diff --git a/bin/dipper-service/src/admin_rpc_server/handlers/indexing_agreements.rs b/bin/dipper-service/src/admin_rpc_server/handlers/indexing_agreements.rs index 689b7415..827537ff 100644 --- a/bin/dipper-service/src/admin_rpc_server/handlers/indexing_agreements.rs +++ b/bin/dipper-service/src/admin_rpc_server/handlers/indexing_agreements.rs @@ -135,5 +135,6 @@ fn into_indexing_agreement_status( IndexingAgreementRecordStatus::AbandonedByIndexer => { IndexingAgreementStatus::AbandonedByIndexer } + IndexingAgreementRecordStatus::Cancelling => IndexingAgreementStatus::Cancelling, } } diff --git a/bin/dipper-service/src/admin_rpc_server/handlers/indexing_requests.rs b/bin/dipper-service/src/admin_rpc_server/handlers/indexing_requests.rs index 17666cbd..b846dd8e 100644 --- a/bin/dipper-service/src/admin_rpc_server/handlers/indexing_requests.rs +++ b/bin/dipper-service/src/admin_rpc_server/handlers/indexing_requests.rs @@ -364,14 +364,6 @@ mod tests { Ok(JobId::default()) } - async fn cancel_rejected_agreement_on_chain( - &self, - _agreement_id: IndexingAgreementId, - _priority: crate::worker::queue::JobPriority, - ) -> anyhow::Result { - unimplemented!() - } - async fn submit_offer( &self, _agreement_id: IndexingAgreementId, diff --git a/bin/dipper-service/src/alerts.rs b/bin/dipper-service/src/alerts.rs new file mode 100644 index 00000000..8147ceef --- /dev/null +++ b/bin/dipper-service/src/alerts.rs @@ -0,0 +1,508 @@ +//! Sends the log lines an operator has to act on to Slack. A line is picked out by its `event` +//! tag, so the code that logs it needs no change. Posting happens in the background, at most once +//! per event in each throttle window, and never holds up or fails the code that logged it. + +use std::{ + collections::{HashMap, HashSet}, + sync::{Arc, Mutex, PoisonError}, + time::Duration, +}; + +use tokio::{sync::mpsc, time::Instant}; +use tracing::{ + Event, Level, Subscriber, + field::{Field, Visit}, +}; +use tracing_subscriber::{Layer, layer::Context}; +use url::Url; + +use crate::config::AlertsConfig; + +/// Alerts that can wait to be posted. More are dropped, as a burst that size is mostly held back +/// by the throttle anyway, and counted into the held-back figure for their event. +const QUEUE: usize = 64; + +/// How long a single post to Slack may take. +const POST_TIMEOUT: Duration = Duration::from_secs(10); + +/// How often to check for events whose window has ended with alerts held back, to post a count. +const HELD_BACK_CHECK: Duration = Duration::from_secs(60); + +/// A tagged log line to alert on. +#[derive(Debug, Clone, PartialEq, Eq)] +struct Alert { + event: String, + level: Level, + message: String, + fields: String, +} + +/// Picks the tagged log lines out of dipper's logging and queues them for posting. +pub struct AlertLayer { + events: HashSet, + queue: mpsc::Sender, + dropped: Dropped, +} + +/// Alerts dropped because the queue was full, counted by event. +type Dropped = Arc>>; + +/// The alert layer for dipper's logging, with its poster running in the background, or `None` +/// when no Slack webhook is configured. +pub fn layer(config: &AlertsConfig) -> Option { + let url = config.slack_webhook_url.as_ref()?.as_ref().clone(); + let (queue, alerts) = mpsc::channel(QUEUE); + let dropped = Dropped::default(); + tokio::spawn(post_alerts( + alerts, + url, + config.throttle, + Arc::clone(&dropped), + )); + Some(AlertLayer { + events: config.events.iter().cloned().collect(), + queue, + dropped, + }) +} + +impl Layer for AlertLayer { + fn on_event(&self, event: &Event<'_>, _ctx: Context<'_, S>) { + let mut tag = EventTag::default(); + event.record(&mut tag); + let Some(tag) = tag.0.filter(|tag| self.events.contains(tag)) else { + return; + }; + let mut line = LogLine::default(); + event.record(&mut line); + let alert = Alert { + event: tag, + level: *event.metadata().level(), + message: line.message, + fields: line.fields, + }; + // Waiting here would hold up the code that logged it, so a full queue drops the alert; + // the poster counts it into the next message for its event. + if let Err(full) = self.queue.try_send(alert) { + *self + .dropped + .lock() + .unwrap_or_else(PoisonError::into_inner) + .entry(full.into_inner().event) + .or_default() += 1; + } + } +} + +/// Only a log line's `event` tag, read first so a line no alert is set up for costs no more. +#[derive(Default)] +struct EventTag(Option); + +impl Visit for EventTag { + fn record_str(&mut self, field: &Field, value: &str) { + if field.name() == "event" { + self.0 = Some(value.to_owned()); + } + } + + fn record_debug(&mut self, field: &Field, value: &dyn std::fmt::Debug) { + if field.name() == "event" { + self.0 = Some(format!("{value:?}")); + } + } +} + +/// A log line's message and its fields other than the `event` tag. +#[derive(Default)] +struct LogLine { + message: String, + fields: String, +} + +impl LogLine { + fn record(&mut self, field: &Field, value: String) { + match field.name() { + "event" => {} + "message" => self.message = value, + name => { + if !self.fields.is_empty() { + self.fields.push(' '); + } + self.fields.push_str(&format!("{name}={value}")); + } + } + } +} + +impl Visit for LogLine { + fn record_str(&mut self, field: &Field, value: &str) { + self.record(field, value.to_owned()); + } + + fn record_debug(&mut self, field: &Field, value: &dyn std::fmt::Debug) { + self.record(field, format!("{value:?}")); + } +} + +/// Post each queued alert to Slack, subject to the throttle, until the queue closes. +async fn post_alerts( + mut alerts: mpsc::Receiver, + url: Url, + window: Duration, + dropped: Dropped, +) { + let http = match reqwest::Client::builder().timeout(POST_TIMEOUT).build() { + Ok(http) => http, + Err(err) => { + tracing::error!(error = %err, "Failed to build the HTTP client for Slack alerts; none will be sent"); + return; + } + }; + let mut throttle = Throttle::new(window); + // A window shorter than the usual check is checked as often as it ends. + let mut check = tokio::time::interval(HELD_BACK_CHECK.min(window).max(Duration::from_secs(1))); + loop { + let mut texts = Vec::new(); + tokio::select! { + alert = alerts.recv() => { + let Some(alert) = alert else { + return; + }; + if let Some(held_back) = throttle.admit(&alert.event, Instant::now()) { + texts.push(alert_text(&alert, held_back)); + } + } + _ = check.tick() => { + for (event, held_back) in throttle.due(Instant::now()) { + texts.push(held_back_text(&event, held_back, window)); + } + } + } + let lost = std::mem::take(&mut *dropped.lock().unwrap_or_else(PoisonError::into_inner)); + for (event, count) in lost { + throttle.hold_back(&event, count, Instant::now()); + } + for text in texts { + post(&http, &url, &text).await; + } + } +} + +/// At most 1 message per event in each window; alerts in between are counted and reported in +/// the next message for that event. +struct Throttle { + window: Duration, + events: HashMap, +} + +struct Window { + started: Instant, + held_back: u64, +} + +impl Throttle { + fn new(window: Duration) -> Self { + Self { + window, + events: HashMap::new(), + } + } + + /// Whether to post this alert now and, if so, how many were held back before it. + fn admit(&mut self, event: &str, now: Instant) -> Option { + if let Some(window) = self.events.get_mut(event) + && now.duration_since(window.started) < self.window + { + window.held_back += 1; + return None; + } + let held_back = self + .events + .insert( + event.to_owned(), + Window { + started: now, + held_back: 0, + }, + ) + .map_or(0, |window| window.held_back); + Some(held_back) + } + + /// Count alerts that never reached the throttle into its held-back figure for their event; + /// one with no window yet is reported at the next check. + fn hold_back(&mut self, event: &str, count: u64, now: Instant) { + let window = self.window; + self.events + .entry(event.to_owned()) + .or_insert_with(|| Window { + started: now.checked_sub(window).unwrap_or(now), + held_back: 0, + }) + .held_back += count; + } + + /// Events whose window has ended with alerts held back, each with how many; the message + /// reporting them starts that event's next window. + fn due(&mut self, now: Instant) -> Vec<(String, u64)> { + let mut due = Vec::new(); + for (event, window) in &mut self.events { + if window.held_back > 0 && now.duration_since(window.started) >= self.window { + due.push((event.clone(), window.held_back)); + *window = Window { + started: now, + held_back: 0, + }; + } + } + due + } +} + +fn alert_text(alert: &Alert, held_back: u64) -> String { + let mut text = format!( + "dipper {} `{}`: {}", + alert.level, + slack_escape(&alert.event), + slack_escape(&alert.message) + ); + if !alert.fields.is_empty() { + text.push_str(&format!("\n{}", slack_escape(&alert.fields))); + } + if held_back > 0 { + text.push_str(&format!("\n{held_back} more since the last alert")); + } + text +} + +/// Escape the 3 characters Slack reads as markup, so text like an HTML error page shows as written +/// instead of turning into links or mentions. +fn slack_escape(text: &str) -> String { + text.replace('&', "&") + .replace('<', "<") + .replace('>', ">") +} + +fn held_back_text(event: &str, held_back: u64, window: Duration) -> String { + format!( + "dipper `{event}`: {held_back} more in the last {}", + describe(window) + ) +} + +/// A throttle window in words: whole minutes where it divides into them, seconds otherwise. +fn describe(window: Duration) -> String { + let seconds = window.as_secs(); + match (seconds / 60, seconds % 60) { + (1, 0) => "minute".to_owned(), + (minutes, 0) if minutes > 0 => format!("{minutes} minutes"), + _ if seconds == 1 => "second".to_owned(), + _ => format!("{seconds} seconds"), + } +} + +/// Post a message to the Slack webhook, logging a failure without the webhook's URL, which is a +/// secret. +async fn post(http: &reqwest::Client, url: &Url, text: &str) { + let sent = http + .post(url.clone()) + .json(&serde_json::json!({ "text": text })) + .send() + .await + .and_then(reqwest::Response::error_for_status); + if let Err(err) = sent { + tracing::warn!(error = %err.without_url(), "Failed to post an alert to Slack"); + } +} + +#[cfg(test)] +mod tests { + use tracing_subscriber::layer::SubscriberExt; + use wiremock::{ + Mock, MockServer, ResponseTemplate, + matchers::{body_json, method}, + }; + + use super::*; + + fn test_layer(events: &[&str]) -> (AlertLayer, mpsc::Receiver) { + let (queue, alerts) = mpsc::channel(QUEUE); + let layer = AlertLayer { + events: events.iter().map(|event| (*event).to_owned()).collect(), + queue, + dropped: Dropped::default(), + }; + (layer, alerts) + } + + #[test] + fn picks_out_only_the_listed_events_whatever_their_level() { + let (layer, mut alerts) = test_layer(&["agreement_cancel_stuck", "nonce_gap_fill_failed"]); + let logging = tracing::Dispatch::new(tracing_subscriber::registry().with(layer)); + + tracing::dispatcher::with_default(&logging, || { + tracing::error!( + event = "agreement_cancel_stuck", + agreement_id = %"0xab", + attempts = 10u32, + "Cancelling an agreement keeps failing" + ); + tracing::warn!( + event = "nonce_gap_fill_failed", + nonce = 7u64, + "Gap fill failed" + ); + }); + tracing::dispatcher::with_default(&logging, || { + tracing::error!(event = "something_else", "Not listed"); + tracing::error!("Not tagged"); + }); + + assert_eq!( + alerts.try_recv().expect("first alert"), + Alert { + event: "agreement_cancel_stuck".to_owned(), + level: Level::ERROR, + message: "Cancelling an agreement keeps failing".to_owned(), + fields: "agreement_id=0xab attempts=10".to_owned(), + } + ); + assert_eq!( + alerts.try_recv().expect("second alert").event, + "nonce_gap_fill_failed" + ); + assert!(alerts.try_recv().is_err(), "nothing else"); + } + + /// Every warning in dipper passes through this layer, so 1 it won't post shouldn't cost + /// formatting its fields. + #[test] + fn leaves_the_fields_of_lines_it_wont_post_unformatted() { + struct Counted<'a>(&'a std::sync::atomic::AtomicUsize); + impl std::fmt::Debug for Counted<'_> { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.0.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + f.write_str("counted") + } + } + let formatted = std::sync::atomic::AtomicUsize::new(0); + let (layer, _alerts) = test_layer(&["agreement_cancel_stuck"]); + let subscriber = tracing_subscriber::registry().with(layer); + + tracing::subscriber::with_default(subscriber, || { + tracing::warn!(event = "something_else", value = ?Counted(&formatted), "Not listed"); + tracing::warn!(value = ?Counted(&formatted), "Not tagged"); + }); + + assert_eq!(formatted.load(std::sync::atomic::Ordering::Relaxed), 0); + } + + #[test] + fn counts_alerts_dropped_when_the_queue_is_full() { + let (queue, _alerts) = mpsc::channel(1); + let layer = AlertLayer { + events: HashSet::from(["rpc_blocks_refused".to_owned()]), + queue, + dropped: Dropped::default(), + }; + let dropped = Arc::clone(&layer.dropped); + let subscriber = tracing_subscriber::registry().with(layer); + + tracing::subscriber::with_default(subscriber, || { + for _ in 0..3 { + tracing::error!(event = "rpc_blocks_refused", "Refused"); + } + }); + + assert_eq!( + *dropped.lock().unwrap(), + HashMap::from([("rpc_blocks_refused".to_owned(), 2)]) + ); + } + + #[test] + fn sends_at_most_1_message_per_event_in_each_window() { + let window = Duration::from_secs(900); + let mut throttle = Throttle::new(window); + let start = Instant::now(); + + assert_eq!(throttle.admit("a", start), Some(0)); + assert_eq!(throttle.admit("a", start + Duration::from_secs(60)), None); + assert_eq!(throttle.admit("a", start + Duration::from_secs(120)), None); + assert_eq!(throttle.admit("b", start), Some(0), "each event on its own"); + + assert!(throttle.due(start + Duration::from_secs(600)).is_empty()); + let after = start + window; + assert_eq!(throttle.due(after), vec![("a".to_owned(), 2)]); + assert_eq!( + throttle.admit("a", after + Duration::from_secs(60)), + None, + "the count starts the next window" + ); + assert_eq!(throttle.admit("a", after + window), Some(1)); + } + + #[test] + fn counts_dropped_alerts_into_the_held_back_figure() { + let window = Duration::from_secs(900); + let mut throttle = Throttle::new(window); + let start = Instant::now(); + assert_eq!(throttle.admit("a", start), Some(0)); + + throttle.hold_back("a", 5, start + Duration::from_secs(60)); + throttle.hold_back("b", 2, start + Duration::from_secs(60)); + + let mut due = throttle.due(start + window); + due.sort(); + assert_eq!(due, vec![("a".to_owned(), 5), ("b".to_owned(), 2)]); + } + + #[test] + fn escapes_what_slack_reads_as_markup() { + let alert = Alert { + event: "nonce_gap_fill_failed".to_owned(), + level: Level::WARN, + message: "Gap fill failed".to_owned(), + fields: "error=502 & more".to_owned(), + }; + + assert!(alert_text(&alert, 0).ends_with("error=<html>502 & more</html>")); + } + + #[test] + fn says_what_happened_and_how_many_more() { + let alert = Alert { + event: "agreement_cancel_stuck".to_owned(), + level: Level::ERROR, + message: "Cancelling an agreement keeps failing".to_owned(), + fields: "agreement_id=0xab".to_owned(), + }; + + assert_eq!( + alert_text(&alert, 3), + "dipper ERROR `agreement_cancel_stuck`: Cancelling an agreement keeps failing\n\ + agreement_id=0xab\n3 more since the last alert" + ); + assert_eq!( + held_back_text("rpc_blocks_refused", 5, Duration::from_secs(900)), + "dipper `rpc_blocks_refused`: 5 more in the last 15 minutes" + ); + assert_eq!(describe(Duration::from_secs(60)), "minute"); + assert_eq!(describe(Duration::from_secs(90)), "90 seconds"); + assert_eq!(describe(Duration::from_secs(30)), "30 seconds"); + } + + #[tokio::test] + async fn posts_the_message_as_slack_text() { + let slack = MockServer::start().await; + Mock::given(method("POST")) + .and(body_json(serde_json::json!({ "text": "hello" }))) + .respond_with(ResponseTemplate::new(200)) + .expect(1) + .mount(&slack) + .await; + let url: Url = slack.uri().parse().expect("URL"); + + post(&reqwest::Client::new(), &url, "hello").await; + } +} diff --git a/bin/dipper-service/src/cancel_dispatch.rs b/bin/dipper-service/src/cancel_dispatch.rs index 08032e4a..cedd4326 100644 --- a/bin/dipper-service/src/cancel_dispatch.rs +++ b/bin/dipper-service/src/cancel_dispatch.rs @@ -1,17 +1,21 @@ -//! On-chain cancel dispatch. Every cancel goes through -//! [`cancel_agreement_on_chain`] so the manager-routed path lives in one place. +//! On-chain cancel dispatch. Every cancel starts with [`start_cancel`] and goes out through +//! `cancel_agreement_on_chain`, so the manager-routed path lives in one place. +use dipper_core::time::now_secs; use thegraph_core::alloy::primitives::B256; use crate::{ - chain_client::{ChainClient, ChainClientError}, + chain_client::{AgreementOnChain, ChainClient, ChainClientError}, config::IndexingAgreementConfig, - registry::IndexingAgreement, + registry::{ + AgreementRegistry, IndexingAgreement, IndexingAgreementStatus, Result as RegistryResult, + }, }; /// Pass both ACTIVE and PENDING; local status lags the chain, so let the -/// collector no-op the absent scope. Never SCOPE_SIGNED (=4): acceptance is -/// offer-based and dipper never retracts a pending offer, so it isn't needed. +/// collector no-op the absent scope. PENDING revokes an offer not yet accepted. +/// Never SCOPE_SIGNED (=4): acceptance is offer-based, so revoking the stored +/// offer is enough. const SCOPE_ACTIVE: u16 = 1; const SCOPE_PENDING: u16 = 2; const SCOPE_BOTH: u16 = SCOPE_ACTIVE | SCOPE_PENDING; @@ -19,47 +23,324 @@ const SCOPE_BOTH: u16 = SCOPE_ACTIVE | SCOPE_PENDING; /// Cancel an agreement on-chain through the RecurringAgreementManager. Passes /// both scope bits so the collector cancels whichever scope the agreement is in, /// and treats a missing or short stored hash as `MissingTermsVersionHash`. -pub async fn cancel_agreement_on_chain( +async fn cancel_agreement_on_chain( chain_client: &T, agreement: &IndexingAgreement, config: &IndexingAgreementConfig, -) -> Result, ChainClientError> { - let version_hash = agreement - .terms_version_hash - .as_deref() - .filter(|h| h.len() == 32) - .map(B256::from_slice) - .ok_or_else(|| ChainClientError::MissingTermsVersionHash { +) -> LiveCancel { + let Some(version_hash) = cancel_hash(agreement) else { + return LiveCancel::CancelFailed(ChainClientError::MissingTermsVersionHash { agreement_id: agreement.id.to_string(), - })?; - // Hazard: the manager's cancel mines successfully even when it does nothing - // (stale/wrong hash, unknown id, already-terminal). So after a submitted - // cancel we re-read on-chain and surface CancelNotConfirmed if still active. - let outcome = chain_client + }); + }; + let tx_hash = match chain_client .cancel_via_manager( config.recurring_collector(), agreement.id.as_bytes(), version_hash, SCOPE_BOTH, ) - .await?; - - // cancel_via_manager only returns Ok(Some) (its tx always submits); - // Ok(None) is reserved. Verify only when a cancel actually mined. - if outcome.is_some() - && chain_client - .agreement_still_active(agreement.id.as_bytes()) - .await? + .await { - return Err(ChainClientError::CancelNotConfirmed { - agreement_id: agreement.id.to_string(), - }); + Ok(tx_hash) => tx_hash, + Err(err) => return LiveCancel::CancelFailed(err), + }; + // The manager's cancel mines even when it does nothing (a stale hash, an unknown id, an + // agreement already ended), and the indexer may have ended it first, so only a read + // afterwards says whether this cancel is what ended it. + match chain_client + .agreement_on_chain(agreement.id.as_bytes()) + .await + { + Ok(AgreementOnChain::NotLive) => LiveCancel::Ended(tx_hash), + Ok(AgreementOnChain::EndedByIndexer) => LiveCancel::NotLive { by_indexer: true }, + Ok(AgreementOnChain::Live) => { + LiveCancel::CancelFailed(ChainClientError::CancelNotConfirmed { + agreement_id: agreement.id.to_string(), + }) + } + Err(err) => LiveCancel::Unconfirmed { tx_hash, err }, + } +} + +/// The stored terms hash an on-chain cancel needs, or `None` when the agreement has no 32-byte +/// one, so no cancel can ever be sent for it. +pub fn cancel_hash(agreement: &IndexingAgreement) -> Option { + agreement + .terms_version_hash + .as_deref() + .filter(|h| h.len() == 32) + .map(B256::from_slice) +} + +/// Why dipper is ending an agreement, which decides the status it ends in once the chain +/// confirms dipper's cancel. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum CancelReason { + /// Dipper no longer wants it: it ends `CanceledByRequester`. + NotWanted, + /// Its indexer stopped serving it: it ends `AbandonedByIndexer`. + Abandoned, +} + +impl CancelReason { + fn ended_status(self) -> &'static str { + match self { + Self::NotWanted => "CANCELED_BY_REQUESTER", + Self::Abandoned => "ABANDONED_BY_INDEXER", + } + } +} + +/// What [`start_cancel`] left an agreement as. Unless `Ended`, it stays `Cancelling` for the +/// cancel retry to finish. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum CancelStarted { + /// It was accepted and its cancel landed: now ended, as its [`CancelReason`] says. + Ended, + /// The chain shows nothing live, so nobody is being paid for it. + NotLive, + /// The chain couldn't be read, or its cancel failed or couldn't be confirmed, so it may + /// still be live and paid. + MayBeLive, +} + +/// Start ending an agreement that may be live on-chain. It is marked `Cancelling` first, so an +/// offer for it still in flight withdraws itself on landing, then cancelled only if the chain +/// shows it live. Fails, sending nothing, when the mark can't be written. +pub async fn start_cancel( + registry: &R, + chain_client: &T, + agreement: &IndexingAgreement, + reason: CancelReason, + config: &IndexingAgreementConfig, +) -> RegistryResult +where + R: AgreementRegistry + Sync, + T: ChainClient, +{ + match reason { + CancelReason::NotWanted => { + registry + .mark_indexing_agreement_as_cancelling(&agreement.id) + .await? + } + CancelReason::Abandoned => { + registry + .mark_indexing_agreement_as_abandoning(&agreement.id) + .await? + } + } + let tx_hash = match cancel_if_live(chain_client, agreement, config).await { + LiveCancel::Ended(tx_hash) => tx_hash, + LiveCancel::NotLive { .. } => return Ok(CancelStarted::NotLive), + LiveCancel::ReadFailed(err) | LiveCancel::CancelFailed(err) => { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "On-chain cancel failed; the cancel retry sends it again" + ); + return Ok(CancelStarted::MayBeLive); + } + LiveCancel::Unconfirmed { tx_hash, err } => { + log_unconfirmed(agreement, tx_hash, &err); + return Ok(CancelStarted::MayBeLive); + } + }; + tracing::info!( + agreement_id = %agreement.id, + tx_hash = ?tx_hash, + "Submitted on-chain cancellation" + ); + // An offer never accepted could still land and be accepted until its deadline. + if agreement.status != IndexingAgreementStatus::AcceptedOnChain { + return Ok(CancelStarted::NotLive); + } + Ok( + if confirm_cancelled(registry, agreement, reason, tx_hash, config).await { + CancelStarted::Ended + } else { + CancelStarted::NotLive + }, + ) +} + +/// Mark an agreement the chain shows dipper ended as ended, recording the cancel when its +/// transaction is known, so the `terminated` sweep announces it. False, logged, when the mark +/// fails; it stays `Cancelling` for the cancel retry. One the chain listener already marked +/// ended counts as ended. +pub async fn confirm_cancelled( + registry: &R, + agreement: &IndexingAgreement, + reason: CancelReason, + tx_hash: Option, + config: &IndexingAgreementConfig, +) -> bool { + // First, so the sweep, which announces the end once the mark lands, finds the transaction. + if tx_hash.is_some() { + record_cancel(registry, agreement, tx_hash, config).await; + } + match registry + .mark_indexing_agreement_as_canceled_by_requester(&agreement.id) + .await + { + Ok(()) => {} + Err(crate::registry::Error::NoRecordsUpdated) => { + tracing::debug!( + agreement_id = %agreement.id, + "Agreement already marked ended, as the chain listener can do first" + ); + return true; + } + Err(err) => { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to mark an ended agreement cancelled; the cancel retry tries again" + ); + return false; + } + } + tracing::info!( + agreement_id = %agreement.id, + indexing_request_id = %agreement.indexing_request_id, + old_status = "CANCELLING", + new_status = reason.ended_status(), + reason = "cancel_confirmed_on_chain", + "agreement state transition" + ); + true +} + +/// Record dipper's own cancel of an accepted agreement, so the `terminated` sweep +/// announces it. +async fn record_cancel( + registry: &R, + agreement: &IndexingAgreement, + tx_hash: Option, + config: &IndexingAgreementConfig, +) { + let manager = config.recurring_agreement_manager().to_string(); + let tx = tx_hash.map(|hash| hash.to_string()); + if let Err(err) = registry + .record_cancel_audit(&agreement.id, now_secs(), &manager, tx.as_deref()) + .await + { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "failed to record cancel audit; terminated event may emit with fallback fields" + ); + } +} + +/// Move an agreement dipper had already rejected or cancelled back into `Cancelling` when the +/// chain shows it live after all, so the cancel retry ends it. The chain is read first, so a +/// subgraph report from before dipper's cancel landed reopens nothing; an unreadable chain +/// reopens it anyway, as the retry reads again before sending. True if it was reopened. +pub async fn reopen_if_live( + registry: &R, + chain_client: &T, + agreement: &IndexingAgreement, +) -> RegistryResult +where + R: AgreementRegistry + Sync, + T: ChainClient, +{ + let seen_live = match chain_client + .agreement_on_chain(agreement.id.as_bytes()) + .await + { + Ok(AgreementOnChain::Live) => true, + Ok(_) => return Ok(false), + Err(err) => { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to read an ended agreement reported live; the cancel retry checks it" + ); + false + } + }; + match registry + .reopen_indexing_agreement_cancel(&agreement.id, seen_live) + .await + { + Ok(()) => {} + Err(crate::registry::Error::NoRecordsUpdated) => return Ok(false), + Err(err) => return Err(err), + } + tracing::warn!( + agreement_id = %agreement.id, + indexer_id = %agreement.indexer.id, + indexing_request_id = %agreement.indexing_request_id, + old_status = %agreement.status, + new_status = "CANCELLING", + reason = "live_on_chain_after_end", + "agreement state transition" + ); + Ok(true) +} + +/// Log a cancel that mined but could not be read back, naming its transaction so it isn't lost; +/// the cancel retry reads the agreement again, and the chain listener records the end. +pub fn log_unconfirmed( + agreement: &IndexingAgreement, + tx_hash: Option, + err: &ChainClientError, +) { + tracing::warn!( + agreement_id = %agreement.id, + tx_hash = ?tx_hash, + error = %err, + "On-chain cancel mined, but whether it ended the agreement couldn't be read; will check again" + ); +} + +/// What [`cancel_if_live`] found and did. +#[derive(Debug)] +pub enum LiveCancel { + /// The chain showed nothing live, so no cancel was sent or the one sent did nothing; + /// `by_indexer` when the indexer ended it. + NotLive { by_indexer: bool }, + /// A cancel went out and the chain confirmed the agreement ended. + Ended(Option), + /// The chain could not be read, so nothing was sent. + ReadFailed(ChainClientError), + /// A cancel mined, but the read after it failed, so whether it ended the agreement, and + /// who did, is not known yet. + Unconfirmed { + tx_hash: Option, + err: ChainClientError, + }, + /// The cancel failed or did not end the agreement. + CancelFailed(ChainClientError), +} + +/// Cancel an agreement on-chain only if the chain shows it live: a pending offer, or +/// accepted and not yet ended. Reading first saves a wasted transaction, since a cancel +/// of an agreement that already ended still mines. +pub async fn cancel_if_live( + chain_client: &T, + agreement: &IndexingAgreement, + config: &IndexingAgreementConfig, +) -> LiveCancel { + match chain_client + .agreement_on_chain(agreement.id.as_bytes()) + .await + { + Err(err) => LiveCancel::ReadFailed(err), + Ok(AgreementOnChain::Live) => { + cancel_agreement_on_chain(chain_client, agreement, config).await + } + Ok(ended) => LiveCancel::NotLive { + by_indexer: ended == AgreementOnChain::EndedByIndexer, + }, } - Ok(outcome) } #[cfg(test)] -mod tests { +pub(crate) mod tests { use std::sync::Mutex; use async_trait::async_trait; @@ -72,9 +353,9 @@ mod tests { use time::OffsetDateTime; use url::Url; - use super::{SCOPE_BOTH, cancel_agreement_on_chain}; + use super::{LiveCancel, SCOPE_BOTH, cancel_agreement_on_chain}; use crate::{ - chain_client::{ChainClient, ChainClientError}, + chain_client::{AgreementOnChain, ChainClient, ChainClientError}, config::IndexingAgreementConfig, registry::{ IndexingAgreement, IndexingAgreementStatus, IndexingAgreementTerms, @@ -92,6 +373,8 @@ mod tests { struct RecordingChainClient { manager_cancels: Mutex>, still_active_after_cancel: bool, + ended_by_indexer: bool, + read_back_fails: bool, active_reads: Mutex, } @@ -142,42 +425,33 @@ mod tests { unimplemented!() } - async fn agreement_still_active( + async fn agreement_on_chain( &self, _agreement_id: &[u8; 16], - ) -> Result { + ) -> Result { *self.active_reads.lock().unwrap() += 1; - Ok(self.still_active_after_cancel) + if self.read_back_fails { + return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); + } + if self.ended_by_indexer { + return Ok(AgreementOnChain::EndedByIndexer); + } + Ok(AgreementOnChain::live_if(self.still_active_after_cancel)) } } fn manager_conf(collector: Address) -> IndexingAgreementConfig { IndexingAgreementConfig { - data_service: Address::ZERO, recurring_collector: collector, recurring_agreement_manager: Address::repeat_byte(0x33), - max_agreement_grt_per_30_days: 0.0, - max_seconds_per_collection: 0, - min_seconds_per_collection: 0, - duration_seconds: 0, - deadline_seconds: 0, - max_grt_per_30_days: std::collections::BTreeMap::new(), - max_grt_per_billion_entities_per_30_days: 0.0, - declined_indexer_lookback_days: 0, - price_rejection_lookback_days: 0, - transient_rejection_lookback_minutes: 0, - uncertain_rejection_lookback_days: 0, - unresponsive_indexer_lookback_days: 0, - mass_unresponsive_trip_fraction: 0.5, - mass_unresponsive_reset_fraction: 0.25, - dips_accepting_snapshot_max_age_hours: 48, - dips_accepting_cache_ttl_seconds: 300, - max_in_flight_offers_per_indexer: None, - max_in_flight_offers_total: None, + ..IndexingAgreementConfig::for_tests() } } - fn agreement(status: IndexingAgreementStatus, hash: Option>) -> IndexingAgreement { + pub(crate) fn agreement( + status: IndexingAgreementStatus, + hash: Option>, + ) -> IndexingAgreement { let deployment_id: DeploymentId = "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" .parse() .unwrap(); @@ -231,9 +505,7 @@ mod tests { Some(vec![7u8; 32]), ); - cancel_agreement_on_chain(&client, &ag, &manager_conf(collector)) - .await - .expect("cancel dispatch"); + cancel_agreement_on_chain(&client, &ag, &manager_conf(collector)).await; let calls = client.manager_cancels.lock().unwrap(); assert_eq!(calls.len(), 1); @@ -253,9 +525,7 @@ mod tests { let client = RecordingChainClient::default(); let ag = agreement(IndexingAgreementStatus::Rejected, Some(vec![9u8; 32])); - cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)) - .await - .expect("cancel dispatch"); + cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)).await; let calls = client.manager_cancels.lock().unwrap(); assert_eq!(calls.len(), 1); @@ -264,19 +534,16 @@ mod tests { #[tokio::test] async fn manager_cancel_missing_hash_is_distinct_error_and_sends_nothing() { - // eh-1: a missing hash must be the distinct MissingTermsVersionHash, not - // a ConfigError the liveness checker reads as "chain client disabled" - // and would silently abandon while the agreement stays live on-chain. + // A missing hash must be the distinct MissingTermsVersionHash: no cancel can be sent + // without it, so the cancel retry spends every attempt at once and raises the alert. let client = RecordingChainClient::default(); let ag = agreement(IndexingAgreementStatus::AcceptedOnChain, None); - let err = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)) - .await - .unwrap_err(); + let out = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)).await; assert!(matches!( - err, - ChainClientError::MissingTermsVersionHash { .. } + out, + LiveCancel::CancelFailed(ChainClientError::MissingTermsVersionHash { .. }) )); assert!(client.manager_cancels.lock().unwrap().is_empty()); } @@ -290,13 +557,11 @@ mod tests { Some(vec![1u8; 16]), ); - let err = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)) - .await - .unwrap_err(); + let out = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)).await; assert!(matches!( - err, - ChainClientError::MissingTermsVersionHash { .. } + out, + LiveCancel::CancelFailed(ChainClientError::MissingTermsVersionHash { .. }) )); assert!(client.manager_cancels.lock().unwrap().is_empty()); } @@ -315,11 +580,12 @@ mod tests { Some(vec![7u8; 32]), ); - let err = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)) - .await - .unwrap_err(); + let out = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)).await; - assert!(matches!(err, ChainClientError::CancelNotConfirmed { .. })); + assert!(matches!( + out, + LiveCancel::CancelFailed(ChainClientError::CancelNotConfirmed { .. }) + )); assert_eq!(client.manager_cancels.lock().unwrap().len(), 1); assert_eq!(*client.active_reads.lock().unwrap(), 1, "verified once"); } @@ -337,12 +603,96 @@ mod tests { Some(vec![7u8; 32]), ); - let out = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)) - .await - .expect("cancel confirmed"); + let out = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)).await; - assert!(out.is_some()); + assert!(matches!(out, LiveCancel::Ended(Some(_)))); assert_eq!(client.manager_cancels.lock().unwrap().len(), 1); assert_eq!(*client.active_reads.lock().unwrap(), 1, "verified once"); } + + #[tokio::test] + async fn a_cancel_the_indexer_beat_to_it_is_not_dipper_s() { + // The indexer's cancel landed first, so dipper's mined as a no-op: crediting the end + // to dipper would announce the wrong canceller. + let client = RecordingChainClient { + ended_by_indexer: true, + ..Default::default() + }; + let ag = agreement( + IndexingAgreementStatus::AcceptedOnChain, + Some(vec![7u8; 32]), + ); + + let out = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)).await; + + assert!(matches!(out, LiveCancel::NotLive { by_indexer: true })); + } + + #[tokio::test] + async fn a_mined_cancel_that_cannot_be_read_back_keeps_its_transaction() { + let client = RecordingChainClient { + read_back_fails: true, + ..Default::default() + }; + let ag = agreement( + IndexingAgreementStatus::AcceptedOnChain, + Some(vec![7u8; 32]), + ); + + let out = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)).await; + + assert!(matches!( + out, + LiveCancel::Unconfirmed { + tx_hash: Some(B256::ZERO), + .. + } + )); + } + + /// Records what each reopen was told about the chain. + #[derive(Default)] + struct ReopenRegistry { + seen_live: Mutex>, + } + + #[async_trait] + impl crate::registry::StubAgreementRegistry for ReopenRegistry { + async fn reopen_indexing_agreement_cancel( + &self, + _id: &IndexingAgreementId, + seen_live: bool, + ) -> crate::registry::Result<()> { + self.seen_live.lock().unwrap().push(seen_live); + Ok(()) + } + } + + #[tokio::test] + async fn a_reopen_clears_the_end_on_record_only_when_the_chain_shows_it_live() { + // An unread chain reopens it all the same, but it may have ended, so its record stays. + let ag = agreement( + IndexingAgreementStatus::CanceledByRequester, + Some(vec![7u8; 32]), + ); + let live = RecordingChainClient { + still_active_after_cancel: true, + ..Default::default() + }; + let unread = RecordingChainClient { + read_back_fails: true, + ..Default::default() + }; + + for (client, seen_live) in [(live, true), (unread, false)] { + let registry = ReopenRegistry::default(); + + let reopened = super::reopen_if_live(®istry, &client, &ag) + .await + .expect("reopen"); + + assert!(reopened); + assert_eq!(*registry.seen_live.lock().unwrap(), vec![seen_live]); + } + } } diff --git a/bin/dipper-service/src/chain_client.rs b/bin/dipper-service/src/chain_client.rs index dd4e4da3..6bc5e3ee 100644 --- a/bin/dipper-service/src/chain_client.rs +++ b/bin/dipper-service/src/chain_client.rs @@ -65,7 +65,12 @@ pub enum ChainClientError { /// tx claimed the nonce with a higher fee. Callers re-sync the nonce and /// resubmit; there is no idempotency guard, so a replay re-sends the call. #[error("tx {tx_hash} did not mine within the receipt-poll window")] - TxDropped { tx_hash: B256 }, + TxDropped { + tx_hash: B256, + /// Whether any receipt check got an answer. When none did, an outage may have hidden a + /// transaction that mined, rather than the chain not mining it. + receipt_checked: bool, + }, /// Tx was mined but reverted on-chain (receipt status = 0). #[error("tx {tx_hash} reverted on-chain")] @@ -80,6 +85,28 @@ pub enum ChainClientError { ContractRevert { selector: [u8; 4], data: Bytes }, } +/// What the chain shows of an agreement's current terms. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum AgreementOnChain { + /// Accepted with no cancellation notice, or an offer still waiting to be accepted. + Live, + /// The indexer ended it. + EndedByIndexer, + /// Neither: never offered, withdrawn, past its offer deadline, or ended by dipper. + NotLive, +} + +impl AgreementOnChain { + pub fn is_live(self) -> bool { + self == Self::Live + } + + #[cfg(test)] + pub fn live_if(live: bool) -> Self { + if live { Self::Live } else { Self::NotLive } + } +} + /// Trait for sending on-chain transactions related to indexing agreements #[async_trait] pub trait ChainClient { @@ -120,13 +147,12 @@ pub trait ChainClient { agreement_id: &[u8; 16], ) -> Result, ChainClientError>; - /// Read whether the agreement is still live on-chain (terms accepted and no - /// cancellation notice given) via the RecurringCollector's + /// Read what the chain shows of the agreement, via the RecurringCollector's /// `getAgreementDetails(id, VERSION_CURRENT)`. - async fn agreement_still_active( + async fn agreement_on_chain( &self, agreement_id: &[u8; 16], - ) -> Result; + ) -> Result; /// Read the latest block's unix timestamp from the chain. Lets agreement /// deadlines be stamped from live chain time when the chain-clock bypass is @@ -175,11 +201,11 @@ impl ChainClient for Arc { (**self).reconcile_agreement(collector, agreement_id).await } - async fn agreement_still_active( + async fn agreement_on_chain( &self, agreement_id: &[u8; 16], - ) -> Result { - (**self).agreement_still_active(agreement_id).await + ) -> Result { + (**self).agreement_on_chain(agreement_id).await } async fn latest_block_timestamp(&self) -> Result { diff --git a/bin/dipper-service/src/chain_client/client.rs b/bin/dipper-service/src/chain_client/client.rs index 67fc7efe..9376c63c 100644 --- a/bin/dipper-service/src/chain_client/client.rs +++ b/bin/dipper-service/src/chain_client/client.rs @@ -8,31 +8,32 @@ use std::{ Arc, atomic::{AtomicU64, Ordering}, }, - time::Duration, + time::{Duration, Instant}, }; use async_trait::async_trait; use dipper_rpc::indexer::indexer_client::sol::RecurringCollectionAgreement; use thegraph_core::alloy::{ - eips::{BlockNumberOrTag, eip2718::Encodable2718}, + eips::{BlockId, BlockNumberOrTag, eip2718::Encodable2718}, network::{EthereumWallet, TransactionBuilder}, primitives::{Address, B256, FixedBytes, U256}, providers::Provider, rpc::types::TransactionRequest, signers::local::PrivateKeySigner, sol_types::{SolCall, SolValue}, - transports::TransportError, + transports::{TransportError, TransportErrorKind}, }; use tokio::sync::Mutex; use super::{ abi::{IRecurringAgreementManager, IRecurringCollector}, gas::{GasEstimator, calculate_max_fee, exceeds_max_gas_price, get_gas_prices}, - rpc_provider::RpcProviderPool, + rpc_provider::{BEHIND_A_SEEN_BLOCK, RpcProviderPool}, }; use crate::{ chain_client::{ - ChainClient, ChainClientError, EscrowAccount, ManagerEscrowReader, TrackedProviders, + AgreementOnChain, ChainClient, ChainClientError, EscrowAccount, ManagerEscrowReader, + TrackedProviders, }, config::ChainClientConfig, worker::service::PROCESS_JOB_TIMEOUT, @@ -53,15 +54,243 @@ const RECEIPT_POLL_TIMEOUT: Duration = Duration::from_secs(15); /// loose enough to avoid hammering the RPC. const RECEIPT_POLL_INTERVAL: Duration = Duration::from_millis(500); +/// What waiting for a transaction's receipt found. +enum ReceiptWait { + /// It mined; whether it succeeded. + Mined(bool), + /// No receipt appeared in time; `checked` when an endpoint did answer that it had none, + /// rather than every check failing. + NotSeen { checked: bool }, +} + /// VERSION_CURRENT index from `IAgreementCollector.sol`: the active (or /// pre-acceptance) terms. `getAgreementDetails(id, 0)` reports their state. const VERSION_CURRENT: u64 = 0; -/// `AgreementDetails.state` flags from `IAgreementCollector.sol` (ACCEPTED=2, -/// NOTICE_GIVEN=4). `getAgreementDetails` keeps ACCEPTED set on a canceled -/// agreement and ORs in NOTICE_GIVEN, so a cancel must clear it, not just lack it. +/// Blocks Arbitrum adds in a week, at its 4 a second. +const WEEK_OF_BLOCKS: u64 = 2_419_200; + +/// The most blocks Arbitrum adds a second. +const BLOCKS_PER_SECOND: u64 = 4; + +/// An hour of blocks. A read every endpoint fails while the nearest is further than this from +/// the newest block dipper has seen raises an ERROR; nearer is lag that clears by itself. +const ALERT_GAP_BLOCKS: u64 = 14_400; + +/// How a read refused by an endpoint whose latest block is too far ahead to be real says so. +const FAR_AHEAD_OF_A_SEEN_BLOCK: &str = "too far ahead of the newest block seen"; + +/// How often, at most, dipper asks every endpoint for its latest block to confirm the newest +/// block it has seen, while that block is unconfirmed. +const CROSS_CHECK_INTERVAL: Duration = Duration::from_secs(60); + +/// An answer takes well under a second, so this only bounds how long a hung endpoint holds a read. +const CROSS_CHECK_DEADLINE: Duration = Duration::from_secs(3); + +/// The newest block dipper has seen, from reads, receipts and latest-block lookups, and when it +/// last moved. It never goes backwards, so a lagging endpoint can't show state from before it. +/// Blocks too far ahead of it are refused only once it is confirmed, by the receipt for one of +/// dipper's transactions or by 2 endpoints agreeing, so an endpoint stuck far behind, or far +/// ahead, that answers first after a restart can't shut out the endpoints that are right. +#[derive(Debug)] +struct SeenBlock { + number: u64, + moved_at: Instant, + confirmed: bool, + cross_checked_at: Option, +} + +impl SeenBlock { + fn new() -> Self { + Self { + number: 0, + moved_at: Instant::now(), + confirmed: false, + cross_checked_at: None, + } + } + + /// The highest block that believably follows the newest one seen: a week of blocks past + /// it, plus what the chain can have added since. + fn believable_limit(&self, now: Instant) -> u64 { + let since = now.saturating_duration_since(self.moved_at).as_secs(); + self.number + .saturating_add(WEEK_OF_BLOCKS) + .saturating_add(since.saturating_mul(BLOCKS_PER_SECOND)) + } + + /// The lowest and highest latest block an endpoint may report to be read from: from the + /// newest seen to its believable limit once confirmed, with no limit until then. + fn bounds(&self, now: Instant) -> (u64, u64) { + let highest = if self.confirmed { + self.believable_limit(now) + } else { + u64::MAX + }; + (self.number, highest) + } + + /// Take `block` as seen, unless it is too far ahead of a confirmed one to be real; false + /// when it is. + fn advance(&mut self, block: u64, now: Instant) -> bool { + if self.confirmed && block > self.believable_limit(now) { + return false; + } + if block > self.number { + self.number = block; + self.moved_at = now; + } + true + } + + /// Take `block` as one the chain is known to have reached. An unconfirmed newest block more + /// than an hour of blocks past it came from a faulty endpoint, so it is replaced. + fn confirm(&mut self, block: u64, now: Instant) { + if self.confirmed { + self.advance(block, now); + return; + } + if block > self.number || self.number > block.saturating_add(ALERT_GAP_BLOCKS) { + self.number = block; + self.moved_at = now; + } + self.confirmed = true; + } + + /// Whether to ask every endpoint for its latest block: only while unconfirmed, and at most + /// once every [`CROSS_CHECK_INTERVAL`]. + fn cross_check_due(&mut self, now: Instant) -> bool { + let checked_lately = self + .cross_checked_at + .is_some_and(|at| now.saturating_duration_since(at) < CROSS_CHECK_INTERVAL); + if self.confirmed || checked_lately { + return false; + } + self.cross_checked_at = Some(now); + true + } +} + +/// The newest of the latest blocks that the most endpoints agree on, at least 2 of them within +/// an hour of blocks of each other, or `None` when no 2 agree. +fn agreed_head(mut heads: Vec) -> Option { + heads.sort_unstable(); + let mut agreed: Option<(usize, u64)> = None; + let mut start = 0; + for (end, &head) in heads.iter().enumerate() { + while head - heads[start] > ALERT_GAP_BLOCKS { + start += 1; + } + let agreeing = end - start + 1; + if agreeing >= 2 && agreed.is_none_or(|(most, _)| agreeing >= most) { + agreed = Some((agreeing, head)); + } + } + agreed.map(|(_, head)| head) +} + +/// Why a read's attempts were refused, to tell a wrong newest block, or endpoints all far out +/// of step, from an outage. +struct Refusals { + /// How far, in blocks, the nearest endpoint refused for its latest block was off. + nearest: AtomicU64, + /// Attempts that failed for any other reason. + other_failures: AtomicU64, +} + +impl Refusals { + fn new() -> Self { + Self { + nearest: AtomicU64::new(u64::MAX), + other_failures: AtomicU64::new(0), + } + } + + /// Refuse an endpoint whose latest block is older than the newest one seen. + fn behind(&self, head: u64, seen: u64) -> Result<(), TransportError> { + if head >= seen { + return Ok(()); + } + self.nearest.fetch_min(seen - head, Ordering::Relaxed); + Err(TransportErrorKind::custom_str(&format!( + "endpoint is at block {head}, {BEHIND_A_SEEN_BLOCK} ({seen})" + ))) + } + + /// Refuse an endpoint whose latest block is too far ahead of the newest one seen to be real. + fn far_ahead(&self, head: u64, seen: u64, highest: u64) -> Result<(), TransportError> { + if head <= highest { + return Ok(()); + } + self.nearest.fetch_min(head - seen, Ordering::Relaxed); + Err(TransportErrorKind::custom_str(&format!( + "endpoint is at block {head}, {FAR_AHEAD_OF_A_SEEN_BLOCK} ({seen})" + ))) + } + + /// Count an attempt that failed for another reason. + fn other(&self, result: Result) -> Result { + if result.is_err() { + self.other_failures.fetch_add(1, Ordering::Relaxed); + } + result + } + + /// Whether every attempt was refused for its block, the nearest over an hour of blocks + /// off: then the endpoints, or the newest block seen, are wrong, not briefly behind. + fn all_far_off(&self) -> bool { + let nearest = self.nearest.load(Ordering::Relaxed); + self.other_failures.load(Ordering::Relaxed) == 0 + && nearest != u64::MAX + && nearest > ALERT_GAP_BLOCKS + } + + /// An ERROR for a read that failed because every endpoint was far out of step. + fn alert(&self, operation: &str, seen: u64) { + if self.all_far_off() { + tracing::error!( + event = "rpc_blocks_refused", + operation, + seen_block = seen, + blocks_off = self.nearest.load(Ordering::Relaxed), + "Every RPC endpoint is over an hour of blocks from the newest block dipper has \ + seen, so chain reads are refused; check the endpoints are in sync and on the \ + right chain, then restart dipper" + ); + } + } +} + +/// `AgreementDetails.state` flags from `IAgreementCollector.sol` (REGISTERED=1, +/// ACCEPTED=2, NOTICE_GIVEN=4, SETTLED=8, BY_PROVIDER=32). `getAgreementDetails` keeps +/// ACCEPTED set on a canceled agreement and ORs in NOTICE_GIVEN, so a cancel must clear it, +/// not just lack it. SETTLED: nothing left to claim. BY_PROVIDER: the indexer cancelled. +const STATE_REGISTERED: u16 = 1; const STATE_ACCEPTED: u16 = 2; const STATE_NOTICE_GIVEN: u16 = 4; +const STATE_SETTLED: u16 = 8; +const STATE_BY_PROVIDER: u16 = 32; + +/// Live iff the terms are accepted and no cancellation notice exists, or an +/// offer is still stored for the indexer to accept. A cancel sets NOTICE_GIVEN +/// while ACCEPTED stays set, so the notice bit tells a live agreement from a +/// cancelled one; a revoked offer reads as an empty state. An offer past its +/// deadline stays stored but is SETTLED, since it can no longer be accepted. +fn still_live(state: u16) -> bool { + let accepted = state & STATE_ACCEPTED != 0; + let pending_offer = state & STATE_REGISTERED != 0 && !accepted && state & STATE_SETTLED == 0; + pending_offer || (accepted && state & STATE_NOTICE_GIVEN == 0) +} + +fn on_chain(state: u16) -> AgreementOnChain { + if still_live(state) { + AgreementOnChain::Live + } else if state & STATE_BY_PROVIDER != 0 { + AgreementOnChain::EndedByIndexer + } else { + AgreementOnChain::NotLive + } +} /// Error patterns that indicate a nonce-related issue. /// @@ -225,6 +454,10 @@ struct AlloyChainClientInner { submit_lock: Mutex<()>, /// How long one submission may hold `submit_lock`; see `derive_submit_deadline`. submit_deadline: Duration, + /// The newest block dipper has seen. An agreement's state is never read from an endpoint + /// behind it, so a lagging endpoint can't undo what dipper saw, nor from one too far + /// ahead of it to be real, so a faulty endpoint can't refuse every read after it. + seen_block: std::sync::Mutex, } impl AlloyChainClient { @@ -275,6 +508,7 @@ impl AlloyChainClient { nonce: AtomicU64::new(NONCE_UNINITIALIZED), submit_lock: Mutex::new(()), submit_deadline, + seen_block: std::sync::Mutex::new(SeenBlock::new()), }), }) } @@ -503,9 +737,8 @@ impl AlloyChainClient { /// already does. Signing happens once, up front, so every endpoint is offered the same /// bytes under one hash and the hash is known before anyone is asked to accept them. async fn send_transaction(&self, tx: &TransactionRequest) -> Result { - // Nothing fills a field in on this path any more, and a request that names no chain is - // signed for chain 1 rather than refused, so check before the signature exists. Not a - // `ConfigError`: the cancel path reads that as the chain client being switched off. + // Nothing fills a field in on this path, and a request that names no chain is signed + // for chain 1 rather than refused, so check before the signature exists. if tx.chain_id() != Some(self.inner.chain_id) { return Err(ChainClientError::SubmitFailed(anyhow::anyhow!( "refusing to sign for chain {:?} while configured for chain {}", @@ -573,9 +806,9 @@ impl AlloyChainClient { .await?; match self.wait_for_receipt(tx_hash, RECEIPT_POLL_TIMEOUT).await? { - Some(true) => Ok(Some(tx_hash)), - Some(false) => Err(ChainClientError::TxReverted { tx_hash }), - None => { + ReceiptWait::Mined(true) => Ok(Some(tx_hash)), + ReceiptWait::Mined(false) => Err(ChainClientError::TxReverted { tx_hash }), + ReceiptWait::NotSeen { checked } => { tracing::warn!( reconciling = subject, tx_hash = %tx_hash, @@ -585,11 +818,108 @@ impl AlloyChainClient { if let Err(err) = self.fill_nonce_gap(dropped_nonce).await { tracing::warn!(nonce = dropped_nonce, error = %err, "Failed to fill mempool nonce gap"); } - Err(ChainClientError::TxDropped { tx_hash }) + Err(ChainClientError::TxDropped { + tx_hash, + receipt_checked: checked, + }) } } } + /// The agreement's `AgreementDetails.state` flags for its current terms. + async fn agreement_state(&self, agreement_id: &[u8; 16]) -> Result { + let call = IRecurringCollector::getAgreementDetailsCall { + agreementId: FixedBytes::<16>::from_slice(agreement_id), + index: thegraph_core::alloy::primitives::U256::from(VERSION_CURRENT), + }; + let collector = self.inner.recurring_collector_address; + Ok(self + .view_at_seen_block(collector, call, "get_agreement_details") + .await? + .state) + } + + /// Run a read-only contract call at the endpoint's latest block, refusing an endpoint + /// whose latest block is older than one dipper has already seen, or too far ahead of it + /// to be real; the pool moves on to the next endpoint instead. + async fn view_at_seen_block( + &self, + to: Address, + call: C, + operation: &'static str, + ) -> Result { + let calldata = call.abi_encode(); + self.cross_check_seen_block().await; + let (seen, highest) = self.block_bounds(); + let refusals = Refusals::new(); + let read = self + .inner + .rpc_pool + .execute(operation, |provider| { + let calldata = calldata.clone(); + let refusals = &refusals; + async move { + let head = refusals.other(provider.get_block_number().await)?; + refusals.behind(head, seen)?; + refusals.far_ahead(head, seen, highest)?; + let tx = TransactionRequest::default().to(to).input(calldata.into()); + let output = + refusals.other(provider.call(tx).block(BlockId::number(head)).await)?; + Ok((head, output)) + } + }) + .await; + let (head, output) = read.inspect_err(|_| refusals.alert(operation, seen))?; + self.note_block(head); + C::abi_decode_returns(&output).map_err(|err| { + ChainClientError::RpcError(anyhow::anyhow!("undecodable {operation} from {to}: {err}")) + }) + } + + fn seen_block(&self) -> std::sync::MutexGuard<'_, SeenBlock> { + self.inner + .seen_block + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + } + + /// The lowest and highest latest block an endpoint may report to be read from. + fn block_bounds(&self) -> (u64, u64) { + self.seen_block().bounds(Instant::now()) + } + + /// Confirm the newest block seen from the endpoints' own agreement, if it isn't yet and + /// there are at least 2 to agree. + async fn cross_check_seen_block(&self) { + let pool = &self.inner.rpc_pool; + if pool.endpoint_count() < 2 || !self.seen_block().cross_check_due(Instant::now()) { + return; + } + match agreed_head(pool.latest_blocks(CROSS_CHECK_DEADLINE).await) { + Some(head) => self.seen_block().confirm(head, Instant::now()), + None => tracing::warn!( + "No 2 RPC endpoints agree on the chain's latest block; reads aren't checked \ + against blocks too far ahead until they do" + ), + } + } + + /// Remember a block dipper has seen, so later reads are never older. One too far ahead to + /// be real is ignored, so it can't refuse every read after it. + fn note_block(&self, block: u64) { + let (taken, seen) = { + let mut seen = self.seen_block(); + (seen.advance(block, Instant::now()), seen.number) + }; + if !taken { + tracing::warn!( + block, + seen_block = seen, + "Ignoring a block too far ahead of the newest block dipper has seen" + ); + } + } + /// Run a read-only contract call and decode its return value. async fn view( &self, @@ -614,15 +944,16 @@ impl AlloyChainClient { }) } - /// Poll `eth_getTransactionReceipt` until the tx has mined or the timeout elapses. - /// `Ok(Some(status))` reports the receipt's success flag; `Ok(None)` says the tx never - /// appeared in time (dropped from the mempool). Transient RPC errors keep polling. + /// Poll `eth_getTransactionReceipt` until the tx has mined or the timeout elapses, saying + /// whether it mined and succeeded, or never appeared in time (dropped from the mempool), and + /// then whether any check got an answer. Transient RPC errors keep polling. async fn wait_for_receipt( &self, tx_hash: B256, timeout: Duration, - ) -> Result, ChainClientError> { + ) -> Result { let deadline = tokio::time::Instant::now() + timeout; + let mut checked = false; loop { let receipt = self .inner @@ -633,8 +964,13 @@ impl AlloyChainClient { .await; match receipt { - Ok(Some(r)) => return Ok(Some(r.status())), - Ok(None) => {} // not mined yet + Ok(Some(r)) => { + if let Some(block) = r.block_number { + self.seen_block().confirm(block, Instant::now()); + } + return Ok(ReceiptWait::Mined(r.status())); + } + Ok(None) => checked = true, // not mined yet Err(e) => { // Transient RPC error: log and keep polling. If it persists, the outer // handler sees the timeout as `Ok(None)` and resubmits, the safe default. @@ -647,7 +983,7 @@ impl AlloyChainClient { } if tokio::time::Instant::now() >= deadline { - return Ok(None); + return Ok(ReceiptWait::NotSeen { checked }); } tokio::time::sleep(RECEIPT_POLL_INTERVAL).await; } @@ -657,16 +993,31 @@ impl AlloyChainClient { #[async_trait] impl ChainClient for AlloyChainClient { async fn latest_block_timestamp(&self) -> Result { - let block = self + self.cross_check_seen_block().await; + let (seen, highest) = self.block_bounds(); + let refusals = Refusals::new(); + let read = self .inner .rpc_pool - .execute("get_latest_block", |provider| async move { - provider.get_block_by_number(BlockNumberOrTag::Latest).await + .execute("get_latest_block", |provider| { + let refusals = &refusals; + async move { + let block = refusals + .other(provider.get_block_by_number(BlockNumberOrTag::Latest).await)?; + // A lagging endpoint's time is only a little early, which brings nothing forward. + if let Some(block) = &block { + refusals.far_ahead(block.header.number, seen, highest)?; + } + Ok(block) + } }) - .await? + .await; + let block = read + .inspect_err(|_| refusals.alert("get_latest_block", seen))? .ok_or_else(|| { ChainClientError::RpcError(anyhow::anyhow!("no latest block returned")) })?; + self.note_block(block.header.number); Ok(block.header.timestamp) } @@ -703,9 +1054,9 @@ impl ChainClient for AlloyChainClient { nonce: dropped_nonce, } = submitted; match self.wait_for_receipt(tx_hash, RECEIPT_POLL_TIMEOUT).await? { - Some(true) => Ok(Some(tx_hash)), - Some(false) => Err(ChainClientError::TxReverted { tx_hash }), - None => { + ReceiptWait::Mined(true) => Ok(Some(tx_hash)), + ReceiptWait::Mined(false) => Err(ChainClientError::TxReverted { tx_hash }), + ReceiptWait::NotSeen { checked } => { tracing::warn!( agreement_id = %format_args!("0x{}", agreement_id.iter().map(|b| format!("{b:02x}")).collect::()), tx_hash = %tx_hash, @@ -715,7 +1066,10 @@ impl ChainClient for AlloyChainClient { if let Err(err) = self.fill_nonce_gap(dropped_nonce).await { tracing::warn!(nonce = dropped_nonce, error = %err, "Failed to fill mempool nonce gap"); } - Err(ChainClientError::TxDropped { tx_hash }) + Err(ChainClientError::TxDropped { + tx_hash, + receipt_checked: checked, + }) } } } @@ -756,9 +1110,9 @@ impl ChainClient for AlloyChainClient { nonce: dropped_nonce, } = submitted; match self.wait_for_receipt(tx_hash, RECEIPT_POLL_TIMEOUT).await? { - Some(true) => Ok(Some(tx_hash)), - Some(false) => Err(ChainClientError::TxReverted { tx_hash }), - None => { + ReceiptWait::Mined(true) => Ok(Some(tx_hash)), + ReceiptWait::Mined(false) => Err(ChainClientError::TxReverted { tx_hash }), + ReceiptWait::NotSeen { checked } => { tracing::warn!( agreement_id = %format_args!("0x{}", agreement_id.iter().map(|b| format!("{b:02x}")).collect::()), tx_hash = %tx_hash, @@ -768,48 +1122,19 @@ impl ChainClient for AlloyChainClient { if let Err(err) = self.fill_nonce_gap(dropped_nonce).await { tracing::warn!(nonce = dropped_nonce, error = %err, "Failed to fill mempool nonce gap"); } - Err(ChainClientError::TxDropped { tx_hash }) + Err(ChainClientError::TxDropped { + tx_hash, + receipt_checked: checked, + }) } } } - async fn agreement_still_active( + async fn agreement_on_chain( &self, agreement_id: &[u8; 16], - ) -> Result { - let calldata = IRecurringCollector::getAgreementDetailsCall { - agreementId: FixedBytes::<16>::from_slice(agreement_id), - index: thegraph_core::alloy::primitives::U256::from(VERSION_CURRENT), - } - .abi_encode(); - - let collector = self.inner.recurring_collector_address; - let output = self - .inner - .rpc_pool - .execute("get_agreement_details", |provider| { - let calldata = calldata.clone(); - async move { - let tx = TransactionRequest::default() - .to(collector) - .input(calldata.into()); - provider.call(tx).await - } - }) - .await?; - - let details = IRecurringCollector::getAgreementDetailsCall::abi_decode_returns(&output) - .map_err(|err| { - ChainClientError::RpcError(anyhow::anyhow!( - "undecodable getAgreementDetails from {collector}: {err}" - )) - })?; - - // Live iff the terms are accepted and no cancellation notice exists. - // A cancel sets NOTICE_GIVEN while ACCEPTED stays set, so checking the - // notice bit is what tells a still-live agreement from a cancelled one. - let state = details.state; - Ok(state & STATE_ACCEPTED != 0 && state & STATE_NOTICE_GIVEN == 0) + ) -> Result { + Ok(on_chain(self.agreement_state(agreement_id).await?)) } async fn reconcile_provider( @@ -1009,6 +1334,48 @@ mod tests { use super::*; + /// A cancel must leave nothing the indexer can still be paid through: neither + /// an accepted agreement without a cancellation notice, nor an offer still + /// stored and waiting to be accepted. The collector reports a revoked offer, + /// or an id it never saw, as an empty state. + #[test] + fn still_live_covers_accepted_agreements_and_pending_offers() { + const BY_PAYER: u16 = 16; + assert!(still_live(STATE_REGISTERED | STATE_ACCEPTED)); + assert!( + still_live(STATE_REGISTERED), + "a pending offer can still be accepted" + ); + assert!( + !still_live(STATE_REGISTERED | STATE_SETTLED), + "an offer past its deadline can't be" + ); + assert!( + still_live(STATE_REGISTERED | STATE_ACCEPTED | STATE_SETTLED), + "an accepted agreement just collected from is still live" + ); + assert!(!still_live( + STATE_REGISTERED | STATE_ACCEPTED | STATE_NOTICE_GIVEN | BY_PAYER | STATE_SETTLED + )); + assert!(!still_live(0), "revoked or never offered"); + } + + #[test] + fn tells_an_end_by_the_indexer_from_any_other() { + const BY_PAYER: u16 = 16; + let ended = STATE_REGISTERED | STATE_ACCEPTED | STATE_NOTICE_GIVEN; + assert_eq!( + on_chain(ended | STATE_BY_PROVIDER), + AgreementOnChain::EndedByIndexer + ); + assert_eq!(on_chain(ended | BY_PAYER), AgreementOnChain::NotLive); + assert_eq!(on_chain(0), AgreementOnChain::NotLive); + assert_eq!( + on_chain(STATE_REGISTERED | STATE_ACCEPTED), + AgreementOnChain::Live + ); + } + /// Answers a send with a fixed transaction hash, echoing the request id so alloy's /// transport accepts the response. Any other call is a mistake in the test rather than /// something to answer with a hash, so say so instead of returning nonsense. @@ -2001,6 +2368,319 @@ mod tests { } } + /// Answers as an endpoint whose latest block is `head`, reporting `state` for any + /// agreement read at that block. + struct AgreementStateResponder { + head: AtomicU64, + /// Blocks the endpoint gains each time it reports its head. + catch_up: u64, + state: u16, + } + + impl Respond for AgreementStateResponder { + fn respond(&self, request: &Request) -> ResponseTemplate { + let body: serde_json::Value = + serde_json::from_slice(&request.body).expect("JSON-RPC request body"); + let result = match body["method"].as_str().unwrap_or_default() { + "eth_blockNumber" => { + let head = self.head.fetch_add(self.catch_up, Ordering::SeqCst); + format!("{head:#x}") + } + "eth_call" => { + let at = body["params"][1].as_str().expect("a block number"); + let at = u64::from_str_radix(at.trim_start_matches("0x"), 16).expect("hex"); + assert!( + at <= self.head.load(Ordering::SeqCst), + "read at a block the endpoint has" + ); + let details = IRecurringCollector::AgreementDetails { + agreementId: FixedBytes::<16>::ZERO, + payer: Address::ZERO, + dataService: Address::ZERO, + serviceProvider: Address::ZERO, + versionHash: B256::ZERO, + state: self.state, + }; + let output = + IRecurringCollector::getAgreementDetailsCall::abi_encode_returns(&details); + format!( + "0x{}", + thegraph_core::alloy::primitives::hex::encode(output) + ) + } + other => panic!("unexpected call {other}"), + }; + ResponseTemplate::new(200).set_body_json(serde_json::json!({ + "jsonrpc": "2.0", + "id": body["id"], + "result": result, + })) + } + } + + async fn server_at_block(head: u64, state: u16) -> MockServer { + server_catching_up(head, 0, state).await + } + + async fn server_catching_up(head: u64, catch_up: u64, state: u16) -> MockServer { + let server = MockServer::start().await; + Mock::given(method("POST")) + .respond_with(AgreementStateResponder { + head: AtomicU64::new(head), + catch_up, + state, + }) + .mount(&server) + .await; + server + } + + #[tokio::test] + async fn waits_for_an_endpoint_a_block_behind_to_catch_up() { + // Hosted endpoints spread calls across nodes, so one a block behind is routine + // rather than a reason to give up on the endpoint. + let endpoint = server_catching_up(94, 1, STATE_REGISTERED | STATE_ACCEPTED).await; + let client = client_over_retrying(vec![endpoint.uri().parse().expect("provider URL")], 1); + client.note_block(95); + + let live = client + .agreement_on_chain(&[0xab; 16]) + .await + .expect("read once the endpoint caught up") + .is_live(); + + assert!(live); + } + + #[tokio::test] + async fn never_reads_an_agreement_from_an_endpoint_behind_a_block_already_seen() { + // The lagging endpoint still shows the offer it hasn't seen accepted and cancelled. + let lagging = server_at_block(90, STATE_REGISTERED).await; + let current = + server_at_block(100, STATE_REGISTERED | STATE_ACCEPTED | STATE_NOTICE_GIVEN).await; + let client = client_over(vec![ + lagging.uri().parse().expect("provider URL"), + current.uri().parse().expect("provider URL"), + ]); + // Not yet confirmed, as after a restart: a lagging endpoint is refused all the same. + client.note_block(95); + + let live = client + .agreement_on_chain(&[0xab; 16]) + .await + .expect("read") + .is_live(); + + assert!(!live, "read from the endpoint that has reached block 95"); + assert_eq!(client.seen_block().number, 100); + } + + #[tokio::test] + async fn fails_a_read_rather_than_answer_from_endpoints_all_behind() { + let lagging = server_at_block(90, STATE_REGISTERED).await; + let client = client_over(vec![lagging.uri().parse().expect("provider URL")]); + client.note_block(95); + + let read = client.agreement_on_chain(&[0xab; 16]).await; + + assert!(read.is_err(), "got {read:?}"); + } + + #[tokio::test] + async fn skips_an_endpoint_too_far_ahead_to_be_real() { + // Otherwise its block would refuse every read from the endpoints that are right. + let faulty = server_at_block(100 + WEEK_OF_BLOCKS + 1_000, STATE_REGISTERED).await; + let current = + server_at_block(150, STATE_REGISTERED | STATE_ACCEPTED | STATE_NOTICE_GIVEN).await; + let client = client_over(vec![ + faulty.uri().parse().expect("provider URL"), + current.uri().parse().expect("provider URL"), + ]); + trust_block(&client, 100); + + let live = client + .agreement_on_chain(&[0xab; 16]) + .await + .expect("read") + .is_live(); + + assert!(!live, "read from the endpoint at block 150"); + assert_eq!(client.seen_block().number, 150); + } + + /// Set the newest block seen as a confirmed one, as a receipt for dipper's transaction does. + fn trust_block(client: &AlloyChainClient, block: u64) { + client.seen_block().confirm(block, Instant::now()); + } + + #[tokio::test] + async fn takes_the_first_block_after_a_restart_however_far_ahead() { + // Dipper may have been down for over a week. + let endpoint = server_at_block(WEEK_OF_BLOCKS * 3, STATE_REGISTERED).await; + let client = client_over(vec![endpoint.uri().parse().expect("provider URL")]); + + client + .agreement_on_chain(&[0xab; 16]) + .await + .expect("read") + .is_live(); + + assert_eq!(client.seen_block().number, WEEK_OF_BLOCKS * 3); + } + + #[test] + fn ignores_a_receipt_block_too_far_ahead_to_be_real() { + let client = client_over(vec!["http://127.0.0.1:1".parse().expect("provider URL")]); + trust_block(&client, 100); + + client.note_block(100 + WEEK_OF_BLOCKS + 1); + assert_eq!(client.seen_block().number, 100); + + client.note_block(100 + WEEK_OF_BLOCKS); + assert_eq!(client.seen_block().number, 100 + WEEK_OF_BLOCKS); + } + + #[test] + fn believes_a_week_of_blocks_ahead_plus_what_the_chain_added_since() { + // A quiet spell with no reads mustn't make the chain's real progress look faulty. + let now = Instant::now(); + let seen = SeenBlock { + number: 100, + moved_at: now.checked_sub(Duration::from_secs(600)).expect("instant"), + confirmed: true, + cross_checked_at: None, + }; + + assert_eq!(seen.bounds(now), (100, 100 + WEEK_OF_BLOCKS + 2_400)); + assert_eq!(SeenBlock::new().bounds(now), (0, u64::MAX)); + } + + #[tokio::test] + async fn says_whether_any_receipt_check_was_answered() { + // A receipt that was never seen only shows the chain didn't mine it when an endpoint + // answered; when every check failed, an outage may be hiding one that did. + let empty = MockServer::start().await; + Mock::given(method("POST")) + .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({ + "jsonrpc": "2.0", + "id": 0, + "result": null, + }))) + .mount(&empty) + .await; + let down = server_answering_500().await; + let wait = Duration::from_millis(100); + + for (server, answered) in [(empty, true), (down, false)] { + let client = client_over(vec![server.uri().parse().expect("provider URL")]); + let found = client + .wait_for_receipt(B256::repeat_byte(0x01), wait) + .await + .expect("waited"); + + assert!( + matches!(found, ReceiptWait::NotSeen { checked } if checked == answered), + "answered: {answered}" + ); + } + } + + #[test] + fn never_goes_back_even_before_it_is_confirmed() { + // So a read just after dipper's cancel mined can't come from an endpoint behind it. + let now = Instant::now(); + let mut seen = SeenBlock::new(); + assert!(seen.advance(100, now)); + assert!(seen.advance(90, now)); + + assert_eq!(seen.bounds(now), (100, u64::MAX)); + } + + #[test] + fn refuses_blocks_far_ahead_only_once_confirmed() { + // An endpoint stuck far behind may answer first after a restart; refusing what is + // far ahead of its block would shut out every endpoint that is right. + let now = Instant::now(); + let mut seen = SeenBlock::new(); + let real = 1_000 + WEEK_OF_BLOCKS * 3; + assert!(seen.advance(1_000, now)); + assert!(seen.advance(real, now), "taken while unconfirmed"); + + seen.confirm(real, now); + + assert!(!seen.advance(real + WEEK_OF_BLOCKS * 2, now)); + assert_eq!(seen.bounds(now).0, real); + } + + #[test] + fn a_confirmed_block_replaces_one_far_ahead_that_was_never_confirmed() { + let now = Instant::now(); + let mut seen = SeenBlock::new(); + let real = 5_000_000; + assert!(seen.advance(real + WEEK_OF_BLOCKS * 10, now)); + + seen.confirm(real, now); + + assert_eq!(seen.bounds(now).0, real); + } + + #[test] + fn agrees_on_the_head_most_endpoints_share() { + let real = 5_000_000; + assert_eq!(agreed_head(vec![1_000, real, real + 3]), Some(real + 3)); + assert_eq!( + agreed_head(vec![real + WEEK_OF_BLOCKS, real + 2, real]), + Some(real + 2) + ); + assert_eq!(agreed_head(vec![1_000, real]), None, "2 that disagree"); + assert_eq!(agreed_head(vec![real]), None, "1 can't agree with itself"); + } + + #[tokio::test] + async fn confirms_the_newest_block_from_endpoints_that_agree() { + // The first endpoint is stuck far behind; the other 2 agree, so it is passed over + // rather than read from. + let stuck = server_at_block(1_000, STATE_REGISTERED).await; + let ended = STATE_REGISTERED | STATE_ACCEPTED | STATE_NOTICE_GIVEN; + let current = server_at_block(5_000_002, ended).await; + let also_current = server_at_block(5_000_002, ended).await; + let client = client_over(vec![ + stuck.uri().parse().expect("provider URL"), + current.uri().parse().expect("provider URL"), + also_current.uri().parse().expect("provider URL"), + ]); + + let live = client + .agreement_on_chain(&[0xab; 16]) + .await + .expect("read") + .is_live(); + + assert!(!live, "read from an endpoint that is right"); + assert!(client.seen_block().confirmed); + assert_eq!(client.seen_block().number, 5_000_002); + } + + #[test] + fn alerts_only_when_every_attempt_is_refused_far_off() { + let far = Refusals::new(); + assert!(far.behind(100, 100 + ALERT_GAP_BLOCKS + 1).is_err()); + assert!(far.all_far_off()); + + let near = Refusals::new(); + assert!(near.behind(97, 100).is_err()); + assert!(!near.all_far_off(), "a few blocks of lag clears by itself"); + + // A primary that is down while a fallback lags is an outage, not a wrong block. + let outage = Refusals::new(); + assert!(outage.behind(100, 100 + ALERT_GAP_BLOCKS + 1).is_err()); + let failed: Result<(), TransportError> = Err(TransportErrorKind::custom_str("down")); + assert!(outage.other(failed).is_err()); + assert!(!outage.all_far_off()); + + assert!(!Refusals::new().all_far_off(), "none refused"); + } + async fn client_over_manager( responder: ManagerViewsResponder, ) -> (AlloyChainClient, MockServer) { diff --git a/bin/dipper-service/src/chain_client/rpc_provider.rs b/bin/dipper-service/src/chain_client/rpc_provider.rs index 98954d46..f05bd1f7 100644 --- a/bin/dipper-service/src/chain_client/rpc_provider.rs +++ b/bin/dipper-service/src/chain_client/rpc_provider.rs @@ -10,7 +10,7 @@ use std::{ use thegraph_core::alloy::{ providers::{ - ProviderBuilder, RootProvider, + Provider, ProviderBuilder, RootProvider, fillers::{BlobGasFiller, ChainIdFiller, FillProvider, GasFiller, JoinFill, NonceFiller}, }, transports::{RpcError, TransportError, TransportErrorKind}, @@ -53,9 +53,22 @@ fn describe_failure(url: &Url, error: &TransportError) -> String { .replace(url.as_str().trim_end_matches('/'), &name) } +/// 1 endpoint's latest block, asked once. A failure is described without the URL, since +/// hosted endpoints carry their API key in it. +async fn latest_block(http: reqwest::Client, url: &Url) -> Result { + let provider = ProviderBuilder::new().connect_reqwest(http, url.clone()); + provider + .get_block_number() + .await + .map_err(|err| describe_failure(url, &err)) +} + /// Error text that indicates a transient failure worth retrying, used only for faults /// that arrive as prose rather than as a status code or JSON-RPC error object. const RETRYABLE_ERROR_PATTERNS: &[&str] = &[ + // A node behind the rest of its provider's fleet, asked for a block it hasn't reached. + "header not found", + "unknown block", "connection refused", "connection reset", "connection closed", @@ -68,6 +81,18 @@ const RETRYABLE_ERROR_PATTERNS: &[&str] = &[ "temporary internal error", ]; +/// How a read refused by an endpoint behind a block dipper has already seen describes it. It +/// gets quick retries, since an endpoint a few blocks behind catches up within a second, then +/// the next endpoint, rather than the backoff for a failing one. +pub(super) const BEHIND_A_SEEN_BLOCK: &str = "behind a block already seen"; + +/// How long a read waits before asking an endpoint behind a block already seen again: 2 blocks. +const LAG_PAUSE: Duration = Duration::from_millis(500); + +/// How many times an endpoint behind a block already seen is asked again: about 4 blocks of +/// lag in all, so the read just after a transaction mines can wait out a node a little behind. +const LAG_RETRIES: u32 = 2; + /// Type alias for the provider with default fillers. pub type HttpProvider = FillProvider< JoinFill< @@ -146,6 +171,39 @@ impl RpcProviderPool { self.worst_case_walk } + /// How many endpoints the pool has. + pub fn endpoint_count(&self) -> usize { + self.providers.len() + } + + /// Each endpoint's latest block, all asked at once with no retries. Endpoints that fail, or + /// don't answer within `deadline`, are left out, so dipper can see whether the rest agree. + pub async fn latest_blocks(&self, deadline: Duration) -> Vec { + let mut asks = tokio::task::JoinSet::new(); + for url in &self.providers { + let (http, url) = (self.http.clone(), url.clone()); + asks.spawn(async move { + let head = tokio::time::timeout(deadline, latest_block(http, &url)) + .await + .unwrap_or_else(|_| Err(format!("no answer within {deadline:?}"))); + (endpoint_name(&url), head) + }); + } + let mut heads = Vec::with_capacity(self.providers.len()); + while let Some(answer) = asks.join_next().await { + match answer { + Ok((_, Ok(head))) => heads.push(head), + Ok((endpoint, Err(reason))) => tracing::debug!( + provider = %endpoint, + error = %reason, + "RPC endpoint didn't give its latest block for a cross-check" + ), + Err(err) => tracing::warn!(error = %err, "Latest-block cross-check task failed"), + } + } + heads + } + /// Rotate to the next provider. /// /// Returns the new provider URL after rotation. @@ -199,41 +257,15 @@ impl RpcProviderPool { let current_url = self.url_at(start + providers_tried).clone(); let endpoint = endpoint_name(¤t_url); - // Reused across this endpoint's attempts, so a retry does not pay for a fresh - // TLS handshake on the path that is already running out of time. - let provider = - ProviderBuilder::new().connect_reqwest(self.http.clone(), current_url.clone()); - - // Retry loop for current provider - let mut endpoint_error: Option = None; - for attempt in 0..=max_retries { - match f(provider.clone()).await { - Ok(result) => return Ok(result), - Err(e) if Self::is_retryable(&e) && attempt < max_retries => { - let delay = Self::backoff_delay(attempt); - tracing::warn!( - operation, - provider = %endpoint, - attempt = attempt + 1, - max_retries, - delay_ms = delay.as_millis(), - error = %describe_failure(¤t_url, &e), - "Retryable RPC error, backing off" - ); - tokio::time::sleep(delay).await; - endpoint_error = Some(e); - } - Err(e) => { - endpoint_error = Some(e); - break; - } - } - } - // Every attempt records why it failed before stopping, so the fallback only - // covers a configuration that somehow allows no attempt at all. - let endpoint_error = endpoint_error - .unwrap_or_else(|| TransportErrorKind::custom_str("no attempt was made")); + let endpoint_error = match self + .ask_endpoint(operation, ¤t_url, max_retries, &f) + .await + { + Ok(result) => return Ok(result), + Err(err) => err, + }; + let lagging = Self::is_behind(&endpoint_error); let reason = describe_failure(¤t_url, &endpoint_error); reasons.push(format!("{endpoint}: {reason}")); providers_tried += 1; @@ -261,16 +293,89 @@ impl RpcProviderPool { // back round to a failing endpoint, costing them the one wasted first ask. self.rotate(); let next_url = self.url_at(start + providers_tried); - tracing::warn!( - operation, - old_provider = %endpoint, - new_provider = %endpoint_name(next_url), - providers_tried, - total_providers = self.providers.len(), - error = %reason, - "Rotating RPC provider after failures" - ); + if lagging { + tracing::debug!( + operation, + old_provider = %endpoint, + new_provider = %endpoint_name(next_url), + error = %reason, + "Rotating RPC provider past one behind a block already seen" + ); + } else { + tracing::warn!( + operation, + old_provider = %endpoint, + new_provider = %endpoint_name(next_url), + providers_tried, + total_providers = self.providers.len(), + error = %reason, + "Rotating RPC provider after failures" + ); + } + } + } + + /// Run a call against one endpoint, retrying a fault worth another go, and return what it + /// answered or why it last failed. + async fn ask_endpoint( + &self, + operation: &str, + url: &Url, + max_retries: u32, + f: &F, + ) -> Result + where + F: Fn(HttpProvider) -> Fut, + Fut: Future>, + { + let endpoint = endpoint_name(url); + // Reused across this endpoint's attempts, so a retry does not pay for a fresh + // TLS handshake on the path that is already running out of time. + let provider = ProviderBuilder::new().connect_reqwest(self.http.clone(), url.clone()); + let mut endpoint_error: Option = None; + let mut lag_retries = 0; + for attempt in 0..=max_retries { + match f(provider.clone()).await { + Ok(result) => return Ok(result), + Err(e) + if Self::is_behind(&e) + && lag_retries < LAG_RETRIES + && attempt < max_retries => + { + tracing::debug!( + operation, + provider = %endpoint, + error = %describe_failure(url, &e), + "RPC endpoint behind a block already seen, asking again" + ); + lag_retries += 1; + tokio::time::sleep(LAG_PAUSE).await; + endpoint_error = Some(e); + } + Err(e) if Self::is_retryable(&e) && attempt < max_retries => { + let delay = Self::backoff_delay(attempt); + tracing::warn!( + operation, + provider = %endpoint, + attempt = attempt + 1, + max_retries, + delay_ms = delay.as_millis(), + error = %describe_failure(url, &e), + "Retryable RPC error, backing off" + ); + tokio::time::sleep(delay).await; + endpoint_error = Some(e); + } + Err(e) => return Err(e), + } } + // Every attempt records why it failed before stopping, so the fallback only + // covers a configuration that somehow allows no attempt at all. + Err(endpoint_error.unwrap_or_else(|| TransportErrorKind::custom_str("no attempt was made"))) + } + + fn is_behind(error: &TransportError) -> bool { + error.to_string().contains(BEHIND_A_SEEN_BLOCK) } /// Whether an error is worth trying again rather than giving up on. Each check can only @@ -501,6 +606,53 @@ mod tests { ); } + /// The latest-block cross-check logs each endpoint's failure, so it must hide the key too. + #[tokio::test] + async fn a_cross_check_failure_hides_the_api_key() { + let keyed: Url = "http://127.0.0.1:1/v2/super-secret-key" + .parse() + .expect("keyed endpoint URL"); + + let reason = latest_block(reqwest::Client::new(), &keyed) + .await + .expect_err("nothing is listening, so the ask fails"); + + assert!( + !reason.contains("super-secret-key"), + "the API key must not appear in the failure: {reason}" + ); + } + + /// Reads wait on the cross-check, so 1 endpoint that never answers must not hold it for the + /// whole request timeout. The healthy endpoint's head still counts. + #[tokio::test] + async fn a_cross_check_stops_waiting_for_a_hung_endpoint() { + let healthy = server_answering_block(0x2a).await; + let hung = MockServer::start().await; + Mock::given(method("POST")) + .respond_with(ResponseTemplate::new(200).set_delay(Duration::from_secs(30))) + .mount(&hung) + .await; + let pool = RpcProviderPool::new( + vec![ + healthy.uri().parse().expect("healthy URL"), + hung.uri().parse().expect("hung URL"), + ], + Duration::from_secs(60), + 0, + ) + .expect("pool"); + + let heads = tokio::time::timeout( + Duration::from_secs(2), + pool.latest_blocks(Duration::from_millis(200)), + ) + .await + .expect("the cross-check should give up on the hung endpoint at its deadline"); + + assert_eq!(heads, vec![0x2a]); + } + /// An endpoint that answers can only describe its own refusal, so it never repeats the /// URL. One that never answers is described by the HTTP client instead, which says which /// URL it was reaching for, and that is where the key sits. Nothing listens on port 1. @@ -633,6 +785,50 @@ mod tests { ); } + /// Hosted endpoints spread calls across nodes, so one a block behind is routine: it is + /// asked again after short pauses, then passed over, never backed off from. + #[tokio::test] + async fn an_endpoint_behind_a_block_already_seen_gets_quick_retries() { + let lagging = server_answering_block(1).await; + let current = server_answering_block(2).await; + let pool = RpcProviderPool::new( + vec![ + lagging.uri().parse().expect("lagging URL"), + current.uri().parse().expect("current URL"), + ], + Duration::from_secs(5), + 3, + ) + .expect("pool"); + let started = std::time::Instant::now(); + + let block = pool + .execute("get_block_number", |provider| async move { + let block = provider.get_block_number().await?; + if block < 2 { + return Err(TransportErrorKind::custom_str(BEHIND_A_SEEN_BLOCK)); + } + Ok(block) + }) + .await + .expect("read from the endpoint that has the block"); + + assert_eq!(block, 2); + assert_eq!( + lagging.received_requests().await.unwrap_or_default().len(), + 3 + ); + assert!(started.elapsed() < Duration::from_secs(2), "no backoff"); + } + + #[test] + fn a_node_that_has_not_reached_a_block_yet_is_retryable() { + let payload = serde_json::from_str(r#"{"code":-32000,"message":"header not found"}"#) + .expect("JSON-RPC error payload"); + let err: TransportError = RpcError::ErrorResp(payload); + assert!(RpcProviderPool::is_retryable(&err)); + } + #[test] fn each_retry_waits_twice_as_long_up_to_a_ceiling() { // 1s, 2s, 4s, 8s, 16s, 32s->30s diff --git a/bin/dipper-service/src/config.rs b/bin/dipper-service/src/config.rs index b480a33e..8bc4619c 100644 --- a/bin/dipper-service/src/config.rs +++ b/bin/dipper-service/src/config.rs @@ -97,6 +97,42 @@ pub struct Config { /// the server off. #[serde(default)] pub health: HealthConfig, + /// Where to send alerts for the log lines an operator has to act on. Without a webhook, + /// none are sent. + #[serde(default)] + pub alerts: AlertsConfig, +} + +/// Alerts for the log lines an operator has to act on, picked out by their `event` tag and +/// posted to Slack, at most once per event in each throttle window. +#[serde_as] +#[derive(Debug, serde::Deserialize)] +#[serde(default, deny_unknown_fields)] +pub struct AlertsConfig { + /// The Slack incoming-webhook URL to post alerts to. Without one, none are sent. + pub slack_webhook_url: Option>, + /// The `event` tags of the log lines to alert on. + pub events: Vec, + /// Least time, in seconds, between 2 messages for the same event; alerts in between are + /// counted into the next one. + #[serde_as(as = "serde_with::DurationSeconds")] + pub throttle: Duration, +} + +impl Default for AlertsConfig { + fn default() -> Self { + Self { + slack_webhook_url: None, + events: [ + "agreement_cancel_stuck", + "rpc_blocks_refused", + "nonce_gap_fill_failed", + ] + .map(str::to_owned) + .to_vec(), + throttle: Duration::from_secs(15 * 60), + } + } } /// Configuration for the HTTP health endpoint used by orchestrator liveness probes. Omitting the @@ -817,8 +853,9 @@ pub struct DipsAgreementConfig { pub max_grt_per_billion_entities_per_30_days: f64, /// Number of days to look back for declined indexers (standard exclusion). Covers - /// CanceledByIndexer, expiries whose offer reached the chain, and structurally - /// persistent rejections (UNSUPPORTED_NETWORK, MANIFEST_TOO_LARGE). Default: 30 days. + /// CanceledByIndexer, AbandonedByIndexer, expiries whose offer reached the chain, and + /// structurally persistent rejections (UNSUPPORTED_NETWORK, MANIFEST_TOO_LARGE). Default: 30 + /// days. #[serde(default = "default_declined_indexer_lookback_days")] pub declined_indexer_lookback_days: i32, @@ -1202,6 +1239,37 @@ pub struct IndexingAgreementConfig { pub max_in_flight_offers_total: Option, } +#[cfg(test)] +impl IndexingAgreementConfig { + /// Addresses and limits all set to 0, with permissive breaker and cache settings, for tests + /// to adjust the fields they care about. + pub fn for_tests() -> Self { + Self { + data_service: Address::ZERO, + recurring_collector: Address::ZERO, + recurring_agreement_manager: Address::ZERO, + max_agreement_grt_per_30_days: 0.0, + max_seconds_per_collection: 0, + min_seconds_per_collection: 0, + duration_seconds: 0, + deadline_seconds: 0, + max_grt_per_30_days: BTreeMap::new(), + max_grt_per_billion_entities_per_30_days: 0.0, + declined_indexer_lookback_days: 0, + price_rejection_lookback_days: 0, + transient_rejection_lookback_minutes: 0, + uncertain_rejection_lookback_days: 0, + unresponsive_indexer_lookback_days: 0, + mass_unresponsive_trip_fraction: 0.5, + mass_unresponsive_reset_fraction: 0.25, + dips_accepting_snapshot_max_age_hours: 48, + dips_accepting_cache_ttl_seconds: 300, + max_in_flight_offers_per_indexer: None, + max_in_flight_offers_total: None, + } + } +} + /// Per-chain pricing for indexing agreements (runtime). #[derive(Debug)] pub struct IndexingAgreementChainPrices { @@ -2050,6 +2118,22 @@ mod tests { ); } + #[test] + fn alerts_config_takes_a_webhook_and_keeps_the_default_events() { + let alerts = serde_json::from_str::( + r#"{"slack_webhook_url": "https://hooks.slack.com/services/T/B/x", "throttle": 60}"#, + ) + .expect("alerts config"); + + assert!(alerts.slack_webhook_url.is_some()); + assert_eq!(alerts.throttle, Duration::from_secs(60)); + assert_eq!(alerts.events, AlertsConfig::default().events); + assert!( + !format!("{alerts:?}").contains("hooks.slack.com"), + "the webhook is a secret, kept out of logs" + ); + } + /// A misspelled key would otherwise be dropped in silence, leaving an operator who meant to /// change the endpoint with the defaults and no sign that the setting never took effect. #[test] diff --git a/bin/dipper-service/src/main.rs b/bin/dipper-service/src/main.rs index 8b22c649..17810755 100644 --- a/bin/dipper-service/src/main.rs +++ b/bin/dipper-service/src/main.rs @@ -8,7 +8,10 @@ use dipper_producer::events::{ use futures_lite::StreamExt; use thegraph_core::alloy::signers::local::PrivateKeySigner; use tokio::task::JoinSet; -use tracing_subscriber::EnvFilter; +use tracing_subscriber::{ + EnvFilter, Layer as _, filter::LevelFilter, layer::SubscriberExt as _, + util::SubscriberInitExt as _, +}; use self::{ config::DEFAULT_MAX_CANDIDATES, registry::RegistryProvider, signing::eip712::Eip712Signer, @@ -17,6 +20,7 @@ use self::{ use crate::config::EventStreamingConfig; mod admin_rpc_server; +mod alerts; mod cancel_dispatch; mod chain_client; mod config; @@ -61,19 +65,24 @@ const STOP_STEP_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(5) reason = "predates this lint; fix when next touched" )] pub async fn main() -> anyhow::Result<()> { - // Set up logging - tracing_subscriber::fmt() - .with_env_filter(EnvFilter::from_default_env()) - .init(); - - // Load the configuration - tracing::debug!("loading configuration"); + // Load the configuration first, since logging needs it to know where to send alerts let conf_path = env::args() .nth(1) .expect("Missing argument for config path") .parse::() .expect("Invalid path"); let conf = config::load_from_file(&conf_path).expect("Failed to load config"); + + // Set up logging. Plain text, with no colour codes, so log stores can search it. Alerts see + // warnings and errors whatever the log level is set to. + tracing_subscriber::registry() + .with( + tracing_subscriber::fmt::layer() + .with_ansi(false) + .with_filter(EnvFilter::from_default_env()), + ) + .with(alerts::layer(&conf.alerts).map(|layer| layer.with_filter(LevelFilter::WARN))) + .init(); tracing::debug!(conf=?conf, "configuration loaded"); // Reject a config the protocol-managed path can't run with before building @@ -98,6 +107,7 @@ pub async fn main() -> anyhow::Result<()> { let chain_listener_agreement_conf = agreement_conf.clone(); let liveness_agreement_conf = agreement_conf.clone(); let escrow_reconciler_agreement_conf = agreement_conf.clone(); + let cancel_retry_agreement_conf = agreement_conf.clone(); // Canonical chain id and RecurringCollector address, read once and shared by the // admin signer, the gRPC proposal signer, and the on-chain chain client so their @@ -522,7 +532,6 @@ pub async fn main() -> anyhow::Result<()> { let ctx = network::service::chain_listener::Ctx { registry: registry.clone(), - worker_queue: worker_handle.queue().clone(), event_source, chain_client: chain_client.clone(), agreement_conf: chain_listener_agreement_conf.clone(), @@ -537,6 +546,15 @@ pub async fn main() -> anyhow::Result<()> { _ => None, }; + //- The cancel retry, always on: it alone finishes the cancels dipper starts + let (cancel_retry_handle, cancel_retry_service) = + network::service::cancel_retry::new(network::service::cancel_retry::Ctx { + registry: registry.clone(), + chain_client: chain_client.clone(), + agreement_conf: cancel_retry_agreement_conf, + worker_queue: worker_handle.queue().clone(), + }); + //- The liveness checker service (optional, enabled by config) // Detects indexers who silently stop indexing active AcceptedOnChain agreements let liveness_checker_handle = match conf.liveness_checker { @@ -675,6 +693,9 @@ pub async fn main() -> anyhow::Result<()> { None }; + let cancel_retry_task_handle = task_tree.spawn(cancel_retry_service); + tracing::debug!(task_id=%cancel_retry_task_handle.id(), "Cancel retry service started"); + // Spawn the escrow reconciler service if enabled let escrow_reconciler_stop_handle = if let Some((handle, service)) = escrow_reconciler_handle { let task_handle = task_tree.spawn(service); @@ -755,6 +776,9 @@ pub async fn main() -> anyhow::Result<()> { all_stopped &= stop_service("Chain listener", handle.stop()).await; } + // Stop the cancel retry before worker (it queues replacements) + all_stopped &= stop_service("Cancel retry", cancel_retry_handle.stop()).await; + // Stop escrow reconciler service before the DB pool closes if let Some(handle) = escrow_reconciler_stop_handle { all_stopped &= stop_service("Escrow reconciler", handle.stop()).await; diff --git a/bin/dipper-service/src/network/service.rs b/bin/dipper-service/src/network/service.rs index 86f9bb3e..3fc9d0f0 100644 --- a/bin/dipper-service/src/network/service.rs +++ b/bin/dipper-service/src/network/service.rs @@ -1,3 +1,4 @@ +pub mod cancel_retry; pub mod chain_events; pub mod chain_listener; pub mod domain_refresh; diff --git a/bin/dipper-service/src/network/service/cancel_retry.rs b/bin/dipper-service/src/network/service/cancel_retry.rs new file mode 100644 index 00000000..3417b30e --- /dev/null +++ b/bin/dipper-service/src/network/service/cancel_retry.rs @@ -0,0 +1,1191 @@ +//! Finishes the cancels dipper starts, marked `Cancelling` before they go out: re-sends each +//! while the chain shows it live, then marks it ended, `AbandonedByIndexer` if its indexer +//! stopped serving it. Runs on its own, as nothing else finishes them. + +use std::{future::Future, sync::Arc, time::Duration}; + +use dipper_core::time::now_secs; +use thegraph_core::alloy::primitives::B256; +use tokio::{sync::mpsc, time::MissedTickBehavior}; + +use crate::{ + cancel_dispatch::{ + CancelReason, LiveCancel, cancel_if_live, confirm_cancelled, log_unconfirmed, + }, + chain_client::{ChainClient, ChainClientError}, + config::IndexingAgreementConfig, + registry::{ + AgreementRegistry, CancelKind, CancellingAgreement, IndexingAgreement, + IndexingRequestRegistry, + }, + worker::service::WorkerQueue, +}; + +/// Failed cancels before dipper alerts an operator and retries the agreement only hourly, so a +/// paused manager recovers once unpaused. Outages don't count (see `failed_attempts`). +pub const MAX_CANCEL_ATTEMPTS: u32 = 10; + +/// Agreements a sweep takes on, those that may be paying an indexer first; the time budget +/// below decides how many it gets through. +const BATCH_SIZE: i64 = 50; + +/// How often agreements still being cancelled get their cancel retried. +const SWEEP_INTERVAL: Duration = Duration::from_secs(300); + +/// Time a sweep may take before leaving the rest to the next one, as each cancel can wait up +/// to 15 s to be mined. +const SWEEP_BUDGET: Duration = Duration::from_secs(30); + +/// Time allowed to read an ended agreement's request, and to queue its replacement. +const DB_TIMEOUT: Duration = Duration::from_secs(30); +const QUEUE_TIMEOUT: Duration = Duration::from_secs(10); + +/// Minutes an agreement stays out of the retry after it is marked, so the cancel sent +/// when it was marked can be mined first instead of being sent again. One moved back to +/// cancelling had none sent, so it doesn't wait. +const SETTLE_MINUTES: i32 = 2; + +/// How long the chain listener gets, from when a check first finds an agreement ended, to +/// record when and in which transaction it ended, before the retry marks it without them. +const LISTENER_GRACE: time::Duration = time::Duration::HOUR; + +/// Handle for stopping the cancel retry. +#[derive(Clone)] +pub struct Handle { + tx_stop: mpsc::Sender<()>, +} + +impl Handle { + /// Stop the cancel retry, cutting short a sweep in progress. + pub async fn stop(&self) { + if self.tx_stop.is_closed() { + return; + } + let _ = self.tx_stop.send(()).await; + self.tx_stop.closed().await; + } +} + +/// What the cancel retry needs. +pub struct Ctx { + pub registry: R, + pub chain_client: T, + pub agreement_conf: Arc, + /// Queues the replacement of an agreement ended because its indexer stopped serving it. + pub worker_queue: W, +} + +/// Create the cancel retry. Returns a handle plus a future to spawn, which sweeps at once and +/// then every [`SWEEP_INTERVAL`]. +pub fn new(ctx: Ctx) -> (Handle, impl Future>) +where + R: AgreementRegistry + IndexingRequestRegistry + Send + Sync, + T: ChainClient + Send + Sync, + W: WorkerQueue + Send + Sync, +{ + let (tx_stop, mut rx_stop) = mpsc::channel(1); + let Ctx { + registry, + chain_client, + agreement_conf, + worker_queue, + } = ctx; + let service = async move { + tracing::info!( + interval_secs = SWEEP_INTERVAL.as_secs(), + "cancel retry service started" + ); + let mut timer = tokio::time::interval(SWEEP_INTERVAL); + timer.set_missed_tick_behavior(MissedTickBehavior::Skip); + loop { + tokio::select! { + _ = rx_stop.recv() => break, + _ = timer.tick() => {}, + } + tokio::select! { + _ = rx_stop.recv() => break, + () = async { + retry_cancelling_agreements(®istry, &chain_client, &agreement_conf).await; + replace_ended_abandoned(®istry, &worker_queue).await; + } => {}, + } + } + tracing::debug!("cancel retry service stopped"); + Ok(()) + }; + (Handle { tx_stop }, service) +} + +/// Queue the replacement of each agreement whose indexer stopped serving it once it has ended, +/// whatever ended it: this retry, the chain listener or the indexer. +async fn replace_ended_abandoned(registry: &R, worker_queue: &W) +where + R: AgreementRegistry + IndexingRequestRegistry + Sync, + W: WorkerQueue + Sync, +{ + let ended = match registry + .get_ended_agreements_awaiting_replacement(BATCH_SIZE) + .await + { + Ok(ended) => ended, + Err(err) => { + tracing::warn!(error = %err, "Failed to list ended agreements awaiting replacement"); + return; + } + }; + for agreement in &ended { + super::liveness_checker::replace_abandoned( + agreement, + registry, + worker_queue, + DB_TIMEOUT, + QUEUE_TIMEOUT, + ) + .await; + } +} + +/// Retry the cancel of agreements still `Cancelling`. The chain's own latest block time +/// decides when an offer that was never accepted no longer can be, so a subgraph that has +/// fallen behind doesn't hold that up. +pub async fn retry_cancelling_agreements( + registry: &R, + chain_client: &T, + config: &IndexingAgreementConfig, +) where + R: AgreementRegistry + Sync, + T: ChainClient, +{ + let cancelling = match registry + .get_cancelling_agreements(BATCH_SIZE, MAX_CANCEL_ATTEMPTS, SETTLE_MINUTES) + .await + { + Ok(cancelling) => cancelling, + Err(err) => { + tracing::warn!(error = %err, "Failed to list agreements still being cancelled"); + return; + } + }; + if cancelling.is_empty() { + return; + } + let Some(chain_now) = chain_time(chain_client).await else { + return; + }; + let started = std::time::Instant::now(); + for (done, row) in cancelling.iter().enumerate() { + if started.elapsed() >= SWEEP_BUDGET { + tracing::info!( + left = cancelling.len() - done, + "Cancel retry ran out of time; the rest wait for the next sweep" + ); + break; + } + retry_cancel(registry, chain_client, config, row, chain_now).await; + } +} + +async fn chain_time(chain_client: &T) -> Option { + match chain_client.latest_block_timestamp().await { + Ok(chain_now) => Some(chain_now), + Err(err) => { + tracing::warn!( + error = %err, + "Failed to read the chain's time; cancels are retried next sweep" + ); + None + } + } +} + +async fn retry_cancel( + registry: &R, + chain_client: &T, + config: &IndexingAgreementConfig, + row: &CancellingAgreement, + chain_now: u64, +) where + R: AgreementRegistry + Sync, + T: ChainClient, +{ + let agreement_id = row.agreement.id; + let (tx_hash, by_indexer, failure) = + match cancel_if_live(chain_client, &row.agreement, config).await { + LiveCancel::ReadFailed(err) => { + tracing::warn!( + %agreement_id, + error = %err, + "Failed to read a cancelling agreement on-chain, will retry" + ); + // Unread, it may still be live, so it can't be confirmed ended. + return note_check(registry, row, None, None).await; + } + LiveCancel::NotLive { by_indexer } => (None, by_indexer, None), + LiveCancel::Ended(tx_hash) => { + tracing::info!( + %agreement_id, + tx_hash = ?tx_hash, + "Cancelled an agreement still live on-chain" + ); + (tx_hash, false, None) + } + LiveCancel::CancelFailed(err) => (None, false, Some(err)), + LiveCancel::Unconfirmed { tx_hash, err } => { + log_unconfirmed(&row.agreement, tx_hash, &err); + return note_check(registry, row, None, None).await; + } + }; + if failure.is_none() + && confirm_if_over(registry, config, row, tx_hash, by_indexer, chain_now).await + { + return; + } + // A cancel that failed found it live; otherwise it is over, or withdrawn until its deadline. + note_check(registry, row, failure.as_ref(), Some(failure.is_none())).await; +} + +/// Mark the agreement ended by dipper once it can't go live again: this sweep's cancel +/// ended it, or nobody accepted its offer before the deadline to. One ended otherwise is left +/// to the chain listener for a while; one the indexer ended then becomes `CanceledByIndexer`. +async fn confirm_if_over( + registry: &R, + config: &IndexingAgreementConfig, + row: &CancellingAgreement, + tx_hash: Option, + by_indexer: bool, + chain_now: u64, +) -> bool { + let agreement = &row.agreement; + let past_grace = row + .ended_seen_at + .is_some_and(|seen| seen < time::OffsetDateTime::now_utc() - LISTENER_GRACE); + let can_confirm = if row.accepted_on_chain { + tx_hash.is_some() || past_grace + } else { + chain_now > agreement.terms.deadline + }; + if !can_confirm { + return false; + } + if by_indexer { + tracing::info!( + agreement_id = %agreement.id, + "The indexer ended an agreement dipper was cancelling" + ); + return past_grace && record_end_by_indexer(registry, agreement).await; + } + let reason = if row.abandoned { + CancelReason::Abandoned + } else { + CancelReason::NotWanted + }; + confirm_cancelled(registry, agreement, reason, tx_hash, config).await +} + +/// Mark an agreement the indexer ended `CanceledByIndexer` when the chain listener hasn't in +/// time, so it doesn't stay cancelling for good. The indexer is recorded as ending it first, +/// so its announcement names them; the time recorded is when dipper noticed. +async fn record_end_by_indexer( + registry: &R, + agreement: &IndexingAgreement, +) -> bool { + let indexer = agreement.indexer.id.to_string(); + let marked = match registry + .record_cancel_audit(&agreement.id, now_secs(), &indexer, None) + .await + { + Ok(()) => { + registry + .apply_reconciliation(&agreement.id, false, Some(CancelKind::ByIndexer)) + .await + } + Err(err) => Err(err), + }; + match marked { + Ok(outcome) => { + tracing::info!( + agreement_id = %agreement.id, + indexing_request_id = %agreement.indexing_request_id, + old_status = "CANCELLING", + new_status = "CANCELED_BY_INDEXER", + applied = outcome.did_cancel, + reason = "indexer_cancel_seen_on_chain", + "agreement state transition" + ); + true + } + Err(err) => { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to mark an agreement the indexer ended, will retry" + ); + false + } + } +} + +/// Record that the agreement was checked and is still cancelling, and whether it was found +/// ended, counting a cancel the chain answered without ending it; at the limit, an ERROR. +async fn note_check( + registry: &R, + row: &CancellingAgreement, + failure: Option<&ChainClientError>, + ended: Option, +) { + let agreement = &row.agreement; + let failed_attempts = failure.map_or(0, failed_attempts); + if let Some(err) = failure + && failed_attempts == 0 + { + log_uncounted_failure(agreement, err); + } + match registry + .record_cancel_check(&agreement.id, failed_attempts, ended) + .await + { + Ok(attempts) => { + if let Some(err) = failure.filter(|_| failed_attempts > 0) { + log_failed_cancel(row, attempts, failed_attempts, err); + } + } + Err(err) => tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to record a check of a cancelling agreement" + ), + } +} + +fn log_uncounted_failure(agreement: &IndexingAgreement, err: &ChainClientError) { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Cancel of an agreement failed or could not be confirmed, will retry" + ); +} + +/// One ERROR as an agreement reaches the limit, for an operator to look into; a WARN for +/// every other failed cancel. +fn log_failed_cancel( + row: &CancellingAgreement, + attempts: u32, + failed: u32, + err: &ChainClientError, +) { + let agreement = &row.agreement; + let reached_limit = + attempts >= MAX_CANCEL_ATTEMPTS && attempts.saturating_sub(failed) < MAX_CANCEL_ATTEMPTS; + if !reached_limit { + tracing::warn!( + agreement_id = %agreement.id, + attempts, + error = %err, + "Cancel failed, never mined, or did not end the agreement; will retry" + ); + return; + } + tracing::error!( + event = "agreement_cancel_stuck", + agreement_id = %agreement.id, + indexer_id = %agreement.indexer.id, + indexing_request_id = %agreement.indexing_request_id, + abandoned = row.abandoned, + attempts, + error = %err, + "Cancelling an agreement keeps failing; it may still be live. Dipper now retries it hourly" + ); +} + +/// How many of an agreement's cancel attempts a failed cancel uses up. The chain was read just +/// before, so any failure counts, a refusal to send (gas over the cap, signer out of funds) +/// included, except a cancel whose receipt checks all failed, which may have mined unseen. +fn failed_attempts(err: &ChainClientError) -> u32 { + match err { + ChainClientError::TxDropped { + receipt_checked: false, + .. + } => 0, + ChainClientError::MissingTermsVersionHash { .. } => MAX_CANCEL_ATTEMPTS, + _ => 1, + } +} + +#[cfg(test)] +mod tests { + use std::sync::{ + Mutex, + atomic::{AtomicBool, AtomicU32, AtomicU64, Ordering}, + }; + + use async_trait::async_trait; + use dipper_core::ids::{IndexingAgreementId, IndexingRequestId}; + use dipper_rpc::indexer::indexer_client::sol::RecurringCollectionAgreement; + use thegraph_core::alloy::primitives::Address; + + use super::*; + use crate::{ + cancel_dispatch::tests::agreement, + chain_client::AgreementOnChain, + registry::{IndexingAgreementStatus, StubAgreementRegistry}, + worker::service::JobPriority, + }; + + const DEADLINE: u64 = 1_000; + + #[derive(Default)] + struct MockRegistry { + cancelling: Vec, + marked_cancelled: Mutex>, + marked_by_indexer: Mutex>, + audits: Mutex>>, + audited_by: Mutex>, + attempts: AtomicU32, + checks: AtomicU32, + found_ended: Mutex>>, + /// The chain listener marks it ended before the retry's own mark lands. + listener_ended_it: bool, + writes: Mutex>, + /// Ended agreements awaiting replacement, until noted as replaced. + awaiting_replacement: Vec, + replacements_noted: Mutex>, + } + + #[async_trait] + impl StubAgreementRegistry for MockRegistry { + async fn get_cancelling_agreements( + &self, + _batch_size: i64, + _max_attempts: u32, + _min_age_minutes: i32, + ) -> crate::registry::Result> { + Ok(self.cancelling.clone()) + } + async fn get_ended_agreements_awaiting_replacement( + &self, + _batch_size: i64, + ) -> crate::registry::Result> { + let noted = self.replacements_noted.lock().unwrap(); + Ok(self + .awaiting_replacement + .iter() + .filter(|agreement| !noted.contains(&agreement.id)) + .cloned() + .collect()) + } + async fn mark_replacement_queued( + &self, + id: &IndexingAgreementId, + ) -> crate::registry::Result<()> { + self.replacements_noted.lock().unwrap().push(*id); + Ok(()) + } + async fn mark_indexing_agreement_as_canceled_by_requester( + &self, + id: &IndexingAgreementId, + ) -> crate::registry::Result<()> { + if self.listener_ended_it { + return Err(crate::registry::Error::NoRecordsUpdated); + } + self.marked_cancelled.lock().unwrap().push(*id); + self.writes.lock().unwrap().push("ended"); + Ok(()) + } + async fn record_cancel_audit( + &self, + _id: &IndexingAgreementId, + _canceled_at: u64, + canceled_by: &str, + canceled_tx: Option<&str>, + ) -> crate::registry::Result<()> { + self.audited_by.lock().unwrap().push(canceled_by.to_owned()); + self.writes.lock().unwrap().push("cancel recorded"); + self.audits + .lock() + .unwrap() + .push(canceled_tx.map(str::to_owned)); + Ok(()) + } + async fn apply_reconciliation( + &self, + id: &IndexingAgreementId, + _apply_accept: bool, + cancel: Option, + ) -> crate::registry::Result { + assert_eq!(cancel, Some(CancelKind::ByIndexer)); + self.marked_by_indexer.lock().unwrap().push(*id); + Ok(crate::registry::ReconciliationOutcome { + did_accept: false, + did_cancel: true, + }) + } + async fn record_cancel_check( + &self, + _id: &IndexingAgreementId, + failed_attempts: u32, + ended: Option, + ) -> crate::registry::Result { + self.checks.fetch_add(1, Ordering::SeqCst); + self.found_ended.lock().unwrap().push(ended); + Ok(self.attempts.fetch_add(failed_attempts, Ordering::SeqCst) + failed_attempts) + } + } + + /// An agreement live on-chain until a cancel ends it, unless set to ignore cancels. + #[derive(Default)] + struct MockChain { + live: AtomicBool, + read_fails: bool, + /// What every send fails with, when set. + send_error: Option ChainClientError>, + mined_cancel_reverts: bool, + never_mines: bool, + receipt_unreadable: bool, + reverts_before_sending: bool, + cancel_has_no_effect: bool, + clock_fails: bool, + clock_reads: AtomicU32, + now: AtomicU64, + ended_by_indexer: bool, + indexer_ends_it_first: bool, + read_back_fails: bool, + cancels_sent: AtomicU32, + reads: AtomicU32, + } + + #[async_trait] + impl ChainClient for MockChain { + async fn offer_via_manager( + &self, + _rca: &RecurringCollectionAgreement, + ) -> Result, ChainClientError> { + unimplemented!() + } + async fn cancel_via_manager( + &self, + _collector: Address, + _agreement_id: &[u8; 16], + _version_hash: B256, + _options: u16, + ) -> Result, ChainClientError> { + if let Some(send_error) = self.send_error { + return Err(send_error()); + } + if self.reverts_before_sending { + return Err(ChainClientError::ContractRevert { + selector: [0xde, 0xad, 0xbe, 0xef], + data: Default::default(), + }); + } + if self.mined_cancel_reverts { + return Err(ChainClientError::TxReverted { + tx_hash: B256::repeat_byte(0xee), + }); + } + if self.never_mines || self.receipt_unreadable { + return Err(ChainClientError::TxDropped { + tx_hash: B256::repeat_byte(0xdd), + receipt_checked: self.never_mines, + }); + } + self.cancels_sent.fetch_add(1, Ordering::SeqCst); + if !self.cancel_has_no_effect || self.indexer_ends_it_first { + self.live.store(false, Ordering::SeqCst); + } + Ok(Some(B256::repeat_byte(0xcd))) + } + async fn reconcile_provider( + &self, + _collector: Address, + _provider: Address, + ) -> Result, ChainClientError> { + unimplemented!() + } + async fn reconcile_agreement( + &self, + _collector: Address, + _agreement_id: &[u8; 16], + ) -> Result, ChainClientError> { + unimplemented!() + } + async fn agreement_on_chain( + &self, + _agreement_id: &[u8; 16], + ) -> Result { + let earlier_reads = self.reads.fetch_add(1, Ordering::SeqCst); + if self.read_fails || (self.read_back_fails && earlier_reads > 0) { + return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); + } + Ok(if self.live.load(Ordering::SeqCst) { + AgreementOnChain::Live + } else if self.ended_by_indexer || self.indexer_ends_it_first { + AgreementOnChain::EndedByIndexer + } else { + AgreementOnChain::NotLive + }) + } + async fn latest_block_timestamp(&self) -> Result { + self.clock_reads.fetch_add(1, Ordering::SeqCst); + if self.clock_fails { + return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); + } + Ok(self.now.load(Ordering::SeqCst)) + } + } + + fn registry_with_one(accepted_on_chain: bool) -> MockRegistry { + let mut cancelling = agreement(IndexingAgreementStatus::Cancelling, Some(vec![7u8; 32])); + cancelling.terms.deadline = DEADLINE; + MockRegistry { + cancelling: vec![CancellingAgreement { + agreement: cancelling, + accepted_on_chain, + ended_seen_at: None, + abandoned: false, + }], + ..MockRegistry::default() + } + } + + fn live_chain() -> MockChain { + MockChain { + live: AtomicBool::new(true), + ..MockChain::default() + } + } + + async fn retry(registry: &MockRegistry, chain: &MockChain, chain_now: u64) { + let config = IndexingAgreementConfig::for_tests(); + chain.now.store(chain_now, Ordering::SeqCst); + retry_cancelling_agreements(registry, chain, &config).await; + } + + #[async_trait] + impl IndexingRequestRegistry for MockRegistry { + async fn set_indexing_target_candidates( + &self, + _requested_by: Address, + _deployment_id: thegraph_core::DeploymentId, + _deployment_chain_id: u64, + _num_candidates: usize, + ) -> crate::registry::Result { + unimplemented!() + } + async fn get_all_indexing_requests( + &self, + ) -> crate::registry::Result> { + unimplemented!() + } + async fn get_indexing_request_by_id( + &self, + id: &IndexingRequestId, + ) -> crate::registry::Result> { + let agreement = ended_abandoned(); + Ok(Some(crate::registry::IndexingRequest { + id: *id, + created_at: time::OffsetDateTime::now_utc(), + updated_at: time::OffsetDateTime::now_utc(), + status: crate::registry::IndexingRequestStatus::Open, + requested_by: Address::ZERO, + deployment_id: agreement.terms.metadata.subgraph_deployment_id, + deployment_chain_id: agreement.terms.metadata.chain_id, + num_candidates: 3, + })) + } + async fn get_indexing_requests_by_deployment_id( + &self, + _deployment_id: &thegraph_core::DeploymentId, + ) -> crate::registry::Result> { + unimplemented!() + } + async fn get_open_indexing_requests_for_reassessment( + &self, + _min_age_seconds: i64, + _batch_size: i64, + ) -> crate::registry::Result> { + unimplemented!() + } + } + + /// Records the requests it is asked to reassess. + #[derive(Default)] + struct MockQueue(Arc>>); + + #[async_trait] + impl WorkerQueue for MockQueue { + async fn send_indexing_agreement_proposal( + &self, + _candidate_url: url::Url, + _agreement_id: IndexingAgreementId, + _indexing_request_id: IndexingRequestId, + _deployment_id: thegraph_core::DeploymentId, + _deployment_chain_id: u64, + _priority: JobPriority, + ) -> anyhow::Result { + unimplemented!() + } + async fn reassess_indexing_request( + &self, + indexing_request_id: IndexingRequestId, + _deployment_id: thegraph_core::DeploymentId, + _deployment_chain_id: u64, + _num_candidates: usize, + _priority: JobPriority, + ) -> anyhow::Result { + self.0.lock().unwrap().push(indexing_request_id); + Ok(dipper_pgmq::JobId::default()) + } + async fn submit_offer( + &self, + _agreement_id: IndexingAgreementId, + _indexing_request_id: IndexingRequestId, + _indexer_url: url::Url, + _deployment_id: thegraph_core::DeploymentId, + _deployment_chain_id: u64, + _priority: JobPriority, + ) -> anyhow::Result { + unimplemented!() + } + } + + fn ended_abandoned() -> IndexingAgreement { + agreement( + IndexingAgreementStatus::AbandonedByIndexer, + Some(vec![7u8; 32]), + ) + } + + /// Nothing else finishes a cancel, so the retry can't wait on the chain listener, which + /// config can turn off. + #[tokio::test(start_paused = true)] + async fn sweeps_on_its_own_and_replaces_an_abandoned_agreement_once_it_has_ended() { + let ended = ended_abandoned(); + let registry = MockRegistry { + awaiting_replacement: vec![ended.clone()], + ..registry_with_one(true) + }; + let chain = Arc::new(live_chain()); + let queue = MockQueue::default(); + let reassessed = Arc::clone(&queue.0); + let (handle, service) = new(Ctx { + registry, + chain_client: Arc::clone(&chain), + agreement_conf: Arc::new(IndexingAgreementConfig::for_tests()), + worker_queue: queue, + }); + let service = tokio::spawn(service); + + tokio::time::sleep(SWEEP_INTERVAL + Duration::from_secs(1)).await; + handle.stop().await; + + service.await.unwrap().unwrap(); + assert_eq!( + chain.clock_reads.load(Ordering::SeqCst), + 2, + "1 sweep at start, 1 later" + ); + assert_eq!( + *reassessed.lock().unwrap(), + vec![ended.indexing_request_id], + "queued once, then noted as replaced" + ); + } + + #[tokio::test] + async fn waits_for_the_next_sweep_when_the_chain_time_cannot_be_read() { + let registry = registry_with_one(true); + let chain = MockChain { + clock_fails: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 0); + assert_eq!(registry.checks.load(Ordering::SeqCst), 0); + } + + #[tokio::test] + async fn leaves_the_chain_alone_when_nothing_is_being_cancelled() { + let registry = MockRegistry::default(); + let chain = live_chain(); + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.clock_reads.load(Ordering::SeqCst), 0); + } + + #[tokio::test] + async fn cancels_a_live_accepted_agreement_and_records_the_cancel() { + let registry = registry_with_one(true); + let chain = live_chain(); + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 1); + assert_eq!(registry.marked_cancelled.lock().unwrap().len(), 1); + let tx = B256::repeat_byte(0xcd).to_string(); + assert_eq!(*registry.audits.lock().unwrap(), vec![Some(tx)]); + } + + #[tokio::test] + async fn records_the_cancel_before_marking_the_agreement_ended() { + // The end is announced once the mark lands; recorded after, the announcement could + // go out without its transaction and never be sent again. + let registry = registry_with_one(true); + + retry(®istry, &live_chain(), 0).await; + + assert_eq!( + *registry.writes.lock().unwrap(), + vec!["cancel recorded", "ended"] + ); + } + + #[tokio::test] + async fn counts_an_agreement_the_listener_marked_ended_first_as_ended() { + // Not a failure: there is nothing left to retry. + let registry = MockRegistry { + listener_ended_it: true, + ..registry_with_one(true) + }; + + retry(®istry, &live_chain(), 0).await; + + assert_eq!(registry.checks.load(Ordering::SeqCst), 0); + } + + #[tokio::test] + async fn leaves_an_accepted_agreement_that_already_ended_to_the_listener() { + // The indexer may have ended it, or an earlier cancel whose result went unread; + // the chain listener reads which, and records when and in which transaction. + let registry = registry_with_one(true); + let chain = MockChain::default(); + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 0); + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + assert!(registry.audits.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn marks_an_ended_accepted_agreement_itself_once_the_listener_has_had_long_enough() { + // In case the listener never reads its end, it would otherwise stay cancelling. + let mut registry = registry_with_one(true); + registry.cancelling[0].ended_seen_at = + Some(time::OffsetDateTime::now_utc() - LISTENER_GRACE - time::Duration::MINUTE); + let chain = MockChain::default(); + + retry(®istry, &chain, 0).await; + + assert_eq!(registry.marked_cancelled.lock().unwrap().len(), 1); + assert!(registry.audits.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn gives_the_listener_its_hour_from_when_the_end_is_first_seen() { + // Cancelling for hours, as while the manager was paused, mustn't count towards it. + let mut registry = registry_with_one(true); + registry.cancelling[0].agreement.updated_at = + time::OffsetDateTime::now_utc() - LISTENER_GRACE * 3; + let chain = MockChain::default(); + + retry(®istry, &chain, 0).await; + + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + assert_eq!(*registry.found_ended.lock().unwrap(), vec![Some(true)]); + } + + #[tokio::test] + async fn notes_a_live_agreement_as_not_ended_and_an_unread_one_as_unknown() { + let registry = registry_with_one(true); + let chain = MockChain { + cancel_has_no_effect: true, + ..live_chain() + }; + retry(®istry, &chain, 0).await; + + let unread = MockChain { + read_fails: true, + ..MockChain::default() + }; + retry(®istry, &unread, 0).await; + + assert_eq!( + *registry.found_ended.lock().unwrap(), + vec![Some(false), None] + ); + } + + #[tokio::test] + async fn leaves_an_end_by_the_indexer_to_the_listener_for_a_while() { + // The listener records when and in which transaction. An accepted agreement can lack + // an accept time, if accepted before accepts were recorded, so both kinds are checked. + for accepted_on_chain in [true, false] { + let registry = registry_with_one(accepted_on_chain); + let chain = MockChain { + ended_by_indexer: true, + ..MockChain::default() + }; + + retry(®istry, &chain, DEADLINE + 1).await; + + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + assert!(registry.marked_by_indexer.lock().unwrap().is_empty()); + assert_eq!(registry.checks.load(Ordering::SeqCst), 1); + } + } + + #[tokio::test] + async fn marks_an_end_by_the_indexer_as_theirs_once_the_listener_has_had_long_enough() { + for accepted_on_chain in [true, false] { + let mut registry = registry_with_one(accepted_on_chain); + registry.cancelling[0].ended_seen_at = + Some(time::OffsetDateTime::now_utc() - LISTENER_GRACE - time::Duration::MINUTE); + let chain = MockChain { + ended_by_indexer: true, + ..MockChain::default() + }; + + retry(®istry, &chain, DEADLINE + 1).await; + + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + assert_eq!(registry.marked_by_indexer.lock().unwrap().len(), 1); + let indexer = registry.cancelling[0].agreement.indexer.id.to_string(); + assert_eq!(*registry.audited_by.lock().unwrap(), vec![indexer]); + assert_eq!(*registry.audits.lock().unwrap(), vec![None]); + } + } + + #[tokio::test] + async fn reads_an_ended_agreement_once_to_learn_who_ended_it() { + let mut registry = registry_with_one(true); + registry.cancelling[0].ended_seen_at = + Some(time::OffsetDateTime::now_utc() - LISTENER_GRACE - time::Duration::MINUTE); + let chain = MockChain { + ended_by_indexer: true, + ..MockChain::default() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.reads.load(Ordering::SeqCst), 1); + assert_eq!(registry.marked_by_indexer.lock().unwrap().len(), 1); + } + + #[tokio::test] + async fn leaves_an_end_the_indexer_beat_dipper_to_as_theirs() { + // Dipper's cancel mined as a no-op after the indexer's; recording it as dipper's + // would announce the wrong canceller, and the listener's details would be ignored. + let registry = registry_with_one(true); + let chain = MockChain { + indexer_ends_it_first: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 1); + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + assert!(registry.audits.lock().unwrap().is_empty()); + assert_eq!(registry.attempts.load(Ordering::SeqCst), 0); + } + + #[tokio::test] + async fn neither_counts_nor_confirms_a_mined_cancel_it_could_not_read_back() { + // It may well have worked; the next check reads the agreement again. + let registry = registry_with_one(true); + let chain = MockChain { + read_back_fails: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 1); + assert_eq!(registry.attempts.load(Ordering::SeqCst), 0); + assert_eq!(registry.checks.load(Ordering::SeqCst), 1); + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn keeps_an_unaccepted_agreement_cancelling_until_its_deadline() { + // An offer still in flight could land and be accepted until then. + let registry = registry_with_one(false); + let chain = MockChain::default(); + + retry(®istry, &chain, DEADLINE).await; + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + + retry(®istry, &chain, DEADLINE + 1).await; + assert_eq!(registry.marked_cancelled.lock().unwrap().len(), 1); + assert!(registry.audits.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn withdraws_a_live_offer_before_its_deadline_and_keeps_watching() { + let registry = registry_with_one(false); + let chain = live_chain(); + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 1); + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn notes_each_check_that_leaves_an_agreement_cancelling() { + // So the next sweep starts with the agreements checked longest ago. + let registry = registry_with_one(false); + let chain = MockChain::default(); + + retry(®istry, &chain, DEADLINE).await; + + assert_eq!(registry.checks.load(Ordering::SeqCst), 1); + assert_eq!(registry.attempts.load(Ordering::SeqCst), 0); + } + + #[tokio::test] + async fn never_confirms_an_agreement_it_could_not_read() { + // Past its deadline but unread, it may be live: an accept the listener hasn't + // recorded, or one from before accepts were recorded. + let registry = registry_with_one(false); + let chain = MockChain { + read_fails: true, + ..live_chain() + }; + + retry(®istry, &chain, DEADLINE + 1).await; + + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + assert_eq!(registry.checks.load(Ordering::SeqCst), 1); + } + + #[tokio::test] + async fn an_unreachable_chain_neither_sends_nor_counts_an_attempt() { + let registry = registry_with_one(true); + let chain = MockChain { + read_fails: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 0); + assert_eq!(registry.attempts.load(Ordering::SeqCst), 0); + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn counts_a_cancel_that_could_not_be_sent() { + // The chain was just read, so a send that fails every time (gas over the cap, signer + // out of funds) is no outage, and only counting it reaches the alert. + let refusals: [fn() -> ChainClientError; 2] = [ + || ChainClientError::SubmitFailed(anyhow::anyhow!("Gas price exceeds maximum")), + || { + ChainClientError::RpcError(anyhow::anyhow!( + "Gas estimation failed: insufficient funds" + )) + }, + ]; + for err in refusals { + let registry = registry_with_one(true); + let chain = MockChain { + send_error: Some(err), + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 0); + assert_eq!(registry.attempts.load(Ordering::SeqCst), 1); + } + } + + #[tokio::test] + async fn gives_up_at_once_on_an_agreement_it_can_never_cancel() { + // Without a stored terms hash no cancel can be sent, so retrying only delays the alert. + let mut registry = registry_with_one(true); + registry.cancelling[0].agreement.terms_version_hash = None; + let chain = live_chain(); + + retry(®istry, &chain, 0).await; + + assert_eq!( + registry.attempts.load(Ordering::SeqCst), + MAX_CANCEL_ATTEMPTS + ); + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 0); + } + + #[tokio::test] + async fn counts_a_cancel_that_is_mined_and_reverts() { + // Each one costs gas, so it can't be retried without limit. + let registry = registry_with_one(true); + let chain = MockChain { + mined_cancel_reverts: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(registry.attempts.load(Ordering::SeqCst), 1); + } + + #[tokio::test] + async fn counts_a_cancel_that_never_mines() { + // Otherwise one that keeps being dropped is sent every sweep for ever, never alerting. + let registry = registry_with_one(true); + let chain = MockChain { + never_mines: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(registry.attempts.load(Ordering::SeqCst), 1); + } + + #[tokio::test] + async fn does_not_count_a_cancel_whose_receipt_could_not_be_checked() { + // Every receipt check failing is an outage, which may hide a cancel that mined. + let registry = registry_with_one(true); + let chain = MockChain { + receipt_unreadable: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(registry.attempts.load(Ordering::SeqCst), 0); + } + + #[tokio::test] + async fn counts_a_cancel_the_contract_refuses_before_it_is_sent() { + // Otherwise one that always reverts is retried, and alerted on, for ever. + let registry = registry_with_one(true); + let chain = MockChain { + reverts_before_sending: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(registry.attempts.load(Ordering::SeqCst), 1); + } + + #[tokio::test] + async fn counts_a_cancel_that_mines_without_ending_the_agreement() { + let registry = registry_with_one(true); + let chain = MockChain { + cancel_has_no_effect: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(registry.attempts.load(Ordering::SeqCst), 1); + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + } +} diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index 2082646b..a3ce327c 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -56,7 +56,6 @@ use crate::{ AgreementRegistry, CancelKind, IndexingAgreement, IndexingAgreementStatus, PendingCancellationRegistry, ReconciliationItem, }, - worker::service::{JobPriority, WorkerQueue}, }; /// Idle interval used when no `Created` agreements are awaiting acceptance. @@ -83,7 +82,6 @@ const SWEEP_BATCH_SIZE: i64 = 1000; /// crash-recovery; the steady-state fan-out fires from finalize on a /// fresh accept, so per-poll execution is wasted DB work. const SWEEP_POLLS: u64 = 60; - /// Handle for controlling the chain listener service lifecycle #[derive(Clone)] pub struct Handle { @@ -103,11 +101,9 @@ impl Handle { } /// Context required by the chain listener service -pub struct Ctx { +pub struct Ctx { /// Registry for querying and updating agreements pub registry: R, - /// Worker queue (still used by reconciliation paths that hand work back to the worker) - pub worker_queue: W, /// Chain event source (subgraph) pub event_source: E, /// Chain client used to cancel on-chain via the RecurringAgreementManager @@ -147,10 +143,9 @@ pub struct ChainListenerState { clippy::too_many_lines, reason = "predates this lint; fix when next touched" )] -pub fn new(ctx: Ctx) -> (Handle, impl Future>) +pub fn new(ctx: Ctx) -> (Handle, impl Future>) where R: AgreementRegistry + ChainListenerStateRegistry + PendingCancellationRegistry + Send + Sync, - W: WorkerQueue + Send + Sync, E: ChainEventSource, T: ChainClient + Send + Sync, { @@ -158,7 +153,6 @@ where let Ctx { registry, - worker_queue, event_source, chain_client, agreement_conf, @@ -305,7 +299,6 @@ where chain_ts_drift_tolerance_secs, bypass_chain_clock_defenses, ®istry, - &worker_queue, &chain_client, &event_source, &mut rx_stop, @@ -424,7 +417,7 @@ struct DrainOutcome { clippy::too_many_lines, reason = "predates this lint; fix when next touched" )] -async fn drain_once( +async fn drain_once( cursor: &mut Cursor, last_persisted_timestamp: &mut Option, last_chain_ts_persist_wall: &mut std::time::Instant, @@ -436,7 +429,6 @@ async fn drain_once( chain_ts_drift_tolerance_secs: u64, bypass_chain_clock_defenses: bool, registry: &R, - worker_queue: &W, chain_client: &T, event_source: &E, rx_stop: &mut mpsc::Receiver<()>, @@ -444,7 +436,6 @@ async fn drain_once( ) -> Result where R: AgreementRegistry + ChainListenerStateRegistry + PendingCancellationRegistry + Send + Sync, - W: WorkerQueue + Send + Sync, E: ChainEventSource, T: ChainClient + Send + Sync, { @@ -587,7 +578,7 @@ where } let agreement = agreements_by_id.remove(&snapshot.agreement_id); - match prepare_reconciliation(&snapshot, agreement, registry, worker_queue).await { + match prepare_reconciliation(&snapshot, agreement, registry, chain_client).await { Ok(Some(prep)) => prepared.push(prep), Ok(None) => {} Err(err) => { @@ -807,15 +798,15 @@ fn apply_chain_ts_drift_cap( clippy::cognitive_complexity, reason = "predates this lint; fix when next touched" )] -async fn prepare_reconciliation( +async fn prepare_reconciliation( snapshot: &AgreementStateSnapshot, agreement: Option, registry: &R, - worker_queue: &W, + chain_client: &T, ) -> anyhow::Result> where R: AgreementRegistry + Sync, - W: WorkerQueue, + T: ChainClient, { tracing::debug!( agreement_id = %snapshot.agreement_id, @@ -852,12 +843,9 @@ where tracing::warn!( agreement_id = %agreement.id, indexer = %snapshot.indexer, - "Rejected agreement accepted on-chain, queuing cancellation" + "Rejected agreement accepted on-chain, cancelling it" ); - worker_queue - // Background: on-chain cleanup of a rejected-then-accepted agreement. - .cancel_rejected_agreement_on_chain(agreement.id, JobPriority::Background) - .await?; + crate::cancel_dispatch::reopen_if_live(registry, chain_client, &agreement).await?; // This row goes Rejected -> Canceled without ever transiting // AcceptedOnChain, so `apply_reconciliation` never records the accept. @@ -876,17 +864,26 @@ where } } + let reopened = + reopen_if_cancelled_but_accepted(snapshot, &agreement, registry, chain_client).await?; + record_accept_of_cancelling(snapshot, &agreement, reopened, registry).await; + // Both transitions are applied atomically downstream so the // Accept-then-Cancel-in-one-snapshot path can't leak an intermediate // AcceptedOnChain to concurrent readers. + // A withdrawn offer reads as cancelled with no accept time: it never went live. + let withdrawn_offer = snapshot.state.is_canceled() && snapshot.accepted_at == 0; let apply_accept = matches!( agreement.status, IndexingAgreementStatus::Created | IndexingAgreementStatus::Expired, - ) && snapshot.state.reached_accepted(); + ) && snapshot.state.reached_accepted() + && !withdrawn_offer; let already_terminal_cancel = matches!( agreement.status, - IndexingAgreementStatus::CanceledByRequester | IndexingAgreementStatus::CanceledByIndexer, + IndexingAgreementStatus::CanceledByRequester + | IndexingAgreementStatus::CanceledByIndexer + | IndexingAgreementStatus::AbandonedByIndexer, ); // Classify off the on-chain state, which carries the contract's own // "canceled by" flag. The canceler address is the payer, not dipper's @@ -926,13 +923,101 @@ where tracing::debug!( agreement_id = %agreement.id, status = %agreement.status, - "Agreement already canceled, ignoring snapshot" + "Agreement already canceled, recording any accept dipper missed" ); + record_accept_and_cancel_from_chain(snapshot, &agreement, registry).await; } Ok(None) } } +/// Record the accept and cancel of an agreement dipper had already marked +/// cancelled that went live on-chain first, so its accepted and terminated +/// events go out. Both go in 1 write, as the terminated sweep waits only for the accept. +/// Existing values win, so an agreement dipper already recorded is unchanged, +/// except an end recorded before the accept, such as its offer's withdrawal, which gives way. +async fn record_accept_and_cancel_from_chain( + snapshot: &AgreementStateSnapshot, + agreement: &IndexingAgreement, + registry: &R, +) { + if snapshot.accepted_at == 0 { + return; + } + let recorded = registry + .record_accept_and_cancel_audit( + &agreement.id, + snapshot.accepted_at, + &snapshot.accepted_tx, + snapshot.canceled_at, + &snapshot.canceled_by.to_string(), + Some(&snapshot.canceled_tx), + ) + .await; + if let Err(err) = recorded { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "failed to record the on-chain accept and cancel of a cancelled agreement; \ + its events go out only if the listener reads this agreement again" + ); + } +} + +/// Safety net for an agreement dipper cancelled whose offer the indexer accepted +/// anyway, such as one that landed after dipper's cancel. Nothing else would end +/// it: reconciliation ignores an accept on a cancelled row. It goes back to +/// `Cancelling` for the cancel retry, unless the chain shows it already ended. Returns whether it +/// moved it back. +async fn reopen_if_cancelled_but_accepted( + snapshot: &AgreementStateSnapshot, + agreement: &IndexingAgreement, + registry: &R, + chain_client: &T, +) -> anyhow::Result +where + R: AgreementRegistry + Sync, + T: ChainClient, +{ + if agreement.status == IndexingAgreementStatus::CanceledByRequester + && snapshot.state.reached_accepted() + && !snapshot.state.is_canceled() + { + return Ok( + crate::cancel_dispatch::reopen_if_live(registry, chain_client, agreement).await?, + ); + } + Ok(false) +} + +/// An agreement dipper is cancelling, or has just moved back to cancelling, stays `Cancelling` +/// when the chain shows it accepted, so its accept is recorded here; its end is then announced, +/// along with the accept, once the cancel lands. A withdrawn offer reads as cancelled with no +/// accept time. +async fn record_accept_of_cancelling( + snapshot: &AgreementStateSnapshot, + agreement: &IndexingAgreement, + reopened: bool, + registry: &R, +) { + if (agreement.status != IndexingAgreementStatus::Cancelling && !reopened) + || !snapshot.state.reached_accepted() + || snapshot.accepted_at == 0 + { + return; + } + if let Err(err) = registry + .record_accepted_audit(&agreement.id, snapshot.accepted_at, &snapshot.accepted_tx) + .await + { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "failed to record the on-chain accept of an agreement being cancelled" + ); + } +} + /// Log the transition that landed and, on fresh accepts, fan out the /// linked pending cancellations. #[expect( @@ -974,7 +1059,7 @@ where match prep.item.cancel { Some(CancelKind::ByRequester) => tracing::info!( agreement_id = %prep.agreement.id, - "Agreement marked as CanceledByRequester (on-chain confirmation)" + "Agreement marked as ended by dipper (on-chain confirmation)" ), Some(CancelKind::ByIndexer) => tracing::info!( agreement_id = %prep.agreement.id, @@ -1006,22 +1091,20 @@ where /// and applies whatever transitions the diff implies. See the module-level /// transition table for the full mapping. #[cfg(test)] -async fn reconcile_agreement( +async fn reconcile_agreement( snapshot: &AgreementStateSnapshot, registry: &R, - worker_queue: &W, chain_client: &T, config: &crate::config::IndexingAgreementConfig, ) -> anyhow::Result<()> where R: AgreementRegistry + PendingCancellationRegistry + Sync, - W: WorkerQueue, T: ChainClient, { let agreement = registry .get_indexing_agreement_by_id(&snapshot.agreement_id) .await?; - let Some(prep) = prepare_reconciliation(snapshot, agreement, registry, worker_queue).await? + let Some(prep) = prepare_reconciliation(snapshot, agreement, registry, chain_client).await? else { return Ok(()); }; @@ -1036,16 +1119,9 @@ where /// Execute pending cancellations linked to a newly-accepted agreement. /// /// Called from the Created -> AcceptedOnChain and Expired -> AcceptedOnChain -/// transitions. For each pending cancellation, fires -/// `cancelIndexingAgreementByPayer` against the RecurringCollector contract, -/// then flips the dipper DB row to `CanceledByRequester`. Each pending row is -/// deleted individually after both steps succeed; transient failures retain -/// the record so the next reconcile pass can retry. -#[expect( - clippy::cognitive_complexity, - clippy::too_many_lines, - reason = "predates this lint; fix when next touched" -)] +/// transitions. Each replaced agreement is marked `Cancelling` and its on-chain cancel +/// sent once; the cancel retry finishes any that don't end at once. +/// A pending row is deleted once its agreement is marked; a failed mark keeps it for retry. async fn execute_pending_cancellations( agreement_id: &IndexingAgreementId, registry: &R, @@ -1060,153 +1136,140 @@ where .get_pending_cancellations_by_new_agreement(*agreement_id) .await?; - if pending.is_empty() { - return Ok(()); - } - let mut transient_failures: u32 = 0; - for cancellation in &pending { - let old_agreement = match registry - .get_indexing_agreement_by_id(&cancellation.old_agreement_id) - .await? - { - None => { - tracing::warn!( - old_agreement_id = %cancellation.old_agreement_id, - "Pending cancellation references non-existent agreement, cleaning up" - ); - registry - .delete_pending_cancellation(*agreement_id, cancellation.old_agreement_id) - .await?; - continue; - } - Some(a) => a, - }; - - let mut on_chain_cancel_tx: Option = None; - match crate::cancel_dispatch::cancel_agreement_on_chain( + let old_agreement_id = cancellation.old_agreement_id; + if start_replaced_cancel( + agreement_id, + &old_agreement_id, + registry, chain_client, - &old_agreement, config, ) - .await + .await? { - Ok(Some(tx_hash)) => { - tracing::info!( - new_agreement_id = %agreement_id, - old_agreement_id = %cancellation.old_agreement_id, - %tx_hash, - "Submitted on-chain cancellation for replaced agreement" - ); - on_chain_cancel_tx = Some(tx_hash.to_string()); - } - Ok(None) => { - tracing::info!( - new_agreement_id = %agreement_id, - old_agreement_id = %cancellation.old_agreement_id, - "Agreement already canceled on-chain; proceeding with local cleanup" - ); - } - Err(err) => { - tracing::warn!( - old_agreement_id = %cancellation.old_agreement_id, - error = %err, - "On-chain cancel failed, retaining pending cancellation for retry" - ); - transient_failures += 1; - continue; - } + registry + .delete_pending_cancellation(*agreement_id, old_agreement_id) + .await?; + } else { + transient_failures += 1; } + } - match registry - .mark_indexing_agreement_as_canceled_by_requester(&cancellation.old_agreement_id) - .await - { - Ok(()) => {} - Err(crate::registry::Error::NoRecordsUpdated) => { - tracing::debug!( - old_agreement_id = %cancellation.old_agreement_id, - "Old agreement already in terminal state, skipping local cancel flip" - ); - registry - .delete_pending_cancellation(*agreement_id, cancellation.old_agreement_id) - .await?; - continue; - } - Err(err) => { - tracing::error!( - old_agreement_id = %cancellation.old_agreement_id, - error = %err, - "On-chain cancel succeeded but DB update failed, retaining pending row" - ); - transient_failures += 1; - continue; - } - } + if transient_failures > 0 { + anyhow::bail!( + "{transient_failures} pending cancellation(s) failed for agreement {}; \ + records retained for retry", + agreement_id, + ); + } - registry - .delete_pending_cancellation(*agreement_id, cancellation.old_agreement_id) - .await?; + Ok(()) +} - tracing::info!( - new_agreement_id = %agreement_id, - old_agreement_id = %cancellation.old_agreement_id, - "Canceled old agreement on-chain and in dipper DB after replacement confirmed" +/// Start cancelling an agreement its accepted replacement supersedes. Returns whether its +/// pending cancellation is done with: marked, already ended, or gone. +async fn start_replaced_cancel( + new_agreement_id: &IndexingAgreementId, + old_agreement_id: &IndexingAgreementId, + registry: &R, + chain_client: &T, + config: &crate::config::IndexingAgreementConfig, +) -> anyhow::Result +where + R: AgreementRegistry + Sync, + T: ChainClient, +{ + let Some(old_agreement) = registry + .get_indexing_agreement_by_id(old_agreement_id) + .await? + else { + tracing::warn!( + %old_agreement_id, + "Pending cancellation references non-existent agreement, cleaning up" ); + return Ok(true); + }; + if old_agreement.status == IndexingAgreementStatus::Expired { + match expired_but_live(chain_client, &old_agreement).await { + Some(true) => {} + Some(false) => return Ok(true), + None => return Ok(false), + } + } + let started = crate::cancel_dispatch::start_cancel( + registry, + chain_client, + &old_agreement, + crate::cancel_dispatch::CancelReason::NotWanted, + config, + ) + .await; + Ok(note_replaced_cancel( + new_agreement_id, + &old_agreement, + started, + )) +} - // Record the cancel audit so `sweep_pending_terminated_events` can emit - // the `terminated` durably. Crucially, the sweep only emits for rows that - // were genuinely accepted on-chain (`accepted_at IS NOT NULL`): a - // proposed-but-never-accepted replacement records audit here but is never - // swept, so it produces no spurious `terminated`. - let manager = config.recurring_agreement_manager().to_string(); - if let Err(err) = registry - .record_cancel_audit( - &cancellation.old_agreement_id, - dipper_core::time::now_secs(), - &manager, - on_chain_cancel_tx.as_deref(), - ) - .await - { +/// Whether an agreement marked `Expired` is live on-chain after all, accepted unseen by a +/// lagging listener; `None`, logged, when the chain can't be read. One that really expired +/// stays `Expired`. +async fn expired_but_live( + chain_client: &T, + agreement: &IndexingAgreement, +) -> Option { + match chain_client + .agreement_on_chain(agreement.id.as_bytes()) + .await + { + Ok(on_chain) => Some(on_chain.is_live()), + Err(err) => { tracing::warn!( - old_agreement_id = %cancellation.old_agreement_id, + old_agreement_id = %agreement.id, error = %err, - "failed to record cancel audit; terminated event may emit with fallback fields" + "Failed to read whether a replaced expired agreement is live, retaining pending row" ); + None } } +} - if transient_failures > 0 { - anyhow::bail!( - "{transient_failures} pending cancellation(s) failed for agreement {}; \ - records retained for retry", - agreement_id, - ); +/// Log how cancelling a replaced agreement started; false when it couldn't be marked. +fn note_replaced_cancel( + new_agreement_id: &IndexingAgreementId, + old_agreement: &IndexingAgreement, + started: crate::registry::Result, +) -> bool { + let old_agreement_id = old_agreement.id; + match started { + Ok(started) => tracing::info!( + %new_agreement_id, + %old_agreement_id, + old_status = %old_agreement.status, + ?started, + reason = "replacement_accepted", + "Cancelling replaced agreement" + ), + Err(crate::registry::Error::NoRecordsUpdated) => tracing::debug!( + %old_agreement_id, + "Replaced agreement already ended or being cancelled" + ), + Err(err) => { + tracing::error!( + %old_agreement_id, + error = %err, + "Failed to mark replaced agreement cancelling, retaining pending row" + ); + return false; + } } - - Ok(()) + true } -/// Retry on-chain cancels for agreements orphaned by a failed shrink-to-zero. -/// -/// When a `set_indexing_target_candidates(num_candidates = 0)` call flips the -/// request row to `Canceled`, reassessment fires `cancelIndexingAgreementByPayer` -/// for every agreement under it. A transient chain-client error during that -/// fan-out leaves the request row `Canceled` and at least one agreement still -/// `AcceptedOnChain` — the local intent and on-chain state disagree, and the -/// admin RPC has nothing left to trigger. -/// -/// This sweep runs periodically on the chain_listener tick and re-fires the -/// on-chain cancel for each such orphan. The chain-side cancel is idempotent -/// (the `Ok(None)` revert path handles already-canceled agreements), and the -/// DB transition is gated on chain success, so this is safe to run on every -/// sweep without coordination with reassessment. -#[expect( - clippy::cognitive_complexity, - reason = "predates this lint; fix when next touched" -)] +/// Start cancelling agreements orphaned by a failed shrink-to-zero: still `AcceptedOnChain` +/// although their request is `Canceled`, because reassessment couldn't mark them. Each starts +/// like any other cancel, so the cancel retry finishes it with the same limit. async fn sweep_orphan_canceled_agreements( registry: &R, chain_client: &T, @@ -1229,76 +1292,39 @@ async fn sweep_orphan_canceled_agreements( } }; - if orphans.is_empty() { - return; - } - - tracing::debug!( - count = orphans.len(), - "Sweeping orphan agreements whose parent request is Canceled" - ); - for agreement in orphans { - let mut on_chain_cancel_tx: Option = None; - match crate::cancel_dispatch::cancel_agreement_on_chain(chain_client, &agreement, config) - .await - { - Ok(Some(tx_hash)) => { - tracing::info!( - agreement_id = %agreement.id, - %tx_hash, - "Submitted on-chain cancel for orphan agreement" - ); - on_chain_cancel_tx = Some(tx_hash.to_string()); - } - Ok(None) => { - tracing::info!( - agreement_id = %agreement.id, - "Orphan agreement already canceled on-chain; cleaning up local state" - ); - } - Err(err) => { - tracing::warn!( - error = %err, - agreement_id = %agreement.id, - "Failed to cancel orphan agreement on-chain; will retry next sweep" - ); - continue; - } - } - - if let Err(err) = registry - .mark_indexing_agreement_as_canceled_by_requester(&agreement.id) - .await - { - tracing::error!( - error = %err, - agreement_id = %agreement.id, - "Failed to mark orphan agreement as canceled in local DB" - ); - continue; - } + let started = crate::cancel_dispatch::start_cancel( + registry, + chain_client, + &agreement, + crate::cancel_dispatch::CancelReason::NotWanted, + config, + ) + .await; + log_orphan_cancel(&agreement, started); + } +} - // Orphan (previously accepted) agreement canceled on-chain by dipper. - // Record the cancel audit; `sweep_pending_terminated_events` emits the - // `terminated` durably (the row is `AcceptedOnChain` -> terminal, so it is - // sweep-eligible). - let manager = config.recurring_agreement_manager().to_string(); - if let Err(err) = registry - .record_cancel_audit( - &agreement.id, - dipper_core::time::now_secs(), - &manager, - on_chain_cancel_tx.as_deref(), - ) - .await - { - tracing::warn!( - agreement_id = %agreement.id, - error = %err, - "failed to record cancel audit; terminated event may emit with fallback fields" - ); - } +fn log_orphan_cancel( + agreement: &IndexingAgreement, + started: crate::registry::Result, +) { + match started { + Ok(started) => tracing::info!( + agreement_id = %agreement.id, + ?started, + reason = "request_canceled", + "Cancelling orphan agreement" + ), + Err(crate::registry::Error::NoRecordsUpdated) => tracing::debug!( + agreement_id = %agreement.id, + "Orphan agreement already ended or being cancelled" + ), + Err(err) => tracing::warn!( + error = %err, + agreement_id = %agreement.id, + "Failed to mark orphan agreement cancelling; will retry next sweep" + ), } } @@ -1533,7 +1559,7 @@ where /// confirmed send. /// /// Eligibility (`get_agreements_pending_accepted_emission`) requires -/// `accepted_at IS NOT NULL` (so pre-feature rows are never backfilled) but is +/// `accepted_at IS NOT NULL` (only rows whose accept was recorded) but is /// NOT gated on current status: an agreement accepted and then cancelled in a /// single snapshot is already terminal yet must still emit its `accepted` (which /// is why this sweep runs before the terminated sweep). @@ -1765,7 +1791,6 @@ mod tests { use dipper_core::ids::{IndexingAgreementId, IndexingRequestId}; use thegraph_core::{DeploymentId, IndexerId, alloy::primitives::ChainId}; use time::OffsetDateTime; - use url::Url; use super::{super::chain_events::AgreementState, *}; use crate::registry::{ @@ -1799,29 +1824,7 @@ mod tests { } fn test_agreement_conf() -> std::sync::Arc { - std::sync::Arc::new(crate::config::IndexingAgreementConfig { - data_service: thegraph_core::alloy::primitives::Address::ZERO, - recurring_collector: thegraph_core::alloy::primitives::Address::ZERO, - recurring_agreement_manager: thegraph_core::alloy::primitives::Address::ZERO, - max_agreement_grt_per_30_days: 0.0, - max_seconds_per_collection: 0, - min_seconds_per_collection: 0, - duration_seconds: 0, - deadline_seconds: 0, - max_grt_per_30_days: std::collections::BTreeMap::new(), - max_grt_per_billion_entities_per_30_days: 0.0, - declined_indexer_lookback_days: 0, - price_rejection_lookback_days: 0, - transient_rejection_lookback_minutes: 0, - uncertain_rejection_lookback_days: 0, - unresponsive_indexer_lookback_days: 0, - mass_unresponsive_trip_fraction: 0.5, - mass_unresponsive_reset_fraction: 0.25, - dips_accepting_snapshot_max_age_hours: 48, - dips_accepting_cache_ttl_seconds: 300, - max_in_flight_offers_per_indexer: None, - max_in_flight_offers_total: None, - }) + std::sync::Arc::new(crate::config::IndexingAgreementConfig::for_tests()) } // Deterministic audit values used by `make_snapshot` so emit tests can assert @@ -1871,10 +1874,16 @@ mod tests { agreements: std::collections::HashMap, marked_accepted_on_chain: Vec, marked_canceled_by_requester: Vec, + marked_cancelling: Vec, + reopened: Vec, marked_canceled_by_indexer: Vec, /// Ids passed to `record_cancel_audit` -- the signal a cancel path drives /// the terminated event (the sweep emits from this audit). recorded_cancel_audit: Vec, + /// Every audit write in order, as ("cancel" | "accept" | "accept and cancel", id). + audit_writes: Vec<(&'static str, IndexingAgreementId)>, + /// When true, writes that record a cancel fail. + fail_cancel_audit: bool, pending_cancellations: std::collections::HashMap< IndexingAgreementId, Vec, @@ -1941,6 +1950,14 @@ mod tests { self.state.lock().unwrap().canceled_request_ids.insert(id); } + /// Set when the offer could last be accepted, as seconds from now. + fn set_agreement_deadline_from_now(&self, agreement_id: IndexingAgreementId, secs: i64) { + if let Some(a) = self.state.lock().unwrap().agreements.get_mut(&agreement_id) { + let deadline = OffsetDateTime::now_utc().unix_timestamp() + secs; + a.terms.deadline = u64::try_from(deadline).unwrap(); + } + } + fn set_agreement_request_id( &self, agreement_id: IndexingAgreementId, @@ -1959,6 +1976,14 @@ mod tests { .contains(id) } + fn was_marked_cancelling(&self, id: &IndexingAgreementId) -> bool { + self.state.lock().unwrap().marked_cancelling.contains(id) + } + + fn was_reopened(&self, id: &IndexingAgreementId) -> bool { + self.state.lock().unwrap().reopened.contains(id) + } + fn was_marked_canceled_by_requester(&self, id: &IndexingAgreementId) -> bool { self.state .lock() @@ -1975,6 +2000,10 @@ mod tests { .contains(id) } + fn audit_writes(&self) -> Vec<(&'static str, IndexingAgreementId)> { + self.state.lock().unwrap().audit_writes.clone() + } + fn was_cancel_audit_recorded(&self, id: &IndexingAgreementId) -> bool { self.state .lock() @@ -2017,7 +2046,7 @@ mod tests { } #[async_trait::async_trait] - impl AgreementRegistry for MockRegistry { + impl crate::registry::StubAgreementRegistry for MockRegistry { async fn get_indexing_agreement_by_id( &self, id: &IndexingAgreementId, @@ -2130,18 +2159,77 @@ mod tests { Ok(()) } + async fn mark_indexing_agreement_as_cancelling( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()> { + let mut state = self.state.lock().unwrap(); + if state.fail_cancel_for.contains(id) { + return Err(crate::registry::Error::BackendError( + dipper_pgregistry::Error::DbError(sqlx::Error::Protocol( + "simulated transient failure".into(), + )), + )); + } + state.marked_cancelling.push(*id); + Ok(()) + } + + async fn reopen_indexing_agreement_cancel( + &self, + id: &IndexingAgreementId, + _seen_live: bool, + ) -> RegistryResult<()> { + self.state.lock().unwrap().reopened.push(*id); + Ok(()) + } + async fn record_cancel_audit( &self, agreement_id: &IndexingAgreementId, _canceled_at: u64, _canceled_by: &str, _canceled_tx: Option<&str>, + ) -> RegistryResult<()> { + let mut state = self.state.lock().unwrap(); + if state.fail_cancel_audit { + return Err(crate::registry::Error::NoRecordsUpdated); + } + state.recorded_cancel_audit.push(*agreement_id); + state.audit_writes.push(("cancel", *agreement_id)); + Ok(()) + } + + async fn record_accept_and_cancel_audit( + &self, + agreement_id: &IndexingAgreementId, + _accepted_at: u64, + _accepted_tx: &str, + _canceled_at: u64, + _canceled_by: &str, + _canceled_tx: Option<&str>, + ) -> RegistryResult<()> { + let mut state = self.state.lock().unwrap(); + if state.fail_cancel_audit { + return Err(crate::registry::Error::NoRecordsUpdated); + } + state + .audit_writes + .push(("accept and cancel", *agreement_id)); + Ok(()) + } + + async fn record_accepted_audit( + &self, + agreement_id: &IndexingAgreementId, + _accepted_at: u64, + _accepted_tx: &str, ) -> RegistryResult<()> { self.state .lock() .unwrap() - .recorded_cancel_audit - .push(*agreement_id); + .audit_writes + .push(("accept", *agreement_id)); Ok(()) } @@ -2187,12 +2275,16 @@ mod tests { Some( IndexingAgreementStatus::Created | IndexingAgreementStatus::AcceptedOnChain - | IndexingAgreementStatus::Rejected, + | IndexingAgreementStatus::Rejected + | IndexingAgreementStatus::Cancelling, ), ), Some(crate::registry::CancelKind::ByIndexer) => matches!( effective_status_for_cancel, - Some(IndexingAgreementStatus::AcceptedOnChain), + Some( + IndexingAgreementStatus::AcceptedOnChain + | IndexingAgreementStatus::Cancelling, + ), ), None => false, }; @@ -2239,9 +2331,13 @@ mod tests { } let mut outcomes = std::collections::HashMap::with_capacity(items.len()); for item in items { - let outcome = self - .apply_reconciliation(&item.agreement_id, item.apply_accept, item.cancel) - .await?; + let outcome = crate::registry::StubAgreementRegistry::apply_reconciliation( + self, + &item.agreement_id, + item.apply_accept, + item.cancel, + ) + .await?; outcomes.insert(item.agreement_id, outcome); } Ok(outcomes) @@ -2346,13 +2442,6 @@ mod tests { Ok((per_indexer, global)) } - async fn mark_indexing_agreement_as_abandoned( - &self, - _id: &IndexingAgreementId, - ) -> RegistryResult { - Err(crate::registry::Error::NoRecordsUpdated) - } - async fn get_agreement_fee_rates(&self) -> RegistryResult> { Ok(vec![]) } @@ -2436,36 +2525,32 @@ mod tests { } } - // Mock worker queue + /// Minimal `ChainClient` mock for chain_listener tests. Records every on-chain cancel + /// attempt. Every agreement reads as not live, so nothing is sent, unless + /// `live_until_cancelled` is set. #[derive(Clone, Default)] - struct MockWorkerQueue { - cancel_jobs: Arc>>, + struct MockChainClient { + cancels: Arc>>, + /// When set, each cancel records whether its agreement was already marked. + registry: Option, + marked_at_cancel: Arc>>, + fail_cancels: bool, + /// When set, every agreement reads as live until a cancel is sent for it. + live_until_cancelled: bool, } - impl MockWorkerQueue { - fn was_cancellation_queued(&self, id: &IndexingAgreementId) -> bool { - self.cancel_jobs.lock().unwrap().contains(id) + /// A chain on which every agreement is live until a cancel is sent for it. + fn live_chain() -> MockChainClient { + MockChainClient { + live_until_cancelled: true, + ..MockChainClient::default() } } - /// Minimal `ChainClient` mock for chain_listener tests. Records every - /// on-chain cancel attempt. Tests can mark specific agreements as - /// already-canceled-on-chain (cancel returns `Ok(None)`); unmarked - /// agreements get a successful `Ok(Some(zero))`. - #[derive(Clone, Default)] - struct MockChainClient { - cancels: Arc>>, - already_canceled: Arc>>, - } - impl MockChainClient { fn was_on_chain_cancel_attempted(&self, id: &IndexingAgreementId) -> bool { self.cancels.lock().unwrap().contains(id.as_bytes()) } - - fn mark_already_canceled_on_chain(&self, id: &IndexingAgreementId) { - self.already_canceled.lock().unwrap().push(*id.as_bytes()); - } } #[async_trait::async_trait] @@ -2500,9 +2585,19 @@ mod tests { Option, crate::chain_client::ChainClientError, > { + if self.fail_cancels { + return Err(crate::chain_client::ChainClientError::RpcError( + anyhow::anyhow!("rpc down"), + )); + } // Record manager-routed cancels through the same recorder so the // existing assertions hold. self.cancels.lock().unwrap().push(*agreement_id); + if let Some(registry) = &self.registry { + let id = IndexingAgreementId::from_bytes(*agreement_id); + let was_marked = registry.was_marked_cancelling(&id); + self.marked_at_cancel.lock().unwrap().push(was_marked); + } // A manager-routed cancel has no "already canceled" result: the // contract silently no-ops a stale cancel and the tx still succeeds, // so the real cancel_via_manager never returns Ok(None). @@ -2530,60 +2625,16 @@ mod tests { Ok(None) } - async fn agreement_still_active( + async fn agreement_on_chain( &self, - _agreement_id: &[u8; 16], - ) -> Result { + agreement_id: &[u8; 16], + ) -> Result + { // Cancel dispatch always reads back after a mined cancel; reporting // not-active here means "cancel confirmed", which these tests expect. - Ok(false) - } - } - - #[async_trait::async_trait] - impl crate::worker::service::WorkerQueue for MockWorkerQueue { - async fn send_indexing_agreement_proposal( - &self, - _candidate_url: Url, - _agreement_id: IndexingAgreementId, - _indexing_request_id: IndexingRequestId, - _deployment_id: DeploymentId, - _deployment_chain_id: ChainId, - _priority: JobPriority, - ) -> anyhow::Result { - Ok(dipper_pgmq::JobId::default()) - } - - async fn reassess_indexing_request( - &self, - _indexing_request_id: IndexingRequestId, - _deployment_id: DeploymentId, - _deployment_chain_id: ChainId, - _num_candidates: usize, - _priority: JobPriority, - ) -> anyhow::Result { - Ok(dipper_pgmq::JobId::default()) - } - - async fn cancel_rejected_agreement_on_chain( - &self, - agreement_id: IndexingAgreementId, - _priority: JobPriority, - ) -> anyhow::Result { - self.cancel_jobs.lock().unwrap().push(agreement_id); - Ok(dipper_pgmq::JobId::default()) - } - - async fn submit_offer( - &self, - _agreement_id: IndexingAgreementId, - _indexing_request_id: IndexingRequestId, - _indexer_url: Url, - _deployment_id: DeploymentId, - _deployment_chain_id: ChainId, - _priority: JobPriority, - ) -> anyhow::Result { - Ok(dipper_pgmq::JobId::default()) + Ok(crate::chain_client::AgreementOnChain::live_if( + self.live_until_cancelled && !self.cancels.lock().unwrap().contains(agreement_id), + )) } } @@ -2593,7 +2644,6 @@ mod tests { async fn test_reconcile_transitions_created_to_accepted() { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::Created); @@ -2602,7 +2652,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2610,7 +2659,131 @@ mod tests { assert!(result.is_ok()); assert!(registry.was_marked_accepted_on_chain(&agreement_id)); - assert!(!worker_queue.was_cancellation_queued(&agreement_id)); + assert!(!registry.was_reopened(&agreement_id)); + } + + #[tokio::test] + async fn reconcile_records_the_accept_of_an_agreement_being_cancelled() { + // It stays cancelling, and the recorded accept lets its end be announced. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::Cancelling); + + let snapshot = make_snapshot(agreement_id, AgreementState::Accepted, Address::ZERO); + reconcile_agreement( + &snapshot, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await + .expect("reconcile ok"); + + assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); + assert_eq!(registry.audit_writes(), vec![("accept", agreement_id)]); + assert!(!registry.was_reopened(&agreement_id)); + } + + #[tokio::test] + async fn reconcile_records_no_accept_for_a_withdrawn_offer() { + // The subgraph reports a withdrawn offer as cancelled by the payer with no accept + // time; recording it would announce an agreement that was never live. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::Cancelling); + let mut snapshot = + make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); + snapshot.accepted_at = 0; + + reconcile_agreement( + &snapshot, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await + .expect("reconcile ok"); + + assert!(registry.was_marked_canceled_by_requester(&agreement_id)); + assert!( + registry + .audit_writes() + .iter() + .all(|(kind, _)| *kind != "accept") + ); + } + + #[tokio::test] + async fn reconcile_does_not_accept_an_offer_withdrawn_before_anyone_accepted_it() { + // Announcing it would send accepted and terminated events for an agreement that + // was never live; an expired one stays expired. + for status in [ + IndexingAgreementStatus::Created, + IndexingAgreementStatus::Expired, + ] { + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, status); + let mut snapshot = + make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); + snapshot.accepted_at = 0; + + reconcile_agreement( + &snapshot, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await + .expect("reconcile ok"); + + assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); + assert_eq!( + registry.was_marked_canceled_by_requester(&agreement_id), + status == IndexingAgreementStatus::Created + ); + assert!( + registry + .audit_writes() + .iter() + .all(|(kind, _)| *kind != "accept") + ); + } + } + + #[tokio::test] + async fn reconcile_marks_an_agreement_being_cancelled_once_the_chain_shows_it_ended() { + for (state, ended_by_dipper) in [ + (AgreementState::CanceledByPayer, true), + (AgreementState::CanceledByServiceProvider, false), + ] { + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::Cancelling); + + let snapshot = make_snapshot(agreement_id, state, Address::ZERO); + reconcile_agreement( + &snapshot, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await + .expect("reconcile ok"); + + assert_eq!( + registry.was_marked_canceled_by_requester(&agreement_id), + ended_by_dipper + ); + assert_eq!( + registry.was_marked_canceled_by_indexer(&agreement_id), + !ended_by_dipper + ); + } } #[tokio::test] @@ -2619,7 +2792,6 @@ mod tests { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let events = CapturingEventsProducer::new(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::Created); @@ -2628,7 +2800,6 @@ mod tests { reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2651,7 +2822,6 @@ mod tests { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let events = CapturingEventsProducer::new(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::AcceptedOnChain); @@ -2668,7 +2838,6 @@ mod tests { reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2686,10 +2855,12 @@ mod tests { } #[tokio::test] - async fn test_reconcile_queues_cancellation_for_rejected() { + async fn test_reconcile_reopens_the_cancel_of_a_rejected_agreement_live_on_chain() { let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); + let chain_client = MockChainClient { + live_until_cancelled: true, + ..MockChainClient::default() + }; let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::Rejected); @@ -2698,7 +2869,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2706,7 +2876,7 @@ mod tests { assert!(result.is_ok()); assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); - assert!(worker_queue.was_cancellation_queued(&agreement_id)); + assert!(registry.was_reopened(&agreement_id)); } #[tokio::test] @@ -2718,7 +2888,6 @@ mod tests { // the canceler address. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let signer_address: Address = "0xaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" .parse() @@ -2734,14 +2903,13 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) .await; assert!(result.is_ok()); - assert!(!worker_queue.was_cancellation_queued(&agreement_id)); + assert!(!registry.was_reopened(&agreement_id)); assert!(registry.was_marked_canceled_by_requester(&agreement_id)); assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); } @@ -2750,7 +2918,6 @@ mod tests { async fn test_reconcile_ignores_unknown_agreement() { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); // Don't add the agreement to the registry @@ -2758,7 +2925,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2766,14 +2932,13 @@ mod tests { assert!(result.is_ok()); assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); - assert!(!worker_queue.was_cancellation_queued(&agreement_id)); + assert!(!registry.was_reopened(&agreement_id)); } #[tokio::test] async fn test_reconcile_recovers_expired_agreement() { let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); + let chain_client = live_chain(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let old_agreement_id = IndexingAgreementId::from_bytes(rand::random()); @@ -2785,7 +2950,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2801,7 +2965,6 @@ mod tests { async fn test_reconcile_marks_canceled_by_indexer() { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let indexer_address: Address = "0x1234567890123456789012345678901234567890" .parse() @@ -2817,7 +2980,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2832,7 +2994,6 @@ mod tests { async fn test_reconcile_marks_canceled_by_requester() { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let signer_address: Address = "0xaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" .parse() @@ -2848,7 +3009,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2866,7 +3026,6 @@ mod tests { // the state: CanceledByPayer -> ByRequester, the only kind allowed here. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); // A payer address deliberately distinct from dipper's signer key. let payer_address: Address = "0xcccccccccccccccccccccccccccccccccccccccc" @@ -2879,7 +3038,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2890,14 +3048,206 @@ mod tests { assert!(!registry.was_marked_canceled_by_indexer(&agreement_id)); assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); // Already canceled on-chain: must not queue a fresh cancel job. - assert!(!worker_queue.was_cancellation_queued(&agreement_id)); + assert!(!registry.was_reopened(&agreement_id)); + } + + #[tokio::test] + async fn test_reconcile_reopens_the_cancel_of_a_cancelled_agreement_live_on_chain() { + // Dipper cancelled the agreement locally, but the indexer accepted its offer + // (for example one that landed after dipper's cancel). Nothing else would end + // it, so it goes back to cancelling for the cancel retry to end. + let registry = MockRegistry::new(); + let chain_client = MockChainClient { + live_until_cancelled: true, + ..MockChainClient::default() + }; + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); + + let snapshot = make_snapshot(agreement_id, AgreementState::Accepted, Address::ZERO); + let result = reconcile_agreement( + &snapshot, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok()); + assert!(registry.was_reopened(&agreement_id)); + assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); + assert!( + registry.audit_writes().contains(&("accept", agreement_id)), + "its accept is recorded, so the cancel retry treats it as paying" + ); + } + + #[tokio::test] + async fn test_reconcile_leaves_a_cancelled_agreement_the_chain_shows_ended() { + // The subgraph can still report an agreement accepted for a few polls after + // dipper's cancel lands; reopening it would only bring it back to cancelling. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); + + let snapshot = make_snapshot(agreement_id, AgreementState::Accepted, Address::ZERO); + reconcile_agreement( + &snapshot, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await + .expect("reconcile ok"); + + assert!(!registry.was_reopened(&agreement_id)); + } + + #[tokio::test] + async fn test_reconcile_cancelled_agreement_already_cancelled_on_chain_queues_nothing() { + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); + + let snapshot = make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); + let result = reconcile_agreement( + &snapshot, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok()); + assert!(!registry.was_reopened(&agreement_id)); + } + + #[tokio::test] + async fn test_reconcile_records_accept_and_cancel_of_cancelled_agreement_that_went_live() { + // Dipper had marked the agreement cancelled, but it was accepted on-chain + // before being ended there. Recording both lets the accepted and terminated + // events go out, in 1 write, as the terminated sweep only waits for the accept. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); + // The listener lagged: dipper only cancelled it locally long after the offer's + // deadline, which once made this accept look too old to announce. + registry.set_agreement_deadline_from_now(agreement_id, -2 * 86_400); + + let snapshot = make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); + let result = reconcile_agreement( + &snapshot, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok()); + assert_eq!( + registry.audit_writes(), + vec![("accept and cancel", agreement_id)] + ); + } + + #[tokio::test] + async fn test_reconcile_records_the_end_of_an_abandoned_agreement() { + // Dipper can mark it ended without the transaction that ended it, and its terminated + // event waits for that transaction, which only this record supplies. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::AbandonedByIndexer); + + let snapshot = make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); + reconcile_agreement( + &snapshot, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await + .expect("reconcile ok"); + + assert_eq!( + registry.audit_writes(), + vec![("accept and cancel", agreement_id)] + ); + } + + #[tokio::test] + async fn test_reconcile_survives_a_failed_accept_and_cancel_record() { + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); + registry.state.lock().unwrap().fail_cancel_audit = true; + + let snapshot = make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); + let result = reconcile_agreement( + &snapshot, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok(), "a failed record must not fail the snapshot"); + assert!(registry.audit_writes().is_empty()); + } + + #[tokio::test] + async fn test_reconcile_records_nothing_for_cancelled_agreement_never_accepted() { + // A withdrawn offer was never accepted, so there is nothing to announce. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); + + let mut snapshot = + make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); + snapshot.accepted_at = 0; + let result = reconcile_agreement( + &snapshot, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok()); + assert!(registry.audit_writes().is_empty()); + } + + #[tokio::test] + async fn test_reconcile_records_nothing_while_cancelled_agreement_is_still_live() { + // Recording the accept now would let the terminated event go out before the + // agreement has actually ended on-chain. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); + + let snapshot = make_snapshot(agreement_id, AgreementState::Accepted, Address::ZERO); + let result = reconcile_agreement( + &snapshot, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok()); + assert!(registry.audit_writes().is_empty()); } #[tokio::test] async fn test_reconcile_ignores_already_canceled() { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByIndexer); @@ -2910,7 +3260,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2934,7 +3283,6 @@ mod tests { // rather than incrementing its `errors` counter. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let signer_address: Address = "0xaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" .parse() @@ -2950,7 +3298,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2969,8 +3316,7 @@ mod tests { // We should run the acceptance-side bookkeeping (pending cancellations) // AND mark the agreement as CanceledByRequester. let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); + let chain_client = live_chain(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let old_agreement_id = IndexingAgreementId::from_bytes(rand::random()); let signer_address: Address = "0xaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" @@ -2989,7 +3335,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -3009,7 +3354,7 @@ mod tests { #[tokio::test] async fn test_pending_cancellations_all_succeed_records_deleted() { let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); + let chain_client = live_chain(); let new_id = IndexingAgreementId::from_bytes(rand::random()); let old_id_1 = IndexingAgreementId::from_bytes(rand::random()); let old_id_2 = IndexingAgreementId::from_bytes(rand::random()); @@ -3045,7 +3390,7 @@ mod tests { // execute_pending no longer emits `terminated` directly; it records the // cancel audit and the chain_listener sweep announces it durably. let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); + let chain_client = live_chain(); let new_id = IndexingAgreementId::from_bytes(rand::random()); let old_id = IndexingAgreementId::from_bytes(rand::random()); @@ -3069,6 +3414,138 @@ mod tests { ); } + #[tokio::test] + async fn test_pending_cancellations_leave_a_never_accepted_agreement_cancelling() { + // Its offer could still land and be accepted until the deadline, so the cancel + // retry finishes it then. No cancel is recorded: one would win over the chain's + // own if the indexer accepted it after all, reporting an end before the accept. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let new_id = IndexingAgreementId::from_bytes(rand::random()); + let old_id = IndexingAgreementId::from_bytes(rand::random()); + + registry.add_agreement(new_id, IndexingAgreementStatus::AcceptedOnChain); + registry.add_agreement(old_id, IndexingAgreementStatus::Created); + registry.add_pending_cancellation(new_id, old_id); + + let result = execute_pending_cancellations( + &new_id, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok()); + assert!(registry.was_marked_cancelling(&old_id)); + assert!(!registry.was_marked_canceled_by_requester(&old_id)); + assert!(!registry.was_cancel_audit_recorded(&old_id)); + } + + /// Runs the pending cancellation of one old agreement in `status`, returning + /// whether it was already marked cancelling when its on-chain cancel went out. + async fn marked_at_pending_cancel(status: IndexingAgreementStatus) -> Vec { + let registry = MockRegistry::new(); + let chain_client = MockChainClient { + registry: Some(registry.clone()), + live_until_cancelled: true, + ..MockChainClient::default() + }; + let new_id = IndexingAgreementId::from_bytes(rand::random()); + let old_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(new_id, IndexingAgreementStatus::AcceptedOnChain); + registry.add_agreement(old_id, status); + registry.add_pending_cancellation(new_id, old_id); + + let result = execute_pending_cancellations( + &new_id, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok(), "got {result:?}"); + assert!(registry.was_pending_cancellation_deleted(&new_id, &old_id)); + chain_client.marked_at_cancel.lock().unwrap().clone() + } + + #[tokio::test] + async fn test_pending_cancellations_mark_an_unaccepted_agreement_before_its_cancel() { + // Its offer may be in flight. An offer landing after the cancel can't be + // withdrawn by it, so its job must find the agreement already marked. + let marked = marked_at_pending_cancel(IndexingAgreementStatus::Created).await; + assert_eq!(marked, vec![true]); + } + + #[tokio::test] + async fn test_pending_cancellations_cancel_an_expired_agreement_only_if_live() { + // A lagging listener can mark an agreement expired that was in fact accepted; one + // that really expired keeps its status, and no needless cancel is sent. + for live in [true, false] { + let registry = MockRegistry::new(); + let chain_client = MockChainClient { + live_until_cancelled: live, + ..MockChainClient::default() + }; + let new_id = IndexingAgreementId::from_bytes(rand::random()); + let old_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(new_id, IndexingAgreementStatus::AcceptedOnChain); + registry.add_agreement(old_id, IndexingAgreementStatus::Expired); + registry.add_pending_cancellation(new_id, old_id); + + let result = execute_pending_cancellations( + &new_id, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok(), "got {result:?}"); + assert_eq!(registry.was_marked_cancelling(&old_id), live); + assert_eq!(chain_client.was_on_chain_cancel_attempted(&old_id), live); + assert!(registry.was_pending_cancellation_deleted(&new_id, &old_id)); + } + } + + #[tokio::test] + async fn test_pending_cancellations_mark_an_accepted_agreement_before_its_cancel() { + // A failed cancel then leaves it cancelling, which the cancel retry picks up. + let marked = marked_at_pending_cancel(IndexingAgreementStatus::AcceptedOnChain).await; + assert_eq!(marked, vec![true]); + } + + #[tokio::test] + async fn test_pending_cancellations_leave_a_failed_cancel_to_the_cancel_retry() { + // Retried from here, a cancel that never works was resent on every sweep with no + // limit, for as long as the replacement stayed accepted. The cancel retry limits it. + let registry = MockRegistry::new(); + let chain_client = MockChainClient { + fail_cancels: true, + ..MockChainClient::default() + }; + let new_id = IndexingAgreementId::from_bytes(rand::random()); + let old_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(new_id, IndexingAgreementStatus::AcceptedOnChain); + registry.add_agreement(old_id, IndexingAgreementStatus::AcceptedOnChain); + registry.add_pending_cancellation(new_id, old_id); + + let result = execute_pending_cancellations( + &new_id, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok(), "got {result:?}"); + assert!(registry.was_marked_cancelling(&old_id)); + assert!(!registry.was_marked_canceled_by_requester(&old_id)); + assert!(!registry.was_cancel_audit_recorded(&old_id)); + assert!(registry.was_pending_cancellation_deleted(&new_id, &old_id)); + } + #[tokio::test] async fn test_pending_cancellations_failed_cancel_records_no_audit() { let registry = MockRegistry::new(); @@ -3089,9 +3566,10 @@ mod tests { ) .await; - // The chain cancel failed, the row is NOT marked canceled, and no cancel - // audit is recorded (so the sweep emits nothing). + // The row couldn't be marked, so no cancel went out, nothing is recorded (the + // sweep emits nothing), and the pending row stays for a retry. assert!(result.is_err()); + assert!(!chain_client.was_on_chain_cancel_attempted(&old_fail)); assert!(!registry.was_marked_canceled_by_requester(&old_fail)); assert!(!registry.was_cancel_audit_recorded(&old_fail)); } @@ -3099,7 +3577,7 @@ mod tests { #[tokio::test] async fn test_pending_cancellations_transient_failure_retains_record() { let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); + let chain_client = live_chain(); let new_id = IndexingAgreementId::from_bytes(rand::random()); let old_ok = IndexingAgreementId::from_bytes(rand::random()); let old_fail = IndexingAgreementId::from_bytes(rand::random()); @@ -3185,13 +3663,10 @@ mod tests { #[tokio::test] async fn test_pending_cancellations_already_canceled_on_chain_succeeds() { - // Crash-recovery edge case: the cancel tx confirmed on-chain on a - // prior pass, but dipper crashed before deleting the pending row. - // On the next sweep the chain call surfaces as Ok(None) (the - // SubgraphService contract reverts with IndexingAgreementNotActive; - // the chain client translates that into "already canceled"). The - // handler must still flip the local row to CanceledByRequester and - // delete the pending row, not loop forever. + // Crash-recovery edge case: the cancel landed on a prior pass, but dipper + // crashed before deleting the pending row. The chain shows nothing live, so no + // cancel is sent; the row is left cancelling for the cancel retry to confirm + // and the pending row is deleted rather than retried forever. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); let new_id = IndexingAgreementId::from_bytes(rand::random()); @@ -3200,7 +3675,6 @@ mod tests { registry.add_agreement(new_id, IndexingAgreementStatus::AcceptedOnChain); registry.add_agreement(old_id, IndexingAgreementStatus::AcceptedOnChain); registry.add_pending_cancellation(new_id, old_id); - chain_client.mark_already_canceled_on_chain(&old_id); let result = execute_pending_cancellations( &new_id, @@ -3214,8 +3688,8 @@ mod tests { result.is_ok(), "expected idempotent success, got {result:?}" ); - assert!(chain_client.was_on_chain_cancel_attempted(&old_id)); - assert!(registry.was_marked_canceled_by_requester(&old_id)); + assert!(!chain_client.was_on_chain_cancel_attempted(&old_id)); + assert!(registry.was_marked_cancelling(&old_id)); assert!(registry.was_pending_cancellation_deleted(&new_id, &old_id)); } @@ -3232,7 +3706,6 @@ mod tests { registry.add_agreement(new_id, IndexingAgreementStatus::AcceptedOnChain); registry.add_agreement(old_id, IndexingAgreementStatus::AcceptedOnChain); registry.add_pending_cancellation(new_id, old_id); - chain_client.mark_already_canceled_on_chain(&old_id); sweep_executable_pending_cancellations( ®istry, @@ -3241,7 +3714,7 @@ mod tests { ) .await; - assert!(registry.was_marked_canceled_by_requester(&old_id)); + assert!(registry.was_marked_cancelling(&old_id)); assert!(registry.was_pending_cancellation_deleted(&new_id, &old_id)); let remaining = registry .get_pending_cancellations_by_new_agreement(new_id) @@ -3260,7 +3733,7 @@ mod tests { // and the old agreement is still alive. The sweep must complete // the cancellation without needing another snapshot to arrive. let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); + let chain_client = live_chain(); let new_id = IndexingAgreementId::from_bytes(rand::random()); let old_id = IndexingAgreementId::from_bytes(rand::random()); @@ -3415,7 +3888,6 @@ mod tests { let ctx = Ctx { registry: registry.clone(), - worker_queue: MockWorkerQueue::default(), chain_client: MockChainClient::default(), event_source, config, @@ -3499,7 +3971,6 @@ mod tests { let ctx = Ctx { registry: registry.clone(), - worker_queue: MockWorkerQueue::default(), chain_client: MockChainClient::default(), event_source, config, @@ -3580,7 +4051,6 @@ mod tests { let ctx = Ctx { registry: registry.clone(), - worker_queue: MockWorkerQueue::default(), chain_client: MockChainClient::default(), event_source, config, @@ -3665,7 +4135,6 @@ mod tests { 10, false, ®istry, - &MockWorkerQueue::default(), &chain_client, &event_source, &mut rx_stop, @@ -3730,7 +4199,6 @@ mod tests { 10, false, ®istry, - &MockWorkerQueue::default(), &chain_client, &event_source, &mut rx_stop, @@ -3758,7 +4226,6 @@ mod tests { 10, false, ®istry, - &MockWorkerQueue::default(), &chain_client, &event_source, &mut rx_stop, @@ -3862,7 +4329,6 @@ mod tests { 10, false, ®istry, - &MockWorkerQueue::default(), &chain_client, &event_source, &mut rx_stop, @@ -3903,7 +4369,6 @@ mod tests { let ctx = Ctx { registry: MockRegistry::new(), - worker_queue: MockWorkerQueue::default(), chain_client: MockChainClient::default(), event_source: TimingEventSource { poll_times: poll_times.clone(), @@ -3942,7 +4407,7 @@ mod tests { // Canceled is the orphan signature: reassessment fired the chain // cancel and failed, then bailed out. The sweep must pick it up. let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); + let chain_client = live_chain(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let request_id = IndexingRequestId::new(); @@ -3963,12 +4428,33 @@ mod tests { ); } + #[tokio::test] + async fn test_orphan_sweep_leaves_a_failed_cancel_to_the_cancel_retry() { + // Retried by this sweep, a cancel that never works was resent with no limit. + let registry = MockRegistry::new(); + let chain_client = MockChainClient { + fail_cancels: true, + ..MockChainClient::default() + }; + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + let request_id = IndexingRequestId::new(); + registry.add_agreement(agreement_id, IndexingAgreementStatus::AcceptedOnChain); + registry.set_agreement_request_id(agreement_id, request_id); + registry.mark_request_canceled(request_id); + + sweep_orphan_canceled_agreements(®istry, &chain_client, test_agreement_conf().as_ref()) + .await; + + assert!(registry.was_marked_cancelling(&agreement_id)); + assert!(!registry.was_marked_canceled_by_requester(&agreement_id)); + } + #[tokio::test] async fn test_orphan_sweep_records_cancel_audit() { // The orphan sweep no longer emits `terminated` directly; it records the // cancel audit and `sweep_pending_terminated_events` announces it. let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); + let chain_client = live_chain(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let request_id = IndexingRequestId::new(); @@ -3988,8 +4474,8 @@ mod tests { #[tokio::test] async fn test_orphan_sweep_handles_already_canceled_on_chain() { - // Idempotency check: the chain reports the agreement is already - // canceled (Ok(None)). The sweep must still clean up the local row. + // The chain shows the agreement already ended, so no cancel is sent; it is + // left cancelling for the cancel retry to confirm who ended it. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); @@ -3998,13 +4484,13 @@ mod tests { registry.add_agreement(agreement_id, IndexingAgreementStatus::AcceptedOnChain); registry.set_agreement_request_id(agreement_id, request_id); registry.mark_request_canceled(request_id); - chain_client.mark_already_canceled_on_chain(&agreement_id); sweep_orphan_canceled_agreements(®istry, &chain_client, test_agreement_conf().as_ref()) .await; - assert!(chain_client.was_on_chain_cancel_attempted(&agreement_id)); - assert!(registry.was_marked_canceled_by_requester(&agreement_id)); + assert!(!chain_client.was_on_chain_cancel_attempted(&agreement_id)); + assert!(registry.was_marked_cancelling(&agreement_id)); + assert!(!registry.was_marked_canceled_by_requester(&agreement_id)); } #[tokio::test] diff --git a/bin/dipper-service/src/network/service/escrow_reconciler.rs b/bin/dipper-service/src/network/service/escrow_reconciler.rs index f24fbaf2..7151042b 100644 --- a/bin/dipper-service/src/network/service/escrow_reconciler.rs +++ b/bin/dipper-service/src/network/service/escrow_reconciler.rs @@ -540,6 +540,7 @@ mod tests { use thegraph_core::alloy::primitives::B256; use super::*; + use crate::chain_client::AgreementOnChain; const NOW: u64 = 1_800_000_000; @@ -692,10 +693,10 @@ mod tests { } Ok(Some(B256::ZERO)) } - async fn agreement_still_active( + async fn agreement_on_chain( &self, _agreement_id: &[u8; 16], - ) -> Result { + ) -> Result { unimplemented!() } async fn reconcile_agreement( diff --git a/bin/dipper-service/src/network/service/expiration.rs b/bin/dipper-service/src/network/service/expiration.rs index 14652149..12802fcf 100644 --- a/bin/dipper-service/src/network/service/expiration.rs +++ b/bin/dipper-service/src/network/service/expiration.rs @@ -538,13 +538,6 @@ mod tests { ) -> anyhow::Result { Ok(JobId::default()) } - async fn cancel_rejected_agreement_on_chain( - &self, - _agreement_id: IndexingAgreementId, - _priority: JobPriority, - ) -> anyhow::Result { - unimplemented!() - } async fn submit_offer( &self, _agreement_id: IndexingAgreementId, diff --git a/bin/dipper-service/src/network/service/liveness_checker.rs b/bin/dipper-service/src/network/service/liveness_checker.rs index 09716aba..a95e2682 100644 --- a/bin/dipper-service/src/network/service/liveness_checker.rs +++ b/bin/dipper-service/src/network/service/liveness_checker.rs @@ -43,7 +43,8 @@ use tokio::{sync::mpsc, time::MissedTickBehavior}; use url::Url; use crate::{ - chain_client::{ChainClient, ChainClientError}, + cancel_dispatch::{CancelReason, CancelStarted}, + chain_client::ChainClient, config::LivenessCheckerConfig, network::provider::NetworkProviderService, registry::{ @@ -480,16 +481,10 @@ async fn record_progress( } } -/// Cancel a stale agreement on-chain and queue reassessment. -/// -/// If the on-chain cancel fails, the DB is not updated and reassessment is not -/// queued, leaving the agreement in `AcceptedOnChain` for the next cycle to retry. +/// End a stale agreement and queue a reassessment to replace it once it can't be paid. It is +/// marked `Cancelling`, and as awaiting replacement, before any cancel is sent, so the cancel +/// retry finishes one that fails here and queues its replacement once it has ended. #[allow(clippy::too_many_arguments)] -#[expect( - clippy::cognitive_complexity, - clippy::too_many_lines, - reason = "predates this lint; fix when next touched" -)] async fn cancel_and_reassess( agreement: &IndexingAgreement, registry: &R, @@ -503,102 +498,108 @@ async fn cancel_and_reassess( W: WorkerQueue + Send + Sync, C: ChainClient + Send + Sync, { - // 1. Cancel on-chain (mode-aware dispatch) - let mut on_chain_cancel_tx: Option = None; - match crate::cancel_dispatch::cancel_agreement_on_chain(chain_client, agreement, agreement_conf) - .await + // Without a cancel that can ever be sent, replacing it would pay 2 indexers until an + // operator ends it, so it stays as it is, active, for one to deal with. + if crate::cancel_dispatch::cancel_hash(agreement).is_none() { + tracing::error!( + agreement_id = %agreement.id, + "cannot cancel stale agreement: missing terms_version_hash; leaving active for operator action" + ); + return; + } + let Some(started) = + start_abandoned_cancel(agreement, registry, chain_client, agreement_conf).await + else { + return; + }; + forget_replaced_agreement(agreement, registry).await; + if started == CancelStarted::MayBeLive { + tracing::info!( + agreement_id = %agreement.id, + "Stale agreement may still be paid; the cancel retry replaces it once it has ended" + ); + return; + } + replace_abandoned(agreement, registry, worker_queue, db_timeout, queue_timeout).await; +} + +/// Queue the replacement of an agreement whose indexer stopped serving it, and note it queued +/// so it isn't queued again. +pub(crate) async fn replace_abandoned( + agreement: &IndexingAgreement, + registry: &R, + worker_queue: &W, + db_timeout: Duration, + queue_timeout: Duration, +) where + R: AgreementRegistry + IndexingRequestRegistry + Sync, + W: WorkerQueue + Sync, +{ + if !queue_replacement(agreement, registry, worker_queue, db_timeout, queue_timeout).await { + return; + } + if let Err(err) = registry.mark_replacement_queued(&agreement.id).await { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to note an abandoned agreement's replacement queued; it may be queued again" + ); + } +} + +/// Mark a stale agreement abandoned and send its cancel; `None`, logged, when it isn't marked. +async fn start_abandoned_cancel( + agreement: &IndexingAgreement, + registry: &R, + chain_client: &C, + agreement_conf: &crate::config::IndexingAgreementConfig, +) -> Option +where + R: AgreementRegistry + Sync, + C: ChainClient, +{ + match crate::cancel_dispatch::start_cancel( + registry, + chain_client, + agreement, + CancelReason::Abandoned, + agreement_conf, + ) + .await { - Ok(Some(tx_hash)) => { - tracing::info!( - agreement_id = %agreement.id, - tx_hash = %tx_hash, - "canceled stale agreement on-chain" - ); - on_chain_cancel_tx = Some(tx_hash.to_string()); - } - Ok(None) => { + Ok(started) => { tracing::info!( agreement_id = %agreement.id, - "stale agreement already canceled on-chain; proceeding to mark abandoned" + ?started, + reason = "indexer_stale", + "Cancelling stale agreement" ); + Some(started) } - Err(err @ ChainClientError::MissingTermsVersionHash { .. }) => { - // Permanent per-agreement condition: the on-chain agreement is - // still live, so do NOT mark abandoned (that would hide a - // money-draining agreement). Surface for operator action. - tracing::error!( - agreement_id = %agreement.id, - error = %err, - "cannot cancel stale agreement: missing terms_version_hash; leaving active for operator action" - ); - return; - } - Err(ChainClientError::ConfigError(_)) => { - // Chain client disabled: still proceed to mark and reassess so the - // DB reflects the detected abandonment even without an on-chain tx. - tracing::warn!( + Err(crate::registry::Error::NoRecordsUpdated) => { + tracing::debug!( agreement_id = %agreement.id, - "chain client not configured, skipping on-chain cancellation" + "Stale agreement already ended or being cancelled" ); + None } Err(err) => { tracing::error!( agreement_id = %agreement.id, error = %err, - "failed to cancel stale agreement on-chain, will retry next cycle" + "failed to mark stale agreement cancelling, will retry next cycle" ); - return; + None } } +} - // 2. Mark as abandoned in DB - let abandoned = match tokio::time::timeout( - db_timeout, - registry.mark_indexing_agreement_as_abandoned(&agreement.id), - ) - .await - { - Ok(Ok(a)) => a, - Ok(Err(err)) => { - tracing::error!( - agreement_id = %agreement.id, - error = %err, - "failed to mark agreement as abandoned" - ); - return; - } - Err(_) => { - tracing::error!( - agreement_id = %agreement.id, - "timeout marking agreement as abandoned" - ); - return; - } - }; - - // The accepted agreement was canceled on-chain because the indexer went - // stale. Record the cancel audit so the chain_listener's `terminated` sweep - // announces it durably: this row is marked `AbandonedByIndexer` (terminal) - // and was accepted on-chain, so it is sweep-eligible. - let manager = agreement_conf.recurring_agreement_manager().to_string(); - if let Err(err) = registry - .record_cancel_audit( - &agreement.id, - dipper_core::time::now_secs(), - &manager, - on_chain_cancel_tx.as_deref(), - ) - .await - { - tracing::warn!( - agreement_id = %agreement.id, - error = %err, - "failed to record cancel audit; terminated event may emit with fallback fields" - ); - } - - // Clean up pending cancellations: if this abandoned agreement was a - // replacement, the old agreement it was replacing should stay active. +/// Drop the pending cancellations an abandoned agreement holds as a replacement, so the +/// agreement it was to replace stays active. +async fn forget_replaced_agreement( + agreement: &IndexingAgreement, + registry: &R, +) { if let Err(err) = registry .delete_pending_cancellations_by_new_agreement(agreement.id) .await @@ -609,47 +610,32 @@ async fn cancel_and_reassess( "failed to clean up pending cancellations for abandoned agreement" ); } +} - // 3. Fetch the indexing request for num_candidates - let request = match tokio::time::timeout( - db_timeout, - registry.get_indexing_request_by_id(&abandoned.indexing_request_id), - ) - .await - { - Ok(Ok(Some(r))) => r, - Ok(Ok(None)) => { - tracing::warn!( - agreement_id = %agreement.id, - indexing_request_id = %abandoned.indexing_request_id, - "indexing request not found for abandoned agreement" - ); - return; - } - Ok(Err(err)) => { - tracing::warn!( - agreement_id = %agreement.id, - error = %err, - "failed to fetch indexing request for abandoned agreement" - ); - return; - } - Err(_) => { - tracing::warn!( - agreement_id = %agreement.id, - "timeout fetching indexing request for abandoned agreement" - ); - return; - } +/// Queue a reassessment of an abandoned agreement's request, which replaces its indexer. True +/// once nothing is left to do: queued, or its request is gone. +async fn queue_replacement( + agreement: &IndexingAgreement, + registry: &R, + worker_queue: &W, + db_timeout: Duration, + queue_timeout: Duration, +) -> bool +where + R: IndexingRequestRegistry + Sync, + W: WorkerQueue + Sync, +{ + let request = match abandoned_request(agreement, registry, db_timeout).await { + Ok(Some(request)) => request, + Ok(None) => return true, + Err(()) => return false, }; - - // 4. Queue reassessment let push_result = tokio::time::timeout( queue_timeout, worker_queue.reassess_indexing_request( - abandoned.indexing_request_id, - abandoned.terms.metadata.subgraph_deployment_id, - abandoned.terms.metadata.chain_id, + agreement.indexing_request_id, + agreement.terms.metadata.subgraph_deployment_id, + agreement.terms.metadata.chain_id, request.num_candidates, // Background: abandonment remediation yields to interactive work. JobPriority::Background, @@ -661,9 +647,10 @@ async fn cancel_and_reassess( Ok(Ok(_job_id)) => { tracing::info!( agreement_id = %agreement.id, - indexing_request_id = %abandoned.indexing_request_id, + indexing_request_id = %agreement.indexing_request_id, "queued reassessment for abandoned agreement" ); + true } Ok(Err(err)) => { tracing::warn!( @@ -671,12 +658,54 @@ async fn cancel_and_reassess( error = %err, "failed to queue reassessment for abandoned agreement" ); + false } Err(_) => { tracing::warn!( agreement_id = %agreement.id, "timeout queuing reassessment for abandoned agreement" ); + false + } + } +} + +/// The request an abandoned agreement served, `None` when it is gone; `Err`, logged, when it +/// can't be read. +async fn abandoned_request( + agreement: &IndexingAgreement, + registry: &R, + db_timeout: Duration, +) -> Result, ()> { + match tokio::time::timeout( + db_timeout, + registry.get_indexing_request_by_id(&agreement.indexing_request_id), + ) + .await + { + Ok(Ok(Some(r))) => Ok(Some(r)), + Ok(Ok(None)) => { + tracing::warn!( + agreement_id = %agreement.id, + indexing_request_id = %agreement.indexing_request_id, + "indexing request not found for abandoned agreement" + ); + Ok(None) + } + Ok(Err(err)) => { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "failed to fetch indexing request for abandoned agreement" + ); + Err(()) + } + Err(_) => { + tracing::warn!( + agreement_id = %agreement.id, + "timeout fetching indexing request for abandoned agreement" + ); + Err(()) } } } @@ -834,7 +863,7 @@ mod tests { record_progress, tolerance_duration, }; use crate::{ - chain_client::{ChainClient, ChainClientError}, + chain_client::{AgreementOnChain, ChainClient, ChainClientError}, config::LivenessCheckerConfig, registry::{ AgreementFeeRate, IndexingAgreement, IndexingAgreementStatus, IndexingAgreementTerms, @@ -915,40 +944,34 @@ mod tests { #[derive(Clone, Default)] struct MockCalls { progress_updates: Arc>>, - abandoned: Arc>>, + /// Ids marked cancelling as abandoned, before any cancel is sent. + abandoning: Arc>>, + /// Ids marked ended once the chain confirmed dipper's cancel. + ended: Arc>>, reassessments: Arc>>, chain_cancels: Arc>>, /// Ids passed to `record_cancel_audit` -- the signal the handler drives /// the terminated event (the chain_listener sweep emits from this audit). cancel_audits: Arc>>, + /// Ids noted as having their replacement queued. + replacements_noted: Arc>>, } struct MockRegistry { calls: MockCalls, - mark_abandoned_result: Arc>>>, + already_ending: bool, get_request_result: Arc>>>>, } impl MockRegistry { fn new(calls: MockCalls, agreement: IndexingAgreement) -> Self { - let abandoned_agreement = { - let mut a = agreement.clone(); - a.status = IndexingAgreementStatus::AbandonedByIndexer; - a - }; let request = make_request(agreement.indexing_request_id, 2); Self { calls, - mark_abandoned_result: Arc::new(Mutex::new(Some(Ok(abandoned_agreement)))), + already_ending: false, get_request_result: Arc::new(Mutex::new(Some(Ok(Some(request))))), } } - - fn with_chain_error(calls: MockCalls, agreement: IndexingAgreement) -> Self { - let mut mock = Self::new(calls, agreement); - mock.mark_abandoned_result = Arc::new(Mutex::new(None)); - mock - } } #[async_trait] @@ -973,16 +996,28 @@ mod tests { .push((*id, block_height)); Ok(()) } - async fn mark_indexing_agreement_as_abandoned( + async fn mark_indexing_agreement_as_abandoning( &self, id: &IndexingAgreementId, - ) -> RegistryResult { - self.calls.abandoned.lock().unwrap().push(*id); - self.mark_abandoned_result - .lock() - .unwrap() - .take() - .expect("mark_abandoned called more than once") + ) -> RegistryResult<()> { + if self.already_ending { + return Err(crate::registry::Error::NoRecordsUpdated); + } + self.calls.abandoning.lock().unwrap().push(*id); + Ok(()) + } + + async fn mark_replacement_queued(&self, id: &IndexingAgreementId) -> RegistryResult<()> { + self.calls.replacements_noted.lock().unwrap().push(*id); + Ok(()) + } + + async fn mark_indexing_agreement_as_canceled_by_requester( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()> { + self.calls.ended.lock().unwrap().push(*id); + Ok(()) } async fn record_cancel_audit( @@ -1098,13 +1133,6 @@ mod tests { self.calls.reassessments.lock().unwrap().push(req_id); Ok(JobId::default()) } - async fn cancel_rejected_agreement_on_chain( - &self, - _agr_id: IndexingAgreementId, - _priority: JobPriority, - ) -> anyhow::Result { - unimplemented!() - } async fn submit_offer( &self, _agreement_id: IndexingAgreementId, @@ -1118,9 +1146,11 @@ mod tests { } } + /// An agreement live on-chain until a cancel that succeeds ends it. struct MockChainClient { calls: MockCalls, result: Result, + live: std::sync::atomic::AtomicBool, } impl MockChainClient { @@ -1128,13 +1158,7 @@ mod tests { Self { calls, result: Ok(B256::ZERO), - } - } - - fn config_error(calls: MockCalls) -> Self { - Self { - calls, - result: Err(ChainClientError::ConfigError("disabled".into())), + live: true.into(), } } @@ -1142,6 +1166,7 @@ mod tests { Self { calls, result: Err(ChainClientError::RpcError(anyhow::anyhow!("network error"))), + live: true.into(), } } } @@ -1174,12 +1199,9 @@ mod tests { // existing cancel-path assertions hold. self.calls.chain_cancels.lock().unwrap().push(*agreement_id); match &self.result { - Ok(hash) => Ok(Some(*hash)), - Err(ChainClientError::ConfigError(s)) => { - Err(ChainClientError::ConfigError(s.clone())) - } - Err(ChainClientError::RpcError(e)) => { - Err(ChainClientError::RpcError(anyhow::anyhow!("{e}"))) + Ok(hash) => { + self.live.store(false, std::sync::atomic::Ordering::SeqCst); + Ok(Some(*hash)) } Err(e) => Err(ChainClientError::RpcError(anyhow::anyhow!("{e}"))), } @@ -1201,13 +1223,13 @@ mod tests { unimplemented!() } - async fn agreement_still_active( + async fn agreement_on_chain( &self, _agreement_id: &[u8; 16], - ) -> Result { - // Cancel dispatch reads back after a mined cancel; reporting - // not-active means "cancel confirmed", which these tests expect. - Ok(false) + ) -> Result { + Ok(AgreementOnChain::live_if( + self.live.load(std::sync::atomic::Ordering::SeqCst), + )) } } @@ -1216,29 +1238,7 @@ mod tests { /// Default agreement config for the cancel-path tests. fn test_agreement_conf() -> crate::config::IndexingAgreementConfig { - crate::config::IndexingAgreementConfig { - data_service: thegraph_core::alloy::primitives::Address::ZERO, - recurring_collector: thegraph_core::alloy::primitives::Address::ZERO, - recurring_agreement_manager: thegraph_core::alloy::primitives::Address::ZERO, - max_agreement_grt_per_30_days: 0.0, - max_seconds_per_collection: 0, - min_seconds_per_collection: 0, - duration_seconds: 0, - deadline_seconds: 0, - max_grt_per_30_days: std::collections::BTreeMap::new(), - max_grt_per_billion_entities_per_30_days: 0.0, - declined_indexer_lookback_days: 0, - price_rejection_lookback_days: 0, - transient_rejection_lookback_minutes: 0, - uncertain_rejection_lookback_days: 0, - unresponsive_indexer_lookback_days: 0, - mass_unresponsive_trip_fraction: 0.5, - mass_unresponsive_reset_fraction: 0.25, - dips_accepting_snapshot_max_age_hours: 48, - dips_accepting_cache_ttl_seconds: 300, - max_in_flight_offers_per_indexer: None, - max_in_flight_offers_total: None, - } + crate::config::IndexingAgreementConfig::for_tests() } // ---- Pure function tests ---- @@ -1474,183 +1474,139 @@ mod tests { // ---- cancel_and_reassess behavior tests ---- - #[tokio::test] - async fn test_cancel_and_reassess_success() { - // Arrange + fn stale_agreement() -> IndexingAgreement { let dep: DeploymentId = "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" .parse() .unwrap(); - let indexer_id = IndexerId::from(Address::ZERO); let url: Url = "http://indexer.example.com/".parse().unwrap(); - let agreement = make_agreement( - indexer_id, + make_agreement( + IndexerId::from(Address::ZERO), url, dep, Some(100), Some(OffsetDateTime::now_utc()), - ); - let req_id = agreement.indexing_request_id; - let agr_id = agreement.id; + ) + } - let calls = MockCalls::default(); - let registry = MockRegistry::new(calls.clone(), agreement.clone()); + async fn end_stale( + agreement: &IndexingAgreement, + registry: &MockRegistry, + chain: &MockChainClient, + ) { let queue = MockWorkerQueue { - calls: calls.clone(), + calls: registry.calls.clone(), }; - let chain = MockChainClient::success(calls.clone()); - - // Act cancel_and_reassess( - &agreement, - ®istry, + agreement, + registry, &queue, - &chain, + chain, &test_agreement_conf(), DB_TIMEOUT, QUEUE_TIMEOUT, ) .await; - - // Assert - assert_eq!( - calls.chain_cancels.lock().unwrap().as_slice(), - &[agreement.id.into_bytes()] - ); - assert_eq!(calls.abandoned.lock().unwrap().as_slice(), &[agr_id]); - assert_eq!(calls.reassessments.lock().unwrap().as_slice(), &[req_id]); } #[tokio::test] - async fn cancel_and_reassess_records_cancel_audit() { - // The stale-agreement cancel no longer emits `terminated` directly: it - // records the cancel audit and the chain_listener sweep announces it. - let dep: DeploymentId = "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" - .parse() - .unwrap(); - let indexer_id = IndexerId::from(Address::ZERO); - let url: Url = "http://indexer.example.com/".parse().unwrap(); - let agreement = make_agreement( - indexer_id, - url, - dep, - Some(100), - Some(OffsetDateTime::now_utc()), - ); - let agr_id = agreement.id; - + async fn ends_a_stale_agreement_through_cancelling_and_reassesses() { + let agreement = stale_agreement(); let calls = MockCalls::default(); let registry = MockRegistry::new(calls.clone(), agreement.clone()); - let queue = MockWorkerQueue { - calls: calls.clone(), - }; let chain = MockChainClient::success(calls.clone()); - cancel_and_reassess( - &agreement, - ®istry, - &queue, - &chain, - &test_agreement_conf(), - DB_TIMEOUT, - QUEUE_TIMEOUT, - ) - .await; + end_stale(&agreement, ®istry, &chain).await; + assert_eq!(calls.abandoning.lock().unwrap().as_slice(), &[agreement.id]); + assert_eq!( + calls.chain_cancels.lock().unwrap().as_slice(), + &[agreement.id.into_bytes()] + ); + assert_eq!(calls.ended.lock().unwrap().as_slice(), &[agreement.id]); assert_eq!( calls.cancel_audits.lock().unwrap().as_slice(), - &[agr_id], - "exactly one cancel audit recorded for the stale agreement" + &[agreement.id], + "its cancel is recorded, so the terminated sweep announces it" + ); + assert_eq!( + calls.reassessments.lock().unwrap().as_slice(), + &[agreement.indexing_request_id] + ); + assert_eq!( + calls.replacements_noted.lock().unwrap().as_slice(), + &[agreement.id], + "so the cancel retry doesn't queue it again" ); } #[tokio::test] - async fn test_cancel_and_reassess_config_error_proceeds() { - // Arrange: chain client disabled (ConfigError) → still mark abandoned and reassess - let dep: DeploymentId = "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" - .parse() - .unwrap(); - let indexer_id = IndexerId::from(Address::ZERO); - let url: Url = "http://indexer.example.com/".parse().unwrap(); - let agreement = make_agreement( - indexer_id, - url, - dep, - Some(100), - Some(OffsetDateTime::now_utc()), - ); - let req_id = agreement.indexing_request_id; - let agr_id = agreement.id; + async fn holds_back_the_replacement_while_a_failed_cancel_leaves_it_paid() { + // Replacing it now would pay both indexers until the retry lands the cancel, which then + // queues the replacement. + let agreement = stale_agreement(); + let calls = MockCalls::default(); + let registry = MockRegistry::new(calls.clone(), agreement.clone()); + let chain = MockChainClient::rpc_error(calls.clone()); + + end_stale(&agreement, ®istry, &chain).await; + + assert_eq!(calls.abandoning.lock().unwrap().as_slice(), &[agreement.id]); + assert!(calls.ended.lock().unwrap().is_empty()); + assert!(calls.cancel_audits.lock().unwrap().is_empty()); + assert!(calls.reassessments.lock().unwrap().is_empty()); + assert!(calls.replacements_noted.lock().unwrap().is_empty()); + } + #[tokio::test] + async fn replaces_at_once_a_stale_agreement_the_chain_shows_ended() { + let agreement = stale_agreement(); let calls = MockCalls::default(); let registry = MockRegistry::new(calls.clone(), agreement.clone()); - let queue = MockWorkerQueue { - calls: calls.clone(), - }; - let chain = MockChainClient::config_error(calls.clone()); + let chain = MockChainClient::rpc_error(calls.clone()); + chain.live.store(false, std::sync::atomic::Ordering::SeqCst); - // Act - cancel_and_reassess( - &agreement, - ®istry, - &queue, - &chain, - &test_agreement_conf(), - DB_TIMEOUT, - QUEUE_TIMEOUT, - ) - .await; + end_stale(&agreement, ®istry, &chain).await; - // Assert: no on-chain cancel (ConfigError is treated as disabled, not a real error) - // but DB mark and reassessment still happen + assert!(calls.chain_cancels.lock().unwrap().is_empty()); assert_eq!( - calls.chain_cancels.lock().unwrap().as_slice(), - &[agreement.id.into_bytes()] + calls.reassessments.lock().unwrap().as_slice(), + &[agreement.indexing_request_id] + ); + assert_eq!( + calls.replacements_noted.lock().unwrap().as_slice(), + &[agreement.id] ); - assert_eq!(calls.abandoned.lock().unwrap().as_slice(), &[agr_id]); - assert_eq!(calls.reassessments.lock().unwrap().as_slice(), &[req_id]); } #[tokio::test] - async fn test_cancel_and_reassess_chain_error_skips() { - // Arrange: transient RPC error → do nothing (retry next cycle) - let dep: DeploymentId = "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" - .parse() - .unwrap(); - let indexer_id = IndexerId::from(Address::ZERO); - let url: Url = "http://indexer.example.com/".parse().unwrap(); - let agreement = make_agreement( - indexer_id, - url, - dep, - Some(100), - Some(OffsetDateTime::now_utc()), - ); + async fn leaves_a_stale_agreement_it_can_never_cancel_for_an_operator() { + // Replacing it would pay both indexers, since its cancel can never be sent. + let mut agreement = stale_agreement(); + agreement.terms_version_hash = None; + let calls = MockCalls::default(); + let registry = MockRegistry::new(calls.clone(), agreement.clone()); + let chain = MockChainClient::success(calls.clone()); + end_stale(&agreement, ®istry, &chain).await; + + assert!(calls.abandoning.lock().unwrap().is_empty()); + assert!(calls.chain_cancels.lock().unwrap().is_empty()); + assert!(calls.reassessments.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn leaves_an_agreement_already_ending_alone() { + let agreement = stale_agreement(); let calls = MockCalls::default(); - let registry = MockRegistry::with_chain_error(calls.clone(), agreement.clone()); - let queue = MockWorkerQueue { - calls: calls.clone(), + let registry = MockRegistry { + already_ending: true, + ..MockRegistry::new(calls.clone(), agreement.clone()) }; - let chain = MockChainClient::rpc_error(calls.clone()); + let chain = MockChainClient::success(calls.clone()); - // Act - cancel_and_reassess( - &agreement, - ®istry, - &queue, - &chain, - &test_agreement_conf(), - DB_TIMEOUT, - QUEUE_TIMEOUT, - ) - .await; + end_stale(&agreement, ®istry, &chain).await; - // Assert: chain cancel attempted, but DB and queue untouched - assert_eq!( - calls.chain_cancels.lock().unwrap().as_slice(), - &[agreement.id.into_bytes()] - ); - assert!(calls.abandoned.lock().unwrap().is_empty()); + assert!(calls.chain_cancels.lock().unwrap().is_empty()); assert!(calls.reassessments.lock().unwrap().is_empty()); } diff --git a/bin/dipper-service/src/registry.rs b/bin/dipper-service/src/registry.rs index 82d9baf6..71470cda 100644 --- a/bin/dipper-service/src/registry.rs +++ b/bin/dipper-service/src/registry.rs @@ -23,8 +23,8 @@ pub use self::agreement_stub::StubAgreementRegistry; use self::result::Result as RegistryResult; pub use self::{ agreement::{ - AgreementFeeRate, AgreementRegistry, CancelKind, IndexingAgreement, NewAgreementParams, - ReconciliationAudit, ReconciliationItem, ReconciliationOutcome, + AgreementFeeRate, AgreementRegistry, CancelKind, CancellingAgreement, IndexingAgreement, + NewAgreementParams, ReconciliationAudit, ReconciliationItem, ReconciliationOutcome, Status as IndexingAgreementStatus, Terms as IndexingAgreementTerms, TermsMetadata as IndexingAgreementTermsMetadata, }, @@ -391,6 +391,65 @@ impl AgreementRegistry for RegistryProvider { .map_err(Into::into) } + async fn mark_indexing_agreement_as_cancelling( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()> { + self.inner + .mark_indexing_agreement_as_cancelling(id) + .await + .map_err(Into::into) + } + + async fn mark_indexing_agreement_as_abandoning( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()> { + self.inner + .mark_indexing_agreement_as_abandoning(id) + .await + .map_err(Into::into) + } + + async fn reopen_indexing_agreement_cancel( + &self, + id: &IndexingAgreementId, + seen_live: bool, + ) -> RegistryResult<()> { + self.inner + .reopen_indexing_agreement_cancel(id, seen_live) + .await + .map_err(Into::into) + } + + async fn get_cancelling_agreements( + &self, + batch_size: i64, + max_attempts: u32, + min_age_minutes: i32, + ) -> RegistryResult> { + Ok(self + .inner + .get_cancelling_agreements(batch_size, max_attempts, min_age_minutes) + .await? + .into_iter() + .map(CancellingAgreement::try_from) + .filter_map(filter_map_with_logging) + .collect()) + } + + async fn record_cancel_check( + &self, + id: &IndexingAgreementId, + failed_attempts: u32, + ended: Option, + ) -> RegistryResult { + self.inner + .record_cancel_check(id, failed_attempts, ended) + .await + .map_err(Into::into) + } + async fn apply_reconciliation( &self, id: &IndexingAgreementId, @@ -567,6 +626,28 @@ impl AgreementRegistry for RegistryProvider { Ok(()) } + async fn record_accept_and_cancel_audit( + &self, + agreement_id: &IndexingAgreementId, + accepted_at: u64, + accepted_tx: &str, + canceled_at: u64, + canceled_by: &str, + canceled_tx: Option<&str>, + ) -> RegistryResult<()> { + self.inner + .record_accept_and_cancel_audit( + agreement_id, + accepted_at, + accepted_tx, + canceled_at, + canceled_by, + canceled_tx, + ) + .await?; + Ok(()) + } + async fn get_expired_created_agreements( &self, batch_size: i64, @@ -631,6 +712,27 @@ impl AgreementRegistry for RegistryProvider { .collect()) } + async fn get_ended_agreements_awaiting_replacement( + &self, + batch_size: i64, + ) -> RegistryResult> { + Ok(self + .inner + .get_ended_agreements_awaiting_replacement(batch_size) + .await? + .into_iter() + .map(IndexingAgreement::try_from) + .filter_map(filter_map_with_logging) + .collect()) + } + + async fn mark_replacement_queued(&self, id: &IndexingAgreementId) -> RegistryResult<()> { + self.inner + .mark_replacement_queued(id) + .await + .map_err(Into::into) + } + async fn update_agreement_sync_progress( &self, id: &IndexingAgreementId, @@ -668,17 +770,6 @@ impl AgreementRegistry for RegistryProvider { .map_err(Into::into) } - async fn mark_indexing_agreement_as_abandoned( - &self, - id: &IndexingAgreementId, - ) -> RegistryResult { - let raw = self.inner.mark_indexing_agreement_as_abandoned(id).await?; - // The conversion only fails for Unknown status; since we just wrote - // AbandonedByIndexer, this cannot fail in practice. - IndexingAgreement::try_from(raw) - .map_err(|_| dipper_pgregistry::Error::NoRecordsUpdated.into()) - } - async fn get_agreement_fee_rates(&self) -> RegistryResult> { self.inner .get_agreement_fee_rates() diff --git a/bin/dipper-service/src/registry/agreement.rs b/bin/dipper-service/src/registry/agreement.rs index 32227b2c..7f8618b1 100644 --- a/bin/dipper-service/src/registry/agreement.rs +++ b/bin/dipper-service/src/registry/agreement.rs @@ -172,9 +172,10 @@ pub trait AgreementRegistry { indexer_ids: &[IndexerId], ) -> RegistryResult>>; - /// Get declined `CanceledByIndexer`/`Expired`/`Rejected` indexers grouped by - /// deployment. Each rejection reason gets its own exclusion window, as does an - /// expiry that never had an offer transaction; see the query for the details. + /// Get declined `CanceledByIndexer`/`Expired`/`Rejected` indexers, and those whose agreement + /// dipper ended `AbandonedByIndexer`, grouped by deployment. Each rejection reason gets its + /// own exclusion window, as does an expiry that never had an offer transaction; see the query + /// for the details. async fn get_declined_indexers_by_deployment( &self, default_lookback_days: i32, @@ -246,10 +247,9 @@ pub trait AgreementRegistry { id: &IndexingAgreementId, ) -> RegistryResult<()>; - /// Record the on-chain tx hash of the most recent `offer()` submission - /// for this agreement. Observability-only; does not transition status. - /// Called once per submit (including resubmits after a dropped tx) so - /// the DB reflects the live hash rather than an evicted one. + /// Record the hash of the latest `offer()` transaction, unless the agreement has ended, so + /// a resubmit replaces a dropped one. [`NoRecordUpdated`](Error::NoRecordsUpdated) when no + /// row took it. async fn update_offer_tx_hash( &self, id: &IndexingAgreementId, @@ -259,13 +259,60 @@ pub trait AgreementRegistry { /// Mark an indexing agreement as `CANCELED_BY_REQUESTER`. /// /// If there is no indexing agreement with the given ID, or if the agreement is not in the - /// `CREATED` or `ACCEPTED_ON_CHAIN` state, this method returns a - /// [`NoRecordUpdated`](Error::NoRecordsUpdated) error. + /// `CREATED`, `ACCEPTED_ON_CHAIN`, `REJECTED` or `CANCELLING` state, this method returns a + /// [`NoRecordUpdated`](Error::NoRecordsUpdated) error. One dipper was cancelling because its + /// indexer stopped serving it becomes `ABANDONED_BY_INDEXER` instead. async fn mark_indexing_agreement_as_canceled_by_requester( &self, id: &IndexingAgreementId, ) -> RegistryResult<()>; + /// Mark a `CREATED`, `ACCEPTED_ON_CHAIN`, `REJECTED` or `EXPIRED` agreement `CANCELLING`, + /// before its cancel is sent; [`NoRecordUpdated`](Error::NoRecordsUpdated) otherwise. + async fn mark_indexing_agreement_as_cancelling( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()>; + + /// Mark an `ACCEPTED_ON_CHAIN` agreement whose indexer stopped serving it `CANCELLING`, + /// before its cancel is sent, so it ends `ABANDONED_BY_INDEXER` once the chain confirms it; + /// [`NoRecordUpdated`](Error::NoRecordsUpdated) otherwise. + async fn mark_indexing_agreement_as_abandoning( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()>; + + /// Move a `CANCELED_BY_REQUESTER` or `REJECTED` agreement the chain shows live back to + /// `CANCELLING`, its cancel attempts reset; [`NoRecordUpdated`](Error::NoRecordsUpdated) + /// otherwise. `seen_live` when the chain was read and showed it live, which clears the end + /// on record and any announcement of it. + async fn reopen_indexing_agreement_cancel( + &self, + id: &IndexingAgreementId, + seen_live: bool, + ) -> RegistryResult<()>; + + /// `CANCELLING` agreements marked over `min_age_minutes` ago, those checked longest ago + /// first; one whose cancel has failed `max_attempts` times only once an hour. One that may be paying + /// an indexer (accepted, or past the offer deadline, which only an accepted one outlives) + /// counts as checked an hour earlier, so it goes first without holding the rest back. + async fn get_cancelling_agreements( + &self, + batch_size: i64, + max_attempts: u32, + min_age_minutes: i32, + ) -> RegistryResult>; + + /// Record a check of a `CANCELLING` agreement that left it cancelling, adding + /// `failed_attempts` to its failed cancels and returning the new count. `ended` says whether + /// the check found it no longer live on-chain, or `None` when the chain couldn't tell. + async fn record_cancel_check( + &self, + id: &IndexingAgreementId, + failed_attempts: u32, + ended: Option, + ) -> RegistryResult; + /// Apply a reconciliation-driven state transition atomically. /// /// Used by `chain_listener::reconcile_agreement` so the @@ -380,6 +427,21 @@ pub trait AgreementRegistry { Ok(()) } + /// Record an agreement's accept and its end together, in 1 write, so nothing reads one + /// without the other. Default no-op so mocks need not override. + #[allow(clippy::too_many_arguments)] + async fn record_accept_and_cancel_audit( + &self, + _agreement_id: &IndexingAgreementId, + _accepted_at: u64, + _accepted_tx: &str, + _canceled_at: u64, + _canceled_by: &str, + _canceled_tx: Option<&str>, + ) -> RegistryResult<()> { + Ok(()) + } + /// Record the cancel audit payload for a dipper-initiated cancel so the /// emission sweep can populate the `terminated` event fields. Default no-op /// so mocks need not override. @@ -450,6 +512,16 @@ pub trait AgreementRegistry { batch_size: i64, ) -> RegistryResult>; + /// Agreements whose indexer stopped serving them that have ended, longest ended first, + /// whose replacement is yet to be queued. + async fn get_ended_agreements_awaiting_replacement( + &self, + batch_size: i64, + ) -> RegistryResult>; + + /// Note that an agreement's replacement has been queued. + async fn mark_replacement_queued(&self, id: &IndexingAgreementId) -> RegistryResult<()>; + /// Update the sync progress for an agreement. /// /// Called when the liveness checker observes the block height has changed @@ -491,17 +563,6 @@ pub trait AgreementRegistry { .map(|m| !m.is_empty()) } - /// Mark an indexing agreement as `ABANDONED_BY_INDEXER`. - /// - /// Transitions `AcceptedOnChain → AbandonedByIndexer`. Returns the full agreement - /// for use in the subsequent reassessment call. - /// Returns [`NoRecordsUpdated`](Error::NoRecordsUpdated) if the agreement doesn't - /// exist or isn't in `AcceptedOnChain` status. - async fn mark_indexing_agreement_as_abandoned( - &self, - id: &IndexingAgreementId, - ) -> RegistryResult; - /// Get per-agreement rate fields from active agreements. /// /// Returns base rate, entity rate, and deployment ID for each active @@ -522,6 +583,32 @@ pub struct AgreementFeeRate { pub tokens_per_entity_per_second: f64, } +/// An agreement dipper is still cancelling on-chain. +#[derive(Debug, Clone)] +pub struct CancellingAgreement { + pub agreement: IndexingAgreement, + /// Whether dipper saw it accepted on-chain, so its end is announced. + pub accepted_on_chain: bool, + /// When a check first found it no longer live on-chain, if one has. + pub ended_seen_at: Option, + /// Whether it is being cancelled because its indexer stopped serving it, so it ends + /// `ABANDONED_BY_INDEXER`. + pub abandoned: bool, +} + +impl TryFrom for CancellingAgreement { + type Error = anyhow::Error; + + fn try_from(value: dipper_pgregistry::CancellingAgreement) -> Result { + Ok(Self { + agreement: value.agreement.try_into()?, + accepted_on_chain: value.accepted_on_chain, + ended_seen_at: value.ended_seen_at, + abandoned: value.abandoned, + }) + } +} + /// An Indexing Agreement represents the contract between the DIPs Gateway (Dipper) and the indexer /// to index the data. /// @@ -684,11 +771,16 @@ pub enum Status { /// The liveness checker detected no indexing progress within the tolerance window. /// - /// Dipper canceled the agreement via `cancelIndexingAgreementByPayer` and will - /// trigger reassignment to find a replacement indexer. + /// Dipper cancelled the agreement on-chain, passing through `Cancelling` until the chain + /// confirmed it, and triggered reassignment to find a replacement indexer. /// /// This is a terminal state. AbandonedByIndexer, + + /// Dipper decided to end the agreement and is cancelling it on-chain, where it may + /// still be live. It becomes `CanceledByRequester`, or `AbandonedByIndexer` when its + /// indexer stopped serving it, announced as ended, only once the chain confirms the end. + Cancelling, } impl std::fmt::Display for Status { @@ -702,6 +794,7 @@ impl std::fmt::Display for Status { Status::AcceptedOnChain => "ACCEPTED_ON_CHAIN", Status::Rejected => "REJECTED", Status::AbandonedByIndexer => "ABANDONED_BY_INDEXER", + Status::Cancelling => "CANCELLING", }; f.write_str(status) } @@ -733,6 +826,7 @@ impl TryFrom for IndexingAgreement { dipper_pgregistry::IndexingAgreementStatus::AbandonedByIndexer => { Status::AbandonedByIndexer } + dipper_pgregistry::IndexingAgreementStatus::Cancelling => Status::Cancelling, _ => { return Err(anyhow::anyhow!("Invalid status: {:?}", value.status)); } diff --git a/bin/dipper-service/src/registry/agreement_stub.rs b/bin/dipper-service/src/registry/agreement_stub.rs index ec9bfd3b..dd2950b4 100644 --- a/bin/dipper-service/src/registry/agreement_stub.rs +++ b/bin/dipper-service/src/registry/agreement_stub.rs @@ -9,9 +9,9 @@ use thegraph_core::{DeploymentId, IndexerId, alloy::primitives::ChainId}; use super::{ agreement::{ - AgreementFeeRate, AgreementRegistry, CancelKind, IndexingAgreement, NewAgreementParams, - PendingAcceptedEvent, PendingExpiredEvent, PendingTerminatedEvent, ReconciliationItem, - ReconciliationOutcome, + AgreementFeeRate, AgreementRegistry, CancelKind, CancellingAgreement, IndexingAgreement, + NewAgreementParams, PendingAcceptedEvent, PendingExpiredEvent, PendingTerminatedEvent, + ReconciliationItem, ReconciliationOutcome, }, result::Result, }; @@ -133,6 +133,40 @@ pub trait StubAgreementRegistry: Send + Sync { unimplemented!("mark_indexing_agreement_as_canceled_by_requester") } + async fn mark_indexing_agreement_as_cancelling(&self, _id: &IndexingAgreementId) -> Result<()> { + unimplemented!("mark_indexing_agreement_as_cancelling") + } + + async fn mark_indexing_agreement_as_abandoning(&self, _id: &IndexingAgreementId) -> Result<()> { + unimplemented!("mark_indexing_agreement_as_abandoning") + } + + async fn reopen_indexing_agreement_cancel( + &self, + _id: &IndexingAgreementId, + _seen_live: bool, + ) -> Result<()> { + unimplemented!("reopen_indexing_agreement_cancel") + } + + async fn get_cancelling_agreements( + &self, + _batch_size: i64, + _max_attempts: u32, + _min_age_minutes: i32, + ) -> Result> { + Ok(Vec::new()) + } + + async fn record_cancel_check( + &self, + _id: &IndexingAgreementId, + failed_attempts: u32, + _ended: Option, + ) -> Result { + Ok(failed_attempts) + } + async fn apply_reconciliation( &self, _id: &IndexingAgreementId, @@ -191,6 +225,17 @@ pub trait StubAgreementRegistry: Send + Sync { unimplemented!("get_agreements_pending_chain_cancel") } + async fn get_ended_agreements_awaiting_replacement( + &self, + _batch_size: i64, + ) -> Result> { + unimplemented!("get_ended_agreements_awaiting_replacement") + } + + async fn mark_replacement_queued(&self, _id: &IndexingAgreementId) -> Result<()> { + unimplemented!("mark_replacement_queued") + } + async fn update_agreement_sync_progress( &self, _id: &IndexingAgreementId, @@ -215,13 +260,6 @@ pub trait StubAgreementRegistry: Send + Sync { .map(|m| !m.is_empty()) } - async fn mark_indexing_agreement_as_abandoned( - &self, - _id: &IndexingAgreementId, - ) -> Result { - unimplemented!("mark_indexing_agreement_as_abandoned") - } - async fn get_agreement_fee_rates(&self) -> Result> { unimplemented!("get_agreement_fee_rates") } @@ -289,6 +327,19 @@ pub trait StubAgreementRegistry: Send + Sync { ) -> Result<()> { Ok(()) } + + #[allow(clippy::too_many_arguments)] + async fn record_accept_and_cancel_audit( + &self, + _agreement_id: &IndexingAgreementId, + _accepted_at: u64, + _accepted_tx: &str, + _canceled_at: u64, + _canceled_by: &str, + _canceled_tx: Option<&str>, + ) -> Result<()> { + Ok(()) + } } // Every stub is a full AgreementRegistry: each method delegates to the stub @@ -415,6 +466,46 @@ impl AgreementRegistry for T { StubAgreementRegistry::mark_indexing_agreement_as_canceled_by_requester(self, id).await } + async fn mark_indexing_agreement_as_cancelling(&self, id: &IndexingAgreementId) -> Result<()> { + StubAgreementRegistry::mark_indexing_agreement_as_cancelling(self, id).await + } + + async fn mark_indexing_agreement_as_abandoning(&self, id: &IndexingAgreementId) -> Result<()> { + StubAgreementRegistry::mark_indexing_agreement_as_abandoning(self, id).await + } + + async fn reopen_indexing_agreement_cancel( + &self, + id: &IndexingAgreementId, + seen_live: bool, + ) -> Result<()> { + StubAgreementRegistry::reopen_indexing_agreement_cancel(self, id, seen_live).await + } + + async fn get_cancelling_agreements( + &self, + batch_size: i64, + max_attempts: u32, + min_age_minutes: i32, + ) -> Result> { + StubAgreementRegistry::get_cancelling_agreements( + self, + batch_size, + max_attempts, + min_age_minutes, + ) + .await + } + + async fn record_cancel_check( + &self, + id: &IndexingAgreementId, + failed_attempts: u32, + ended: Option, + ) -> Result { + StubAgreementRegistry::record_cancel_check(self, id, failed_attempts, ended).await + } + async fn apply_reconciliation( &self, id: &IndexingAgreementId, @@ -466,6 +557,17 @@ impl AgreementRegistry for T { StubAgreementRegistry::get_agreements_pending_chain_cancel(self, batch_size).await } + async fn get_ended_agreements_awaiting_replacement( + &self, + batch_size: i64, + ) -> Result> { + StubAgreementRegistry::get_ended_agreements_awaiting_replacement(self, batch_size).await + } + + async fn mark_replacement_queued(&self, id: &IndexingAgreementId) -> Result<()> { + StubAgreementRegistry::mark_replacement_queued(self, id).await + } + async fn update_agreement_sync_progress( &self, id: &IndexingAgreementId, @@ -488,13 +590,6 @@ impl AgreementRegistry for T { StubAgreementRegistry::exists_active_agreements(self).await } - async fn mark_indexing_agreement_as_abandoned( - &self, - id: &IndexingAgreementId, - ) -> Result { - StubAgreementRegistry::mark_indexing_agreement_as_abandoned(self, id).await - } - async fn get_agreement_fee_rates(&self) -> Result> { StubAgreementRegistry::get_agreement_fee_rates(self).await } @@ -568,4 +663,25 @@ impl AgreementRegistry for T { ) .await } + + async fn record_accept_and_cancel_audit( + &self, + agreement_id: &IndexingAgreementId, + accepted_at: u64, + accepted_tx: &str, + canceled_at: u64, + canceled_by: &str, + canceled_tx: Option<&str>, + ) -> Result<()> { + StubAgreementRegistry::record_accept_and_cancel_audit( + self, + agreement_id, + accepted_at, + accepted_tx, + canceled_at, + canceled_by, + canceled_tx, + ) + .await + } } diff --git a/bin/dipper-service/src/worker/context.rs b/bin/dipper-service/src/worker/context.rs index 4b606026..73175fc5 100644 --- a/bin/dipper-service/src/worker/context.rs +++ b/bin/dipper-service/src/worker/context.rs @@ -234,10 +234,10 @@ impl_from_state!(SendIndexingAgreementProposalCtx { impl_from_state!(CancelRejectedAgreementOnChainCtx { registry, chain_client, - agreement_conf, }); impl_from_state!(SubmitOfferCtx { registry, chain_client, + agreement_conf, }); diff --git a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs index d11e8e72..c6d5735c 100644 --- a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs +++ b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs @@ -1,17 +1,14 @@ -//! Cancel a rejected agreement that was accepted on-chain -//! -//! When an indexer rejects an agreement off-chain but later accepts it on-chain, -//! the chain listener detects this and queues this job to cancel the agreement -//! via the RecurringAgreementManager. +//! Kept so jobs queued before an upgrade still run. The chain listener now moves an +//! agreement dipper rejected or cancelled that went live on-chain anyway back into +//! `Cancelling` itself, and the cancel retry ends it; this job does the same. -use std::{sync::Arc, time::Duration}; +use std::time::Duration; use dipper_core::ids::IndexingAgreementId; use crate::{ - cancel_dispatch::cancel_agreement_on_chain, - chain_client::{ChainClient, ChainClientError}, - config::IndexingAgreementConfig, + cancel_dispatch::reopen_if_live, + chain_client::ChainClient, registry::{AgreementRegistry, IndexingAgreementStatus}, worker::result::{JobError, JobResult}, }; @@ -19,462 +16,91 @@ use crate::{ pub struct Ctx { pub registry: R, pub chain_client: T, - pub agreement_conf: Arc, } -/// Cancel a rejected agreement on-chain. +/// Cancel on-chain an agreement dipper rejected or cancelled that was accepted anyway. #[derive(Debug, serde::Serialize, serde::Deserialize)] pub struct Message { pub agreement_id: IndexingAgreementId, } -/// Cancel a rejected agreement on-chain. -/// -/// This is called when an indexer rejected the proposal off-chain but then accepted -/// on-chain anyway. We cancel the agreement via `cancelIndexingAgreementByPayer` to -/// ensure the indexer doesn't receive payment for work we didn't want. -#[expect( - clippy::cognitive_complexity, - reason = "predates this lint; fix when next touched" -)] +/// Hand an agreement dipper rejected or cancelled that is live on-chain to the cancel retry. pub async fn handle(ctx: Ctx, Message { agreement_id }: &Message) -> JobResult<()> where R: AgreementRegistry + Sync, T: ChainClient, { - // Look up the agreement - let agreement = ctx + let Some(agreement) = ctx .registry .get_indexing_agreement_by_id(agreement_id) .await - .map_err(|err| JobError::Fatal(err.into()))?; - - let agreement = match agreement { - Some(a) => a, - None => { - tracing::error!( - agreement_id = %agreement_id, - "Agreement not found for on-chain cancellation" - ); - return Ok(()); - } + .map_err(|err| JobError::Fatal(err.into()))? + else { + tracing::warn!(%agreement_id, "Agreement not found for on-chain cancellation"); + return Ok(()); }; - - // Verify the agreement is in Rejected status (off-chain rejection that got accepted on-chain) - // The chain listener should only queue this job for Rejected agreements - if agreement.status != IndexingAgreementStatus::Rejected { - tracing::warn!( - agreement_id = %agreement_id, - status = %agreement.status, - "Agreement not in Rejected status, skipping on-chain cancellation" - ); + if !matches!( + agreement.status, + IndexingAgreementStatus::Rejected | IndexingAgreementStatus::CanceledByRequester + ) { return Ok(()); } - - tracing::info!( - agreement_id = %agreement_id, - indexer_id = %agreement.indexer.id, - "Canceling rejected agreement on-chain" - ); - - // Send the cancellation transaction (mode-aware dispatch). - let on_chain_cancel_tx: Option = - match cancel_agreement_on_chain(&ctx.chain_client, &agreement, &ctx.agreement_conf).await { - Ok(Some(tx_hash)) => { - tracing::info!( - agreement_id = %agreement_id, - tx_hash = %tx_hash, - "Successfully submitted on-chain cancellation" - ); - Some(tx_hash.to_string()) - } - Ok(None) => { - tracing::info!( - agreement_id = %agreement_id, - "Rejected agreement already canceled on-chain; reconciling local state" - ); - None - } - Err(err @ ChainClientError::MissingTermsVersionHash { .. }) => { - // Permanent: the hash never appears, so retrying can't help. Fail - // terminally and leave the live agreement for operator action. - tracing::error!( - agreement_id = %agreement_id, - error = %err, - "Cannot cancel rejected agreement: missing terms_version_hash" - ); - return Err(JobError::Fatal(err.into())); - } - Err(err) => { - tracing::warn!( - agreement_id = %agreement_id, - error = %err, - "Failed to cancel agreement on-chain, will retry" - ); - // Retry with backoff - on-chain transactions can fail due to gas issues, nonce, etc. - return Err(JobError::Retryable(err.into(), Duration::from_secs(30))); - } - }; - - // When the row was actually flipped to terminal, record the cancel audit so - // the chain_listener's `terminated` sweep announces it durably. The accept - // was recorded when the rejected-then-accepted anomaly was first detected, so - // the row is sweep-eligible. If the mark failed, the row stays `Rejected` and - // the chain_listener observes the on-chain cancel and flips it itself, then - // the same sweep emits -- so nothing is lost either way. - if mark_cancellation_complete(&ctx.registry, agreement_id).await { - let manager = ctx.agreement_conf.recurring_agreement_manager().to_string(); - if let Err(err) = ctx - .registry - .record_cancel_audit( - agreement_id, - dipper_core::time::now_secs(), - &manager, - on_chain_cancel_tx.as_deref(), - ) - .await - { - tracing::warn!( - agreement_id = %agreement_id, - error = %err, - "failed to record cancel audit; terminated event may emit with fallback fields" - ); - } - } - - Ok(()) -} - -/// Flip the local row to CanceledByRequester after either a fresh on-chain -/// cancel or the discovery that the agreement was already canceled on-chain. -/// Failures here are logged but not fatal — the on-chain side is already in -/// the right state, so the next reconciliation pass can re-attempt the DB -/// update without risking a duplicate transaction. -/// -/// Returns `true` when the row was marked terminal. The caller emits the -/// `terminated` event only on `true`: if the mark failed the row stays -/// `Rejected` (non-terminal), so the chain_listener will observe the on-chain -/// cancel and emit `terminated` itself — emitting here too would duplicate. -async fn mark_cancellation_complete(registry: &R, agreement_id: &IndexingAgreementId) -> bool -where - R: AgreementRegistry + Sync, -{ - match registry - .mark_indexing_agreement_as_canceled_by_requester(agreement_id) + reopen_if_live(&ctx.registry, &ctx.chain_client, &agreement) .await - { - Ok(()) => { - tracing::info!( - agreement_id = %agreement_id, - old_status = "REJECTED", - new_status = "CANCELED_BY_REQUESTER", - reason = "canceled_on_chain_after_rejection", - "agreement state transition" - ); - true - } - Err(err) => { - tracing::error!( - agreement_id = %agreement_id, - error = %err, - "Failed to update agreement status after on-chain cancellation" - ); - false - } - } + .map(|_| ()) + .map_err(|err| JobError::Retryable(err.into(), Duration::from_secs(30))) } #[cfg(test)] mod tests { - use std::sync::Mutex; + use std::sync::{Arc, Mutex}; use async_trait::async_trait; - use dipper_core::ids::IndexingRequestId; use dipper_rpc::indexer::indexer_client::sol::RecurringCollectionAgreement; - use thegraph_core::{ - DeploymentId, IndexerId, - alloy::primitives::{Address, B256, U256}, - }; - use time::OffsetDateTime; - use url::Url; + use thegraph_core::alloy::primitives::{Address, B256}; use super::*; use crate::{ - chain_client::{ChainClient, ChainClientError}, - registry::{ - IndexingAgreement, IndexingAgreementStatus, IndexingAgreementTerms, - IndexingAgreementTermsMetadata, - }, + cancel_dispatch::tests::agreement, + chain_client::{AgreementOnChain, ChainClientError}, + registry::{IndexingAgreement, StubAgreementRegistry}, }; - // ========================================================================= - // Mock implementations - // ========================================================================= - - /// Registry that returns a single configurable agreement and records the - /// terminal-cancel transition + cancel-audit calls the handler drives. - /// `Clone` shares the tracked state (Arc), so a test can clone one into the - /// `Ctx` and still assert on the original after `handle` consumes the ctx. - #[derive(Clone)] struct MockRegistry { - agreement: Arc>>, - marked_canceled: Arc>>, - /// Ids passed to `record_cancel_audit` -- the signal the handler drives - /// the terminated event (the chain_listener sweep emits from this audit). - recorded_cancel_audit: Arc>>, - /// When true, `mark_indexing_agreement_as_canceled_by_requester` errors. - fail_mark: bool, - } - - impl MockRegistry { - fn new(agreement: IndexingAgreement) -> Self { - Self { - agreement: Arc::new(Mutex::new(Some(agreement))), - marked_canceled: Arc::new(Mutex::new(Vec::new())), - recorded_cancel_audit: Arc::new(Mutex::new(Vec::new())), - fail_mark: false, - } - } - - fn with_mark_failure(agreement: IndexingAgreement) -> Self { - Self { - fail_mark: true, - ..Self::new(agreement) - } - } + agreement: IndexingAgreement, + reopened: Arc>>, } #[async_trait] - impl AgreementRegistry for MockRegistry { + impl StubAgreementRegistry for MockRegistry { async fn get_indexing_agreement_by_id( &self, _id: &IndexingAgreementId, ) -> crate::registry::Result> { - Ok(self.agreement.lock().unwrap().clone()) - } - - async fn get_indexing_agreements_by_deployment_id( - &self, - _deployment_id: &DeploymentId, - ) -> crate::registry::Result> { - Ok(vec![]) - } - - async fn get_indexing_agreements_by_indexer_id( - &self, - _indexer_id: &IndexerId, - ) -> crate::registry::Result> { - Ok(vec![]) - } - - async fn get_pending_agreement_indexers_by_deployment( - &self, - _indexer_ids: &[IndexerId], - ) -> crate::registry::Result>> - { - Ok(std::collections::HashMap::new()) - } - - async fn get_declined_indexers_by_deployment( - &self, - _default_lookback_days: i32, - _price_lookback_days: i32, - _transient_lookback_minutes: i32, - _uncertain_lookback_days: i32, - ) -> crate::registry::Result>> - { - Ok(std::collections::HashMap::new()) - } - - async fn get_indexing_agreements_by_indexing_request_id( - &self, - _request_id: &IndexingRequestId, - ) -> crate::registry::Result> { - Ok(vec![]) - } - - async fn get_active_indexing_agreements_by_indexing_request_id( - &self, - _request_id: &IndexingRequestId, - ) -> crate::registry::Result> { - Ok(vec![]) - } - - async fn count_accepted_agreements_by_deployment( - &self, - _deployment_id: &DeploymentId, - ) -> crate::registry::Result { - Ok(0) - } - - async fn record_cancel_audit( - &self, - agreement_id: &IndexingAgreementId, - _canceled_at: u64, - _canceled_by: &str, - _canceled_tx: Option<&str>, - ) -> crate::registry::Result<()> { - self.recorded_cancel_audit - .lock() - .unwrap() - .push(*agreement_id); - Ok(()) - } - - async fn register_new_indexing_agreement( - &self, - _params: crate::registry::NewAgreementParams, - ) -> crate::registry::Result { - Ok(IndexingAgreementId::from_bytes(rand::random())) - } - - async fn register_agreement_with_pending_cancellation( - &self, - _params: crate::registry::NewAgreementParams, - _old_agreement_id: IndexingAgreementId, - ) -> crate::registry::Result { - Ok(IndexingAgreementId::from_bytes(rand::random())) - } - - async fn get_unresponsive_indexers( - &self, - _lookback_days: i32, - _chain_id: thegraph_core::alloy::primitives::ChainId, - ) -> crate::registry::Result> { - unimplemented!() + Ok(Some(self.agreement.clone())) } - - async fn mark_indexing_agreement_as_unresponsive( - &self, - _id: &IndexingAgreementId, - ) -> crate::registry::Result<()> { - unimplemented!() - } - - async fn count_created_agreements_by_indexer( - &self, - ) -> crate::registry::Result<(std::collections::HashMap, u64)> { - unimplemented!() - } - - async fn update_offer_tx_hash( - &self, - _id: &IndexingAgreementId, - _tx_hash: &[u8; 32], - ) -> crate::registry::Result<()> { - Ok(()) - } - - async fn mark_indexing_agreement_as_canceled_by_requester( + async fn reopen_indexing_agreement_cancel( &self, id: &IndexingAgreementId, + _seen_live: bool, ) -> crate::registry::Result<()> { - if self.fail_mark { - return Err(crate::registry::Error::NoRecordsUpdated); - } - self.marked_canceled.lock().unwrap().push(*id); - Ok(()) - } - - async fn apply_reconciliation( - &self, - _id: &IndexingAgreementId, - _apply_accept: bool, - _cancel: Option, - ) -> crate::registry::Result { - Ok(crate::registry::ReconciliationOutcome { - did_accept: false, - did_cancel: false, - }) - } - - async fn get_expired_created_agreements( - &self, - _batch_size: i64, - _chain_timestamp: u64, - ) -> crate::registry::Result> { - Ok(vec![]) - } - - async fn mark_indexing_agreement_as_expired( - &self, - _id: &IndexingAgreementId, - ) -> crate::registry::Result<()> { - Ok(()) - } - - async fn mark_indexing_agreement_as_rejected( - &self, - _id: &IndexingAgreementId, - _rejection_reason: Option<&str>, - ) -> crate::registry::Result<()> { + self.reopened.lock().unwrap().push(*id); Ok(()) } - - async fn get_accepted_on_chain_agreements( - &self, - _batch_size: i64, - ) -> crate::registry::Result> { - Ok(vec![]) - } - - async fn get_agreements_pending_chain_cancel( - &self, - _batch_size: i64, - ) -> crate::registry::Result> { - Ok(vec![]) - } - - async fn update_agreement_sync_progress( - &self, - _id: &IndexingAgreementId, - _block_height: u64, - _progress_at: time::OffsetDateTime, - ) -> crate::registry::Result<()> { - Ok(()) - } - - async fn count_active_agreements_by_deployment( - &self, - ) -> crate::registry::Result> { - Ok(std::collections::HashMap::new()) - } - - async fn mark_indexing_agreement_as_abandoned( - &self, - _id: &IndexingAgreementId, - ) -> crate::registry::Result { - Err(crate::registry::Error::NoRecordsUpdated) - } - - async fn get_agreement_fee_rates( - &self, - ) -> crate::registry::Result> { - Ok(vec![]) - } } - /// Chain client whose manager cancel always mines (returns a tx hash) and - /// whose post-cancel liveness read reports the agreement is no longer active, - /// so the cancel is confirmed. - #[derive(Default)] - struct MockChainClient; + struct MockChain { + live: bool, + } #[async_trait] - impl ChainClient for MockChainClient { - async fn latest_block_timestamp(&self) -> Result { - Err(ChainClientError::RpcError(anyhow::anyhow!( - "latest_block_timestamp not mocked" - ))) - } - + impl ChainClient for MockChain { async fn offer_via_manager( &self, _rca: &RecurringCollectionAgreement, ) -> Result, ChainClientError> { - Ok(None) + unimplemented!() } - async fn cancel_via_manager( &self, _collector: Address, @@ -482,15 +108,14 @@ mod tests { _version_hash: B256, _options: u16, ) -> Result, ChainClientError> { - Ok(Some(B256::ZERO)) + panic!("the cancel retry sends cancels, not this job") } - async fn reconcile_provider( &self, _collector: Address, _provider: Address, ) -> Result, ChainClientError> { - Ok(None) + unimplemented!() } async fn reconcile_agreement( &self, @@ -499,144 +124,57 @@ mod tests { ) -> Result, ChainClientError> { unimplemented!() } - - async fn agreement_still_active( + async fn agreement_on_chain( &self, _agreement_id: &[u8; 16], - ) -> Result { - Ok(false) + ) -> Result { + Ok(AgreementOnChain::live_if(self.live)) + } + async fn latest_block_timestamp(&self) -> Result { + unimplemented!() } } - fn test_agreement_conf() -> Arc { - Arc::new(IndexingAgreementConfig { - data_service: Address::ZERO, - recurring_collector: Address::ZERO, - recurring_agreement_manager: Address::ZERO, - max_agreement_grt_per_30_days: 0.0, - max_seconds_per_collection: 0, - min_seconds_per_collection: 0, - duration_seconds: 0, - deadline_seconds: 0, - max_grt_per_30_days: std::collections::BTreeMap::new(), - max_grt_per_billion_entities_per_30_days: 0.0, - declined_indexer_lookback_days: 0, - price_rejection_lookback_days: 0, - transient_rejection_lookback_minutes: 0, - uncertain_rejection_lookback_days: 0, - unresponsive_indexer_lookback_days: 0, - mass_unresponsive_trip_fraction: 0.5, - mass_unresponsive_reset_fraction: 0.25, - dips_accepting_snapshot_max_age_hours: 48, - dips_accepting_cache_ttl_seconds: 300, - max_in_flight_offers_per_indexer: None, - max_in_flight_offers_total: None, - }) - } - - fn test_deployment_id() -> DeploymentId { - "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" - .parse() - .unwrap() - } - - fn make_agreement(status: IndexingAgreementStatus) -> IndexingAgreement { - IndexingAgreement { - id: IndexingAgreementId::from_bytes(rand::random()), - nonce_uuid: uuid::Uuid::now_v7(), - created_at: OffsetDateTime::now_utc(), - updated_at: OffsetDateTime::now_utc(), - status, - indexing_request_id: IndexingRequestId::new(), - indexer: crate::registry::Indexer { - id: IndexerId::from(Address::ZERO), - url: Url::parse("https://indexer.example").unwrap(), - }, - terms: IndexingAgreementTerms { - payer: Address::ZERO, - service_provider: Address::ZERO, - data_service: Address::ZERO, - deadline: 0, - ends_at: 0, - max_initial_tokens: U256::ZERO, - max_ongoing_tokens_per_second: U256::ZERO, - min_seconds_per_collection: 0, - max_seconds_per_collection: 0, - conditions: 0, - metadata: IndexingAgreementTermsMetadata { - tokens_per_second: U256::ZERO, - tokens_per_entity_per_second: U256::ZERO, - subgraph_deployment_id: test_deployment_id(), - protocol_network: 1u64, - chain_id: 1u64, - proposed_at: 0, - }, + async fn run(status: IndexingAgreementStatus, live: bool) -> Vec { + let reopened = Arc::new(Mutex::new(Vec::new())); + let agreement = agreement(status, Some(vec![7u8; 32])); + let message = Message { + agreement_id: agreement.id, + }; + let ctx = Ctx { + registry: MockRegistry { + agreement, + reopened: Arc::clone(&reopened), }, - last_block_height: None, - last_progress_at: None, - rejection_reason: None, - // 32-byte hash so the on-chain cancel path is exercised. - terms_version_hash: Some(vec![0u8; 32]), - } - } + chain_client: MockChain { live }, + }; - // ========================================================================= - // Tests - // ========================================================================= + handle(ctx, &message).await.expect("job ok"); - fn ctx_for(registry: MockRegistry) -> Ctx { - Ctx { - registry, - chain_client: MockChainClient, - agreement_conf: test_agreement_conf(), - } + reopened.lock().unwrap().clone() } #[tokio::test] - async fn rejected_agreement_records_cancel_audit_once() { - // The handler no longer emits `terminated` directly: it records the cancel - // audit, and the chain_listener sweep announces it durably. Assert exactly - // one audit was recorded for the agreement. - let agreement = make_agreement(IndexingAgreementStatus::Rejected); - let agreement_id = agreement.id; - let registry = MockRegistry::new(agreement); - - let result = handle(ctx_for(registry.clone()), &Message { agreement_id }).await; - assert!(result.is_ok(), "handle should succeed: {result:?}"); - - let recorded = registry.recorded_cancel_audit.lock().unwrap().clone(); - assert_eq!(recorded, vec![agreement_id], "exactly one cancel audit"); + async fn hands_a_live_agreement_dipper_ended_to_the_cancel_retry() { + for status in [ + IndexingAgreementStatus::Rejected, + IndexingAgreementStatus::CanceledByRequester, + ] { + assert_eq!(run(status, true).await.len(), 1, "{status}"); + } } #[tokio::test] - async fn failed_local_mark_records_no_cancel_audit() { - // On-chain cancel succeeds but the local DB mark fails, leaving the row - // non-terminal. The handler must NOT record cancel audit -- the - // chain_listener will observe the on-chain cancel, flip the row, and the - // sweep emits from there. - let agreement = make_agreement(IndexingAgreementStatus::Rejected); - let agreement_id = agreement.id; - let registry = MockRegistry::with_mark_failure(agreement); - - let result = handle(ctx_for(registry.clone()), &Message { agreement_id }).await; - assert!(result.is_ok(), "handle should still return Ok: {result:?}"); + async fn leaves_an_agreement_that_already_ended_or_is_still_wanted() { assert!( - registry.recorded_cancel_audit.lock().unwrap().is_empty(), - "no cancel audit recorded when the local mark failed" + run(IndexingAgreementStatus::CanceledByRequester, false) + .await + .is_empty() ); - } - - #[tokio::test] - async fn non_rejected_agreement_records_nothing() { - let agreement = make_agreement(IndexingAgreementStatus::Created); - let agreement_id = agreement.id; - let registry = MockRegistry::new(agreement); - - let result = handle(ctx_for(registry.clone()), &Message { agreement_id }).await; - assert!(result.is_ok(), "handle should return Ok: {result:?}"); assert!( - registry.recorded_cancel_audit.lock().unwrap().is_empty(), - "no cancel audit recorded for a non-Rejected agreement" + run(IndexingAgreementStatus::AcceptedOnChain, true) + .await + .is_empty() ); } } diff --git a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs index fdd5fb81..584e08d3 100644 --- a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs +++ b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs @@ -633,107 +633,27 @@ where ); } - // Cancel old agreements that have no replacement to pair with. - // These indexers are leaving the target group with nothing taking - // their place. Fire the on-chain cancel first; only mark the local row - // CanceledByRequester after the chain tx is accepted, so retry on a - // transient chain-client failure does the right thing. + // Cancel old agreements that have no replacement to pair with: these indexers + // leave the target group with nothing taking their place. let mut directly_cancelled = 0u32; let mut cancel_failures = 0u32; for old_agreement in old_iter { - // Skip agreements that haven't been accepted on-chain yet -- there is - // nothing on the contract to cancel. The local row goes straight to - // CanceledByRequester so the indexer never picks it up. - let needs_on_chain_cancel = matches!( - old_agreement.status, - crate::registry::IndexingAgreementStatus::AcceptedOnChain - ); - - let mut on_chain_cancel_tx: Option = None; - if needs_on_chain_cancel { - match crate::cancel_dispatch::cancel_agreement_on_chain( - &ctx.chain_client, - old_agreement, - &ctx.agreement_conf, - ) - .await - { - Ok(Some(tx_hash)) => { - tracing::info!( - agreement_id = %old_agreement.id, - indexing_request_id = %indexing_request_id, - %tx_hash, - "Submitted on-chain cancellation for unpaired old agreement" - ); - on_chain_cancel_tx = Some(tx_hash.to_string()); - } - Ok(None) => { - tracing::info!( - agreement_id = %old_agreement.id, - indexing_request_id = %indexing_request_id, - "Unpaired old agreement already canceled on-chain; proceeding with local cleanup" - ); - } - Err(err) => { - tracing::warn!( - error = %err, - agreement_id = %old_agreement.id, - "On-chain cancel failed; will retry on next reassessment" - ); - cancel_failures += 1; - continue; - } + let new_status = match cancel_unpaired(&ctx, old_agreement).await { + Unpaired::Moved(new_status) => new_status, + Unpaired::AlreadyEnding => continue, + Unpaired::Failed => { + cancel_failures += 1; + continue; } - } - - if let Err(err) = ctx - .registry - .mark_indexing_agreement_as_canceled_by_requester(&old_agreement.id) - .await - { - tracing::error!( - error=%err, - agreement_id=%old_agreement.id, - "Failed to mark unpaired old agreement as canceled in local DB" - ); - cancel_failures += 1; - continue; - } - + }; tracing::info!( agreement_id = %old_agreement.id, indexing_request_id = %indexing_request_id, old_status = %old_agreement.status, - new_status = "CANCELED_BY_REQUESTER", + new_status, reason = "reassessment_not_in_target_group", "agreement state transition" ); - - // Record the cancel audit for the accepted-on-chain agreements dipper just - // cancelled, so the chain_listener's `terminated` sweep announces them - // durably. Never-accepted agreements (`!needs_on_chain_cancel`) were never - // live on-chain: they are not sweep-eligible (`accepted_at IS NULL`) and - // correctly emit nothing. - if needs_on_chain_cancel { - let manager = ctx.agreement_conf.recurring_agreement_manager().to_string(); - if let Err(err) = ctx - .registry - .record_cancel_audit( - &old_agreement.id, - now_secs(), - &manager, - on_chain_cancel_tx.as_deref(), - ) - .await - { - tracing::warn!( - agreement_id = %old_agreement.id, - error = %err, - "failed to record cancel audit; terminated event may emit with fallback fields" - ); - } - } - directly_cancelled += 1; } @@ -746,19 +666,14 @@ where } if cancel_failures > 0 { - // Two recovery paths cover any agreements left AcceptedOnChain here: - // - // - Shrink-to-zero (request now Canceled): the chain_listener's - // `sweep_orphan_canceled_agreements` retries on every sweep tick - // (default ~5 min at fast poll, ~5 h at slow poll). - // - Shrink-not-zero (request still Open with too many agreements): - // the periodic reassignment service re-queues reassessment at its - // configured cadence (default 24 h). + // An agreement that couldn't be marked keeps its status: the orphan-cancel sweep + // retries it once the request is Canceled, the periodic reassignment service + // (24 h) while it is Open. tracing::warn!( indexing_request_id=%indexing_request_id, failures=cancel_failures, - "some agreement cancels failed during reassessment; retry will fire \ - via the orphan-cancel sweep (Canceled requests) or the periodic \ + "some agreements could not be marked for cancelling during reassessment; retry \ + will fire via the orphan-cancel sweep (Canceled requests) or the periodic \ reassignment service (Open requests over-target)" ); } @@ -918,7 +833,7 @@ mod lifecycle_event_tests { use super::{Ctx, Message, handle}; use crate::{ - chain_client::{ChainClient, ChainClientError}, + chain_client::{AgreementOnChain, ChainClient, ChainClientError}, config::IndexingAgreementConfig, network::{ provider::NetworkProviderService, @@ -928,11 +843,10 @@ mod lifecycle_event_tests { }, }, registry::{ - AgreementFeeRate, AgreementRegistry, CancelKind, Indexer, IndexerDenylistRegistry, - IndexingAgreement, IndexingAgreementStatus, IndexingAgreementTerms, - IndexingAgreementTermsMetadata, IndexingRequest, IndexingRequestRegistry, - NewAgreementParams, PendingCancellation, PendingCancellationRegistry, - ReconciliationItem, ReconciliationOutcome, Result as RegistryResult, SetTargetOutcome, + AgreementFeeRate, Indexer, IndexerDenylistRegistry, IndexingAgreement, + IndexingAgreementStatus, IndexingAgreementTerms, IndexingAgreementTermsMetadata, + IndexingRequest, IndexingRequestRegistry, NewAgreementParams, PendingCancellation, + PendingCancellationRegistry, Result as RegistryResult, SetTargetOutcome, }, signing::eip712::Eip712Signer, test_support::{CapturedEvent, CapturingEventsProducer}, @@ -987,8 +901,8 @@ mod lifecycle_event_tests { // ---- Mock: worker queue -------------------------------------------------- - /// Records every `send_indexing_agreement_proposal` call's indexer URL. - /// Clone shares the buffer so a caller can inspect proposals after `handle`. + /// Records every `send_indexing_agreement_proposal` call's indexer URL. Clone shares the + /// buffer for inspection after `handle`. #[derive(Default, Clone)] struct MockQueue { proposals: Arc>>, @@ -1020,14 +934,6 @@ mod lifecycle_event_tests { unimplemented!("not exercised by reassess handler") } - async fn cancel_rejected_agreement_on_chain( - &self, - _agreement_id: IndexingAgreementId, - _priority: crate::worker::queue::JobPriority, - ) -> anyhow::Result { - unimplemented!("not exercised by reassess handler") - } - async fn submit_offer( &self, _agreement_id: IndexingAgreementId, @@ -1043,9 +949,17 @@ mod lifecycle_event_tests { // ---- Mock: chain client -------------------------------------------------- - /// Always reports a successful cancel that the post-cancel read confirms. - #[derive(Default)] - struct MockChainClient; + /// Shows every agreement live until a cancel is sent for it, unless nothing is on + /// chain, and records the id of every agreement it was asked to cancel. + #[derive(Default, Clone)] + struct MockChainClient { + cancelled: Arc>>, + nothing_on_chain: bool, + /// When set, each cancel records whether its agreement was already marked cancelling. + marked_cancelling: Option>>>, + marked_at_cancel: Arc>>, + fail_cancel: bool, + } #[async_trait] impl ChainClient for MockChainClient { @@ -1065,10 +979,19 @@ mod lifecycle_event_tests { async fn cancel_via_manager( &self, _collector: Address, - _agreement_id: &[u8; 16], + agreement_id: &[u8; 16], _version_hash: B256, _options: u16, ) -> std::result::Result, ChainClientError> { + if self.fail_cancel { + return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); + } + self.cancelled.lock().unwrap().push(*agreement_id); + if let Some(marked) = &self.marked_cancelling { + let id = IndexingAgreementId::from_bytes(*agreement_id); + let was_marked = marked.lock().unwrap().contains(&id); + self.marked_at_cancel.lock().unwrap().push(was_marked); + } Ok(Some(B256::repeat_byte(0xcd))) } @@ -1087,12 +1010,13 @@ mod lifecycle_event_tests { unimplemented!() } - async fn agreement_still_active( + async fn agreement_on_chain( &self, - _agreement_id: &[u8; 16], - ) -> std::result::Result { - // Cancel confirmed: agreement is no longer active on-chain. - Ok(false) + agreement_id: &[u8; 16], + ) -> std::result::Result { + Ok(AgreementOnChain::live_if( + !self.nothing_on_chain && !self.cancelled.lock().unwrap().contains(agreement_id), + )) } } @@ -1112,6 +1036,9 @@ mod lifecycle_event_tests { /// `set_indexing_request_shortfall_active` flips it and reports whether it /// changed, so tests can exercise the transition-based emit. shortfall_active: std::sync::Mutex, + /// Ids marked CanceledByRequester locally. + marked_cancelled: Arc>>, + marked_cancelling: Arc>>, } #[async_trait] @@ -1160,13 +1087,7 @@ mod lifecycle_event_tests { } #[async_trait] - impl AgreementRegistry for MockRegistry { - async fn get_indexing_agreement_by_id( - &self, - _id: &IndexingAgreementId, - ) -> RegistryResult> { - unimplemented!() - } + impl crate::registry::StubAgreementRegistry for MockRegistry { // gather_selection_context: all active agreements for the deployment. async fn get_indexing_agreements_by_deployment_id( &self, @@ -1174,12 +1095,6 @@ mod lifecycle_event_tests { ) -> RegistryResult> { Ok(self.active_agreements.clone()) } - async fn get_indexing_agreements_by_indexer_id( - &self, - _indexer_id: &IndexerId, - ) -> RegistryResult> { - unimplemented!() - } // gather_selection_context: no pending agreements. async fn get_pending_agreement_indexers_by_deployment( &self, @@ -1197,12 +1112,6 @@ mod lifecycle_event_tests { ) -> RegistryResult>> { Ok(HashMap::new()) } - async fn get_indexing_agreements_by_indexing_request_id( - &self, - _request_id: &IndexingRequestId, - ) -> RegistryResult> { - unimplemented!() - } // The handler's current-state baseline for the diff. async fn get_active_indexing_agreements_by_indexing_request_id( &self, @@ -1237,96 +1146,27 @@ mod lifecycle_event_tests { ) -> RegistryResult> { Ok(Vec::new()) } - async fn mark_indexing_agreement_as_unresponsive( - &self, - _id: &IndexingAgreementId, - ) -> RegistryResult<()> { - unimplemented!() - } async fn count_created_agreements_by_indexer( &self, ) -> RegistryResult<(std::collections::HashMap, u64)> { Ok((std::collections::HashMap::new(), 0)) } - async fn update_offer_tx_hash( - &self, - _id: &IndexingAgreementId, - _tx_hash: &[u8; 32], - ) -> RegistryResult<()> { - unimplemented!() - } // Cancel path: pre-mark the local row terminal. async fn mark_indexing_agreement_as_canceled_by_requester( &self, - _id: &IndexingAgreementId, + id: &IndexingAgreementId, ) -> RegistryResult<()> { + self.marked_cancelled.lock().unwrap().push(*id); Ok(()) } - async fn apply_reconciliation( - &self, - _id: &IndexingAgreementId, - _apply_accept: bool, - _cancel: Option, - ) -> RegistryResult { - unimplemented!() - } - async fn apply_reconciliation_batch( + async fn mark_indexing_agreement_as_cancelling( &self, - _items: &[ReconciliationItem], - ) -> RegistryResult> { - unimplemented!() - } - async fn get_expired_created_agreements( - &self, - _batch_size: i64, - _chain_timestamp: u64, - ) -> RegistryResult> { - unimplemented!() - } - async fn mark_indexing_agreement_as_expired( - &self, - _id: &IndexingAgreementId, - ) -> RegistryResult<()> { - unimplemented!() - } - async fn mark_indexing_agreement_as_rejected( - &self, - _id: &IndexingAgreementId, - _rejection_reason: Option<&str>, + id: &IndexingAgreementId, ) -> RegistryResult<()> { - unimplemented!() - } - async fn get_accepted_on_chain_agreements( - &self, - _batch_size: i64, - ) -> RegistryResult> { - unimplemented!() - } - async fn get_agreements_pending_chain_cancel( - &self, - _batch_size: i64, - ) -> RegistryResult> { - unimplemented!() - } - async fn update_agreement_sync_progress( - &self, - _id: &IndexingAgreementId, - _block_height: u64, - _progress_at: OffsetDateTime, - ) -> RegistryResult<()> { - unimplemented!() - } - async fn count_active_agreements_by_deployment( - &self, - ) -> RegistryResult> { - unimplemented!() - } - async fn mark_indexing_agreement_as_abandoned( - &self, - _id: &IndexingAgreementId, - ) -> RegistryResult { - unimplemented!() + self.marked_cancelling.lock().unwrap().push(*id); + Ok(()) } + // gather_selection_context: optimistic DIPs fees (none). async fn get_agreement_fee_rates(&self) -> RegistryResult> { Ok(Vec::new()) @@ -1416,19 +1256,11 @@ mod lifecycle_event_tests { min_seconds_per_collection: 60, duration_seconds: 86400, deadline_seconds: 3600, - max_grt_per_30_days: std::collections::BTreeMap::new(), - max_grt_per_billion_entities_per_30_days: 0.0, declined_indexer_lookback_days: 30, price_rejection_lookback_days: 1, transient_rejection_lookback_minutes: 30, uncertain_rejection_lookback_days: 1, - unresponsive_indexer_lookback_days: 0, - mass_unresponsive_trip_fraction: 0.5, - mass_unresponsive_reset_fraction: 0.25, - dips_accepting_snapshot_max_age_hours: 48, - dips_accepting_cache_ttl_seconds: 300, - max_in_flight_offers_per_indexer: None, - max_in_flight_offers_total: None, + ..IndexingAgreementConfig::for_tests() } } @@ -1573,7 +1405,7 @@ mod lifecycle_event_tests { selected: vec![selected(new_idx)], }, MockQueue::default(), - MockChainClient, + MockChainClient::default(), events.clone(), snapshot, ); @@ -1626,7 +1458,7 @@ mod lifecycle_event_tests { selected: vec![selected(new_idx)], }, MockQueue::default(), - MockChainClient, + MockChainClient::default(), events.clone(), snapshot, ); @@ -1666,6 +1498,8 @@ mod lifecycle_event_tests { chain_state_lookup_fails: false, // Already in shortfall. shortfall_active: std::sync::Mutex::new(true), + marked_cancelled: Arc::default(), + marked_cancelling: Arc::default(), }; let ctx = build_ctx( registry, @@ -1673,7 +1507,7 @@ mod lifecycle_event_tests { selected: vec![selected(idx)], }, MockQueue::default(), - MockChainClient, + MockChainClient::default(), events.clone(), indexer_urls::Snapshot::new(), ); @@ -1702,6 +1536,8 @@ mod lifecycle_event_tests { accepted_count: 1, chain_state_lookup_fails: false, shortfall_active: std::sync::Mutex::new(false), + marked_cancelled: Arc::default(), + marked_cancelling: Arc::default(), }; let ctx = build_ctx( registry, @@ -1709,7 +1545,7 @@ mod lifecycle_event_tests { selected: vec![selected(idx)], }, MockQueue::default(), - MockChainClient, + MockChainClient::default(), events.clone(), indexer_urls::Snapshot::new(), ); @@ -1734,7 +1570,7 @@ mod lifecycle_event_tests { MockRegistry::default(), // no current agreements, latch false MockIisa { selected: vec![] }, // zero available MockQueue::default(), - MockChainClient, + MockChainClient::default(), events.clone(), indexer_urls::Snapshot::new(), ); @@ -1772,12 +1608,14 @@ mod lifecycle_event_tests { accepted_count: 0, chain_state_lookup_fails: false, shortfall_active: std::sync::Mutex::new(false), + marked_cancelled: Arc::default(), + marked_cancelling: Arc::default(), }; let ctx = build_ctx( registry, MockIisa { selected: vec![] }, // IISA returns nothing -> coverage drops MockQueue::default(), - MockChainClient, + MockChainClient::default(), events.clone(), indexer_urls::Snapshot::new(), ); @@ -1825,6 +1663,8 @@ mod lifecycle_event_tests { accepted_count: 0, chain_state_lookup_fails: false, shortfall_active: std::sync::Mutex::new(false), + marked_cancelled: Arc::default(), + marked_cancelling: Arc::default(), }; let ctx = build_ctx( registry, @@ -1832,7 +1672,7 @@ mod lifecycle_event_tests { selected: vec![selected(new_idx)], }, queue.clone(), - MockChainClient, + MockChainClient::default(), events.clone(), snapshot, ); @@ -1870,9 +1710,8 @@ mod lifecycle_event_tests { async fn never_accepted_unpaired_cancel_does_not_emit_terminated() { // Both old agreements were never accepted on-chain (Created). One add // pairs with the first old agreement; the second, unpaired old agreement - // reaches the cancel loop but, being never-accepted - // (`!needs_on_chain_cancel`), must NOT emit `terminated`. The add still - // emits `proposed`. Net: exactly one event, a `proposed`. + // reaches the cancel loop but, never accepted, must NOT emit `terminated`. + // The add still emits `proposed`. Net: exactly one event, a `proposed`. let new_idx = indexer_id(0x44); let old_paired = indexer_id(0x55); let old_unpaired = indexer_id(0x56); @@ -1889,6 +1728,8 @@ mod lifecycle_event_tests { accepted_count: 0, chain_state_lookup_fails: false, shortfall_active: std::sync::Mutex::new(false), + marked_cancelled: Arc::default(), + marked_cancelling: Arc::default(), }; let ctx = build_ctx( registry, @@ -1896,7 +1737,7 @@ mod lifecycle_event_tests { selected: vec![selected(new_idx)], }, MockQueue::default(), - MockChainClient, + MockChainClient::default(), events.clone(), snapshot, ); @@ -1911,6 +1752,225 @@ mod lifecycle_event_tests { ); assert!(matches!(captured[0], CapturedEvent::Proposed { .. })); } + + /// Ctx whose only active agreement, in `status`, leaves the target group, with the + /// chain mock recording whether the row was already marked cancelling at each cancel. + fn ctx_cancelling_one( + status: IndexingAgreementStatus, + ) -> ( + Ctx, + IndexingAgreement, + ) { + let leaving = agreement(indexer_id(0x11), status); + let mut ctx = build_ctx( + MockRegistry { + active_agreements: vec![leaving.clone()], + ..MockRegistry::default() + }, + MockIisa { selected: vec![] }, + MockQueue::default(), + MockChainClient::default(), + CapturingEventsProducer::new(), + indexer_urls::Snapshot::new(), + ); + ctx.chain_client.marked_cancelling = Some(ctx.registry.marked_cancelling.clone()); + (ctx, leaving) + } + + #[tokio::test] + async fn marks_an_unaccepted_agreement_cancelling_before_cancelling_it_on_chain() { + // Its offer may be in flight. An offer landing after the cancel can't be + // withdrawn by it, so its job must find the agreement already marked. It + // stays cancelling: the offer could still be accepted until its deadline. + let (ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::Created); + let chain = ctx.chain_client.clone(); + let cancelled = ctx.registry.marked_cancelled.clone(); + + let result = handle(ctx, &test_message(0)).await; + + assert!(result.is_ok(), "got {result:?}"); + assert_eq!( + *chain.cancelled.lock().unwrap(), + vec![*leaving.id.as_bytes()] + ); + assert_eq!(*chain.marked_at_cancel.lock().unwrap(), vec![true]); + assert!(cancelled.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn sends_no_cancel_for_an_offer_that_never_reached_the_chain() { + // A cancel of nothing still mines and costs gas. The mark made first means an + // offer still in flight withdraws itself when it lands. + let (mut ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::Created); + ctx.chain_client.nothing_on_chain = true; + let chain = ctx.chain_client.clone(); + let marked = ctx.registry.marked_cancelling.clone(); + + let result = handle(ctx, &test_message(0)).await; + + assert!(result.is_ok(), "got {result:?}"); + assert!(chain.cancelled.lock().unwrap().is_empty()); + assert_eq!(*marked.lock().unwrap(), vec![leaving.id]); + } + + #[tokio::test] + async fn marks_an_accepted_agreement_cancelled_once_its_cancel_lands() { + let (ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::AcceptedOnChain); + let chain = ctx.chain_client.clone(); + let cancelled = ctx.registry.marked_cancelled.clone(); + + let result = handle(ctx, &test_message(0)).await; + + assert!(result.is_ok(), "got {result:?}"); + assert_eq!(*chain.marked_at_cancel.lock().unwrap(), vec![true]); + assert_eq!(*cancelled.lock().unwrap(), vec![leaving.id]); + } + + #[tokio::test] + async fn an_agreement_whose_cancel_fails_is_left_cancelling_for_the_retry() { + // The cancel retry re-sends the cancel of every cancelling agreement until + // the chain shows it ended, so nothing is queued here. + for status in [ + IndexingAgreementStatus::Created, + IndexingAgreementStatus::AcceptedOnChain, + ] { + let (mut ctx, leaving) = ctx_cancelling_one(status); + ctx.chain_client.fail_cancel = true; + let cancelling = ctx.registry.marked_cancelling.clone(); + let cancelled = ctx.registry.marked_cancelled.clone(); + + let result = handle(ctx, &test_message(0)).await; + + assert!(result.is_ok(), "got {result:?}"); + assert_eq!(*cancelling.lock().unwrap(), vec![leaving.id]); + assert!(cancelled.lock().unwrap().is_empty()); + } + } + + #[tokio::test] + async fn cancelling_an_unaccepted_agreement_revokes_its_offer_on_chain() { + // A Created agreement's offer can already be on-chain, where the indexer + // can still accept it. Cancelling the request must revoke that offer, not + // only mark the local row cancelled. + let leaving = agreement(indexer_id(0x11), IndexingAgreementStatus::Created); + let leaving_id = *leaving.id.as_bytes(); + let chain_client = MockChainClient::default(); + let registry = MockRegistry { + active_agreements: vec![leaving], + accepted_count: 0, + chain_state_lookup_fails: false, + shortfall_active: std::sync::Mutex::new(false), + marked_cancelled: Arc::default(), + marked_cancelling: Arc::default(), + }; + let ctx = build_ctx( + registry, + MockIisa { selected: vec![] }, + MockQueue::default(), + chain_client.clone(), + CapturingEventsProducer::new(), + indexer_urls::Snapshot::new(), + ); + + handle(ctx, &test_message(0)).await.expect("handler ok"); + + assert_eq!(*chain_client.cancelled.lock().unwrap(), vec![leaving_id]); + } + + #[test] + fn an_agreement_that_ended_since_it_was_listed_is_not_a_failure() { + let agreement = crate::cancel_dispatch::tests::agreement( + crate::registry::IndexingAgreementStatus::AcceptedOnChain, + None, + ); + let backend_down = crate::registry::Error::BackendError(dipper_pgregistry::Error::DbError( + sqlx::Error::PoolTimedOut, + )); + + assert_eq!( + super::unmarked(&agreement, &crate::registry::Error::NoRecordsUpdated), + super::Unpaired::AlreadyEnding + ); + assert_eq!( + super::unmarked(&agreement, &backend_down), + super::Unpaired::Failed + ); + } +} + +/// What became of an old agreement taken out of the target group. +#[derive(Debug, PartialEq, Eq)] +enum Unpaired { + /// Moved to the status named. + Moved(&'static str), + /// Already ended or being cancelled, as a race with another path can leave it. + AlreadyEnding, + /// Couldn't be marked; logged. + Failed, +} + +/// Take an old agreement out of the target group. One that may be live on-chain is cancelled +/// there too, which the cancel retry re-sends until it ends. +async fn cancel_unpaired( + ctx: &Ctx, + agreement: &crate::registry::IndexingAgreement, +) -> Unpaired +where + R: AgreementRegistry + Sync, + T: ChainClient, +{ + let may_be_live = agreement.status == crate::registry::IndexingAgreementStatus::AcceptedOnChain + || (agreement.status == crate::registry::IndexingAgreementStatus::Created + && agreement.terms_version_hash.is_some()); + if !may_be_live { + return match ctx + .registry + .mark_indexing_agreement_as_canceled_by_requester(&agreement.id) + .await + { + Ok(()) => Unpaired::Moved("CANCELED_BY_REQUESTER"), + Err(err) => unmarked(agreement, &err), + }; + } + match crate::cancel_dispatch::start_cancel( + &ctx.registry, + &ctx.chain_client, + agreement, + crate::cancel_dispatch::CancelReason::NotWanted, + &ctx.agreement_conf, + ) + .await + { + Ok(crate::cancel_dispatch::CancelStarted::Ended) => { + Unpaired::Moved("CANCELED_BY_REQUESTER") + } + Ok( + crate::cancel_dispatch::CancelStarted::NotLive + | crate::cancel_dispatch::CancelStarted::MayBeLive, + ) => Unpaired::Moved("CANCELLING"), + Err(err) => unmarked(agreement, &err), + } +} + +/// Why an unpaired old agreement couldn't be marked, logged at the level it deserves: one that +/// ended, or started cancelling, since it was listed is expected and needs nothing more. +fn unmarked( + agreement: &crate::registry::IndexingAgreement, + err: &crate::registry::Error, +) -> Unpaired { + if matches!(err, crate::registry::Error::NoRecordsUpdated) { + tracing::debug!( + agreement_id = %agreement.id, + "Unpaired old agreement already ended or being cancelled" + ); + return Unpaired::AlreadyEnding; + } + tracing::error!( + error = %err, + agreement_id = %agreement.id, + "Failed to mark unpaired old agreement as ended in local DB" + ); + Unpaired::Failed } /// Olds reserved from cancellation: one per add-cancel pairing lost to a @@ -2267,7 +2327,7 @@ mod deadline_clock_tests { use super::{JobError, resolve_deadline_clock}; use crate::{ - chain_client::{ChainClient, ChainClientError}, + chain_client::{AgreementOnChain, ChainClient, ChainClientError}, network::service::{ chain_events::Cursor, chain_listener::{ChainListenerState, ChainListenerStateRegistry}, @@ -2318,11 +2378,11 @@ mod deadline_clock_tests { unimplemented!() } - async fn agreement_still_active( + async fn agreement_on_chain( &self, _agreement_id: &[u8; 16], - ) -> Result { - Ok(false) + ) -> Result { + Ok(AgreementOnChain::NotLive) } } diff --git a/bin/dipper-service/src/worker/handlers/selection_context.rs b/bin/dipper-service/src/worker/handlers/selection_context.rs index 0f65da62..3485463d 100644 --- a/bin/dipper-service/src/worker/handlers/selection_context.rs +++ b/bin/dipper-service/src/worker/handlers/selection_context.rs @@ -7,7 +7,9 @@ use thegraph_core::{DeploymentId, IndexerId, alloy::primitives::ChainId}; use crate::{ network::service::entity_count_cache::EntityCountCache, - registry::{AgreementRegistry, IndexerDenylistRegistry, IndexingAgreementStatus}, + registry::{ + AgreementRegistry, IndexerDenylistRegistry, IndexingAgreement, IndexingAgreementStatus, + }, worker::result::{JobError, JobResult}, }; @@ -42,11 +44,12 @@ where R: AgreementRegistry + IndexerDenylistRegistry, { // Get indexers that already have active agreements for this deployment - let existing_indexers = registry + let agreements = registry .get_indexing_agreements_by_deployment_id(deployment_id) .await - .map_err(|err| JobError::Fatal(err.into()))? - .into_iter() + .map_err(|err| JobError::Fatal(err.into()))?; + let existing_indexers = agreements + .iter() .filter(|a| is_active_agreement(&a.status)) .map(|a| a.indexer.id) .collect::>(); @@ -58,7 +61,7 @@ where .map_err(|err| JobError::Fatal(err.into()))?; // Get indexers that declined within their respective lookback periods - let declined_indexers = registry + let mut declined_indexers = registry .get_declined_indexers_by_deployment( declined_indexer_lookback_days, price_rejection_lookback_days, @@ -67,6 +70,7 @@ where ) .await .map_err(|err| JobError::Fatal(err.into()))?; + exclude_cancelling_indexers(&mut declined_indexers, *deployment_id, &agreements); // Get denied indexers that should be excluded from selection let indexer_denylist = registry @@ -173,6 +177,30 @@ fn wei_per_second_to_grt_per_28d(wei_per_second: f64) -> f64 { wei_per_second * SECONDS_PER_28_DAYS / WEI_PER_GRT } +/// Add to the deployment's declined list the indexers whose agreement dipper is still +/// cancelling. That agreement may still be live and paid on-chain, so its indexer must not +/// be picked again, but it no longer counts towards the group IISA sizes. +fn exclude_cancelling_indexers( + declined: &mut HashMap>, + deployment_id: DeploymentId, + agreements: &[IndexingAgreement], +) { + let cancelling = agreements + .iter() + .filter(|a| a.status == IndexingAgreementStatus::Cancelling) + .map(|a| a.indexer.id) + .collect::>(); + if cancelling.is_empty() { + return; + } + let excluded = declined.entry(deployment_id).or_default(); + for indexer in cancelling { + if !excluded.contains(&indexer) { + excluded.push(indexer); + } + } +} + /// Check if an agreement status represents an active agreement. fn is_active_agreement(status: &IndexingAgreementStatus) -> bool { matches!( @@ -184,7 +212,46 @@ fn is_active_agreement(status: &IndexingAgreementStatus) -> bool { #[cfg(test)] mod tests { use super::*; - use crate::registry::AgreementFeeRate; + use crate::{cancel_dispatch::tests::agreement, registry::AgreementFeeRate}; + + fn indexer(hex_digit: char) -> IndexerId { + format!("0x{}", hex_digit.to_string().repeat(40)) + .parse() + .unwrap() + } + + #[test] + fn an_indexer_still_being_cancelled_cannot_be_picked_again_for_the_deployment() { + let deployment: DeploymentId = "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" + .parse() + .unwrap(); + let mut cancelling = agreement(IndexingAgreementStatus::Cancelling, None); + cancelling.indexer.id = indexer('b'); + let mut accepted = agreement(IndexingAgreementStatus::AcceptedOnChain, None); + accepted.indexer.id = indexer('c'); + let mut declined = HashMap::from([(deployment, vec![indexer('a')])]); + + exclude_cancelling_indexers(&mut declined, deployment, &[cancelling, accepted]); + + assert_eq!(declined[&deployment], vec![indexer('a'), indexer('b')]); + } + + #[test] + fn a_cancelling_indexer_already_declined_is_listed_once_and_none_adds_no_entry() { + let deployment: DeploymentId = "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" + .parse() + .unwrap(); + let mut cancelling = agreement(IndexingAgreementStatus::Cancelling, None); + cancelling.indexer.id = indexer('a'); + let mut declined = HashMap::from([(deployment, vec![indexer('a')])]); + exclude_cancelling_indexers(&mut declined, deployment, &[cancelling]); + assert_eq!(declined[&deployment], vec![indexer('a')]); + + let mut none_declined = HashMap::new(); + let accepted = agreement(IndexingAgreementStatus::AcceptedOnChain, None); + exclude_cancelling_indexers(&mut none_declined, deployment, &[accepted]); + assert!(none_declined.is_empty()); + } #[test] fn test_wei_per_second_to_grt_per_28d() { diff --git a/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs b/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs index 1aa582ac..e3e6fbc4 100644 --- a/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs +++ b/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs @@ -415,7 +415,7 @@ mod tests { } #[async_trait] - impl AgreementRegistry for MockRegistry { + impl crate::registry::StubAgreementRegistry for MockRegistry { async fn get_indexing_agreement_by_id( &self, _id: &IndexingAgreementId, @@ -522,6 +522,13 @@ mod tests { Ok(()) } + async fn mark_indexing_agreement_as_cancelling( + &self, + _id: &IndexingAgreementId, + ) -> crate::registry::Result<()> { + Ok(()) + } + async fn apply_reconciliation( &self, _id: &IndexingAgreementId, @@ -600,13 +607,6 @@ mod tests { Ok((std::collections::HashMap::new(), 0)) } - async fn mark_indexing_agreement_as_abandoned( - &self, - _id: &IndexingAgreementId, - ) -> crate::registry::Result { - Err(crate::registry::Error::NoRecordsUpdated) - } - async fn get_agreement_fee_rates(&self) -> crate::registry::Result> { Ok(vec![]) } @@ -729,14 +729,6 @@ mod tests { Ok(JobId::default()) } - async fn cancel_rejected_agreement_on_chain( - &self, - _agreement_id: IndexingAgreementId, - _priority: JobPriority, - ) -> anyhow::Result { - Ok(JobId::default()) - } - async fn submit_offer( &self, _agreement_id: IndexingAgreementId, diff --git a/bin/dipper-service/src/worker/handlers/submit_offer.rs b/bin/dipper-service/src/worker/handlers/submit_offer.rs index 7eb0b263..83867064 100644 --- a/bin/dipper-service/src/worker/handlers/submit_offer.rs +++ b/bin/dipper-service/src/worker/handlers/submit_offer.rs @@ -16,16 +16,18 @@ //! `rcaOffers` mapping on `RecurringCollector` sits in an ERC-7201 namespaced storage //! struct with no public getter, so the chain cannot cheaply be asked what already landed. -use std::time::Duration; +use std::{sync::Arc, time::Duration}; use dipper_core::ids::{IndexingAgreementId, IndexingRequestId}; use thegraph_core::{DeploymentId, alloy::primitives::ChainId}; use url::Url; use crate::{ + cancel_dispatch::{LiveCancel, cancel_if_live, log_unconfirmed}, chain_client::{ChainClient, ChainClientError, decode_revert_reason}, + config::IndexingAgreementConfig, indexer_rpc_client::into_sol_rca, - registry::{AgreementRegistry, IndexingAgreementStatus}, + registry::{AgreementRegistry, IndexingAgreement, IndexingAgreementStatus}, worker::result::{JobError, JobResult}, }; @@ -38,6 +40,7 @@ pub const TRANSIENT_RETRY_BASE: Duration = Duration::from_secs(30); pub struct Ctx { pub registry: R, pub chain_client: T, + pub agreement_conf: Arc, } /// Submit an RCA offer on-chain. @@ -68,30 +71,12 @@ where R: AgreementRegistry, T: ChainClient, { - // Fetch the agreement. Skip silently if it's already been transitioned - // out of Created (e.g. expired by the reassignment service). - let agreement = match ctx - .registry - .get_indexing_agreement_by_id(agreement_id) - .await - .map_err(|err| JobError::Fatal(err.into()))? - { - None => { - tracing::error!( - agreement_id = %agreement_id, - "Agreement not found in registry at submit_offer" - ); - return Ok(()); + let agreement = match next_step(&ctx.registry, agreement_id).await? { + NextStep::Offer(agreement) => agreement, + NextStep::Withdraw(agreement) => { + return withdraw_offer_if_stored(&ctx, &agreement).await; } - Some(a) if a.status != IndexingAgreementStatus::Created => { - tracing::warn!( - agreement_id = %agreement_id, - status = %a.status, - "Agreement not in Created status, skipping offer submission" - ); - return Ok(()); - } - Some(a) => a, + NextStep::Skip => return Ok(()), }; // Rebuild the on-chain RCA struct from the stored terms. The bytes must be @@ -136,19 +121,24 @@ where tx_hash = %tx_hash, "Offer submitted on-chain successfully" ); - // Observability only: record which tx hash actually mined. // Any failure here is non-fatal to the overall flow. - if let Err(err) = ctx + match ctx .registry .update_offer_tx_hash(agreement_id, tx_hash.as_ref()) .await { - tracing::warn!( + Ok(()) => {} + Err(crate::registry::Error::NoRecordsUpdated) => tracing::debug!( + agreement_id = %agreement_id, + tx_hash = %tx_hash, + "Agreement ended while its offer was mining, so its offer_tx_hash isn't stored" + ), + Err(err) => tracing::warn!( agreement_id = %agreement_id, tx_hash = %tx_hash, error = %err, "Failed to persist offer_tx_hash; continuing" - ); + ), } } Err(err @ ChainClientError::TxDropped { .. }) => { @@ -189,15 +179,137 @@ where } } + // Every cancel of an unaccepted agreement marks it before it is sent, so a cancel + // that went out ahead of this offer, and found nothing to withdraw, shows here. + if let Some(agreement) = cancelled_meanwhile(&ctx.registry, agreement_id).await { + return withdraw_offer_if_stored(&ctx, &agreement).await; + } + // Offer is confirmed on-chain (or was already there). The indexer-agent will // pick up the pending_rca_proposals row and call acceptIndexingAgreement. No // further enqueue needed; chain_listener detects the acceptance event. Ok(()) } +/// What this run of the job does with its agreement. +enum NextStep { + /// Still wanted: send the offer. + Offer(IndexingAgreement), + /// Dipper cancelled it: withdraw any offer an earlier attempt left on-chain. + Withdraw(IndexingAgreement), + /// Gone, expired or otherwise past offering. + Skip, +} + +async fn next_step( + registry: &R, + agreement_id: &IndexingAgreementId, +) -> JobResult { + let agreement = registry + .get_indexing_agreement_by_id(agreement_id) + .await + .map_err(|err| JobError::Fatal(err.into()))?; + Ok(match agreement { + None => { + tracing::error!( + agreement_id = %agreement_id, + "Agreement not found in registry at submit_offer" + ); + NextStep::Skip + } + Some(a) if a.status == IndexingAgreementStatus::Created => NextStep::Offer(a), + Some(a) if dipper_cancelled(a.status) => NextStep::Withdraw(a), + Some(a) => { + tracing::warn!( + agreement_id = %agreement_id, + status = %a.status, + "Agreement not in Created status, skipping offer submission" + ); + NextStep::Skip + } + }) +} + +/// Withdraw the agreement's offer if one is on-chain: dipper cancelled it while +/// this job's offer was in flight or before this retry. A failure retries the job, +/// which comes back here through its status check. +async fn withdraw_offer_if_stored( + ctx: &Ctx, + agreement: &IndexingAgreement, +) -> JobResult<()> { + match cancel_if_live(&ctx.chain_client, agreement, &ctx.agreement_conf).await { + LiveCancel::NotLive { .. } => Ok(()), + LiveCancel::Ended(tx_hash) => { + tracing::info!( + agreement_id = %agreement.id, + tx_hash = ?tx_hash, + "Withdrew the offer of an agreement dipper had cancelled" + ); + Ok(()) + } + LiveCancel::CancelFailed(err @ ChainClientError::MissingTermsVersionHash { .. }) => { + tracing::error!( + agreement_id = %agreement.id, + error = %err, + "Cannot withdraw the offer of a cancelled agreement; it stays open until its deadline" + ); + Err(JobError::Fatal(err.into())) + } + LiveCancel::Unconfirmed { tx_hash, err } => { + log_unconfirmed(agreement, tx_hash, &err); + Err(retry_withdraw(agreement, err)) + } + LiveCancel::ReadFailed(err) | LiveCancel::CancelFailed(err) => { + Err(retry_withdraw(agreement, err)) + } + } +} + +fn retry_withdraw(agreement: &IndexingAgreement, err: ChainClientError) -> JobError { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to withdraw the offer of a cancelled agreement, will retry" + ); + JobError::Retryable(err.into(), TRANSIENT_RETRY_BASE) +} + +/// Whether dipper has cancelled the agreement, or started to. +fn dipper_cancelled(status: IndexingAgreementStatus) -> bool { + matches!( + status, + IndexingAgreementStatus::Cancelling | IndexingAgreementStatus::CanceledByRequester + ) +} + +/// The agreement, if it was cancelled after this job's status check. After a failed read +/// the job still finishes: retrying would send the offer again, and the cancel retry +/// withdraws the offer of an agreement left cancelling. +async fn cancelled_meanwhile( + registry: &R, + agreement_id: &IndexingAgreementId, +) -> Option { + match registry.get_indexing_agreement_by_id(agreement_id).await { + Ok(Some(agreement)) if dipper_cancelled(agreement.status) => Some(agreement), + Ok(_) => None, + Err(err) => { + tracing::warn!( + agreement_id = %agreement_id, + error = %err, + "Failed to re-read agreement after its offer landed; the cancel retry \ + withdraws the offer if it was cancelled" + ); + None + } + } +} + #[cfg(test)] mod tests { - use std::sync::Mutex; + use std::sync::{ + Arc, Mutex, + atomic::{AtomicBool, AtomicU32, Ordering}, + }; use async_trait::async_trait; use thegraph_core::{ @@ -207,6 +319,7 @@ mod tests { use super::*; use crate::{ + chain_client::AgreementOnChain, indexer_rpc_client::compute_on_chain_id, registry::{ IndexingAgreement, IndexingAgreementTerms, IndexingAgreementTermsMetadata, @@ -214,8 +327,14 @@ mod tests { }, }; + /// Shared with the chain mock so a test can change the row mid-send. + type SharedAgreement = Arc>>; + struct MockRegistry { - agreement: Option, + agreement: SharedAgreement, + /// When set, every read after the first fails. + later_reads_fail: bool, + reads: AtomicU32, } #[async_trait] @@ -224,7 +343,10 @@ mod tests { &self, _id: &IndexingAgreementId, ) -> crate::registry::Result> { - Ok(self.agreement.clone()) + if self.reads.fetch_add(1, Ordering::SeqCst) > 0 && self.later_reads_fail { + return Err(crate::registry::Error::NoRecordsUpdated); + } + Ok(self.agreement.lock().unwrap().clone()) } async fn update_offer_tx_hash( &self, @@ -236,9 +358,18 @@ mod tests { } /// Yields the configured result once; a second call means the handler - /// retried inside one run, which must never happen. + /// retried inside one run, and a call with no result configured means the + /// handler sent when it must not. struct MockChainClient { offer_result: Mutex, ChainClientError>>>, + /// When set, the agreement is cancelled locally while the offer is sent, + /// as the chain listener does when a replacement is accepted. + cancel_mid_send: Option, + cancelled: Arc>>, + /// Whether the agreement's offer (or the agreement) is live on-chain: set + /// by a mined offer, cleared by a cancel. + on_chain: Arc, + fail_cancel: bool, } #[async_trait] @@ -247,20 +378,35 @@ mod tests { &self, _rca: &dipper_rpc::indexer::indexer_client::sol::RecurringCollectionAgreement, ) -> Result, ChainClientError> { - self.offer_result + if let Some(agreement) = &self.cancel_mid_send + && let Some(row) = agreement.lock().unwrap().as_mut() + { + row.status = IndexingAgreementStatus::Cancelling; + } + let result = self + .offer_result .lock() .unwrap() .take() - .expect("offer_via_manager called more than once") + .expect("offer_via_manager called more than once, or when it must not send"); + if matches!(result, Ok(Some(_))) { + self.on_chain.store(true, Ordering::SeqCst); + } + result } async fn cancel_via_manager( &self, _collector: Address, - _agreement_id: &[u8; 16], + agreement_id: &[u8; 16], _version_hash: B256, _options: u16, ) -> Result, ChainClientError> { - unimplemented!() + if self.fail_cancel { + return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); + } + self.cancelled.lock().unwrap().push(*agreement_id); + self.on_chain.store(false, Ordering::SeqCst); + Ok(Some(B256::repeat_byte(0xcd))) } async fn reconcile_provider( &self, @@ -276,11 +422,13 @@ mod tests { ) -> Result, ChainClientError> { unimplemented!() } - async fn agreement_still_active( + async fn agreement_on_chain( &self, _agreement_id: &[u8; 16], - ) -> Result { - unimplemented!() + ) -> Result { + Ok(AgreementOnChain::live_if( + self.on_chain.load(Ordering::SeqCst), + )) } async fn latest_block_timestamp(&self) -> Result { unimplemented!() @@ -348,17 +496,151 @@ mod tests { fn ctx_with_offer_result( agreement: IndexingAgreement, offer_result: Result, ChainClientError>, + ) -> Ctx { + ctx_with(agreement, Some(offer_result)) + } + + /// `offer_result: None` makes any send panic. + fn ctx_with( + agreement: IndexingAgreement, + offer_result: Option, ChainClientError>>, ) -> Ctx { Ctx { registry: MockRegistry { - agreement: Some(agreement), + agreement: Arc::new(Mutex::new(Some(agreement))), + later_reads_fail: false, + reads: AtomicU32::new(0), }, chain_client: MockChainClient { - offer_result: Mutex::new(Some(offer_result)), + offer_result: Mutex::new(offer_result), + cancel_mid_send: None, + cancelled: Arc::default(), + on_chain: Arc::default(), + fail_cancel: false, }, + agreement_conf: Arc::new(test_agreement_conf()), + } + } + + fn test_agreement_conf() -> crate::config::IndexingAgreementConfig { + crate::config::IndexingAgreementConfig::for_tests() + } + + #[tokio::test] + async fn withdraws_its_offer_when_the_agreement_was_cancelled_while_it_was_sent() { + //* Arrange - the agreement is cancelled locally while the offer is in flight, + // so the cancel found nothing to withdraw and the offer would stay open + let mut agreement = make_test_agreement(); + agreement.terms_version_hash = Some(vec![7u8; 32]); + let agreement_id = agreement.id; + let message = make_message(agreement_id); + let mut ctx = ctx_with_offer_result(agreement, Ok(Some(B256::repeat_byte(0xab)))); + ctx.chain_client.cancel_mid_send = Some(ctx.registry.agreement.clone()); + let cancelled = ctx.chain_client.cancelled.clone(); + + //* Act + let result = handle(ctx, &message).await; + + //* Assert + assert!(result.is_ok(), "got {result:?}"); + assert_eq!(*cancelled.lock().unwrap(), vec![*agreement_id.as_bytes()]); + } + + #[tokio::test] + async fn finishes_without_resending_when_it_cannot_recheck_the_agreement() { + //* Arrange - a retry would send the offer again; the mock panics on a second send + let agreement = make_test_agreement(); + let message = make_message(agreement.id); + let mut ctx = ctx_with_offer_result(agreement, Ok(Some(B256::repeat_byte(0xab)))); + ctx.registry.later_reads_fail = true; + + //* Act + let result = handle(ctx, &message).await; + + //* Assert + assert!(result.is_ok(), "got {result:?}"); + } + + #[tokio::test] + async fn keeps_its_offer_when_the_agreement_is_still_wanted() { + //* Arrange + let agreement = make_test_agreement(); + let message = make_message(agreement.id); + let ctx = ctx_with_offer_result(agreement, Ok(Some(B256::repeat_byte(0xab)))); + let cancelled = ctx.chain_client.cancelled.clone(); + + //* Act + let result = handle(ctx, &message).await; + + //* Assert + assert!(result.is_ok(), "got {result:?}"); + assert!(cancelled.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn skips_an_agreement_a_reassessment_already_cancelled() { + //* Arrange - the reassessment ran first and cancelled the agreement; no offer + // result is configured, so a send would panic + let mut agreement = make_test_agreement(); + agreement.status = IndexingAgreementStatus::CanceledByRequester; + let message = make_message(agreement.id); + let ctx = ctx_with(agreement, None); + let cancelled = ctx.chain_client.cancelled.clone(); + + //* Act + let result = handle(ctx, &message).await; + + //* Assert - and with no offer on-chain, nothing to withdraw + assert!(result.is_ok(), "got {result:?}"); + assert!(cancelled.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn withdraws_a_stored_offer_of_an_agreement_already_cancelled() { + for status in [ + IndexingAgreementStatus::Cancelling, + IndexingAgreementStatus::CanceledByRequester, + ] { + //* Arrange - an earlier attempt sent the offer, then the agreement was + // cancelled before this retry; no offer result, so a send would panic + let mut agreement = make_test_agreement(); + agreement.status = status; + agreement.terms_version_hash = Some(vec![7u8; 32]); + let agreement_id = agreement.id; + let message = make_message(agreement_id); + let ctx = ctx_with(agreement, None); + ctx.chain_client.on_chain.store(true, Ordering::SeqCst); + let cancelled = ctx.chain_client.cancelled.clone(); + + //* Act + let result = handle(ctx, &message).await; + + //* Assert + assert!(result.is_ok(), "got {result:?}"); + assert_eq!(*cancelled.lock().unwrap(), vec![*agreement_id.as_bytes()]); } } + #[tokio::test] + async fn retries_a_withdraw_that_fails() { + //* Arrange - cancelled while the offer was in flight, and the withdraw fails + let mut agreement = make_test_agreement(); + agreement.terms_version_hash = Some(vec![7u8; 32]); + let message = make_message(agreement.id); + let mut ctx = ctx_with_offer_result(agreement, Ok(Some(B256::repeat_byte(0xab)))); + ctx.chain_client.cancel_mid_send = Some(ctx.registry.agreement.clone()); + ctx.chain_client.fail_cancel = true; + + //* Act + let result = handle(ctx, &message).await; + + //* Assert - a retry finds the row cancelled and withdraws through the check + assert!( + matches!(result, Err(JobError::Retryable(_, _))), + "got {result:?}" + ); + } + #[tokio::test] async fn contract_revert_fails_the_job_instead_of_retrying() { //* Arrange - the offer reverts with the observed window selector diff --git a/bin/dipper-service/src/worker/service.rs b/bin/dipper-service/src/worker/service.rs index 7605945f..659ca90c 100644 --- a/bin/dipper-service/src/worker/service.rs +++ b/bin/dipper-service/src/worker/service.rs @@ -619,14 +619,14 @@ where } } Err(JobError::Deferred(delay)) => { - // Couldn't run now (another reassessment holds the global lock); - // re-queue at a flat delay without counting a failed attempt. - // Logged at info so sustained contention is visible per job id. + // Couldn't run now (the global reassess lock is busy, or pacing + // held it back); re-queue at a flat delay without counting a failed + // attempt. Logged at info so sustained contention is visible per job id. let scheduled_for = OffsetDateTime::now_utc() + delay; tracing::info!( job = %job.id(), delay_secs = %delay.as_secs(), - "Deferring job; another reassessment holds the global lock, will retry" + "Deferring job; it can't run yet, will retry" ); if let Err(err) = job.reschedule(scheduled_for).await { tracing::error!(error=?err, "Failed to reschedule deferred job"); diff --git a/bin/dipper-service/src/worker/service_queue.rs b/bin/dipper-service/src/worker/service_queue.rs index 064928d8..49ad37e1 100644 --- a/bin/dipper-service/src/worker/service_queue.rs +++ b/bin/dipper-service/src/worker/service_queue.rs @@ -4,10 +4,7 @@ use thegraph_core::{DeploymentId, alloy::primitives::ChainId}; use url::Url; use super::{ - handlers::{ - CancelRejectedAgreementOnChain, ReassessIndexingRequest, SendIndexingAgreementProposal, - SubmitOffer, - }, + handlers::{ReassessIndexingRequest, SendIndexingAgreementProposal, SubmitOffer}, messages::Message, queue::{JobId, JobPriority, Queue}, }; @@ -33,15 +30,6 @@ pub trait WorkerQueue { priority: JobPriority, ) -> anyhow::Result; - /// Cancel a rejected agreement on-chain. When an indexer rejected off-chain - /// but accepted on-chain, this cancels the agreement via - /// `cancelIndexingAgreementByPayer`. - async fn cancel_rejected_agreement_on_chain( - &self, - agreement_id: IndexingAgreementId, - priority: JobPriority, - ) -> anyhow::Result; - /// Submit an RCA offer on-chain as the first step of a new proposal. The /// job retries until the indexer's window to accept has closed, so a /// provider outage costs one offer only if it outlasts that window. @@ -124,21 +112,6 @@ where .await } - async fn cancel_rejected_agreement_on_chain( - &self, - agreement_id: IndexingAgreementId, - priority: JobPriority, - ) -> anyhow::Result { - self.queue - .push( - Message::CancelRejectedAgreementOnChain(CancelRejectedAgreementOnChain { - agreement_id, - }), - priority, - ) - .await - } - async fn submit_offer( &self, agreement_id: IndexingAgreementId, @@ -246,10 +219,10 @@ mod tests { assert_eq!(*queue.queue.pushes.lock().unwrap(), vec![Some(4)]); } - /// Only the offer submission has a deadline to spend its retries against, - /// so every other job keeps the queue-wide budget. + /// A proposal has no deadline of its own to spend retries against, so it + /// keeps the queue-wide budget. #[tokio::test] - async fn other_jobs_keep_the_queue_default_retry_budget() { + async fn a_proposal_keeps_the_queue_default_retry_budget() { //* Arrange let queue = handle(4); @@ -265,15 +238,8 @@ mod tests { ) .await .unwrap(); - queue - .cancel_rejected_agreement_on_chain( - IndexingAgreementId::from_bytes([0; 16]), - JobPriority::Background, - ) - .await - .unwrap(); //* Assert - assert_eq!(*queue.queue.pushes.lock().unwrap(), vec![None, None]); + assert_eq!(*queue.queue.pushes.lock().unwrap(), vec![None]); } } diff --git a/dipper-pgregistry/migrations/20261002000000_add_cancelling_status.sql b/dipper-pgregistry/migrations/20261002000000_add_cancelling_status.sql new file mode 100644 index 00000000..e9c86d29 --- /dev/null +++ b/dipper-pgregistry/migrations/20261002000000_add_cancelling_status.sql @@ -0,0 +1,15 @@ +-- Cancelling (status = 9): dipper has decided to end the agreement and keeps sending the +-- on-chain cancel until the chain confirms it ended; only then does the row become +-- CanceledByRequester, which announces the end. +-- +-- cancel_attempts counts cancels that reached the chain without ending the agreement, +-- so one that can never work stops being retried and is left for an operator. +-- cancel_checked_at lets each retry sweep start with the agreements checked longest ago. +ALTER TABLE dipper_reg_indexing_agreements + ADD COLUMN cancel_attempts INTEGER NOT NULL DEFAULT 0, + ADD COLUMN cancel_checked_at TIMESTAMPTZ; + +-- Partial, so it only covers agreements still being cancelled (normally none). +CREATE INDEX idx_indexing_agreements_cancelling + ON dipper_reg_indexing_agreements (cancel_checked_at NULLS FIRST, updated_at) + WHERE status = 9; diff --git a/dipper-pgregistry/migrations/20261005000000_add_cancel_ended_seen_at.sql b/dipper-pgregistry/migrations/20261005000000_add_cancel_ended_seen_at.sql new file mode 100644 index 00000000..fdd6df7a --- /dev/null +++ b/dipper-pgregistry/migrations/20261005000000_add_cancel_ended_seen_at.sql @@ -0,0 +1,5 @@ +-- ended_seen_at: when dipper's cancel retry first found a Cancelling agreement no longer live +-- on-chain. Where the retry has no cancel of its own to show for the end, the chain listener +-- gets an hour from then to record how it ended before the retry closes it out without that. +ALTER TABLE dipper_reg_indexing_agreements + ADD COLUMN ended_seen_at TIMESTAMPTZ; diff --git a/dipper-pgregistry/migrations/20261005000001_add_abandoned_flag.sql b/dipper-pgregistry/migrations/20261005000001_add_abandoned_flag.sql new file mode 100644 index 00000000..87fd09c3 --- /dev/null +++ b/dipper-pgregistry/migrations/20261005000001_add_abandoned_flag.sql @@ -0,0 +1,5 @@ +-- abandoned: dipper is cancelling the agreement because its indexer stopped serving it, so once +-- the chain confirms dipper's cancel it ends AbandonedByIndexer (8) rather than +-- CanceledByRequester (3). +ALTER TABLE dipper_reg_indexing_agreements + ADD COLUMN abandoned BOOLEAN NOT NULL DEFAULT false; diff --git a/dipper-pgregistry/migrations/20261008000000_add_replacement_pending.sql b/dipper-pgregistry/migrations/20261008000000_add_replacement_pending.sql new file mode 100644 index 00000000..bf58fbb4 --- /dev/null +++ b/dipper-pgregistry/migrations/20261008000000_add_replacement_pending.sql @@ -0,0 +1,8 @@ +-- replacement_pending: the agreement's indexer stopped serving it and dipper has yet to queue a +-- reassessment to replace it, which waits until the agreement can no longer be paid. +ALTER TABLE dipper_reg_indexing_agreements + ADD COLUMN replacement_pending BOOLEAN NOT NULL DEFAULT false; + +CREATE INDEX idx_indexing_agreements_replacement_pending + ON dipper_reg_indexing_agreements (updated_at) + WHERE replacement_pending; diff --git a/dipper-pgregistry/src/indexing_agreement.rs b/dipper-pgregistry/src/indexing_agreement.rs index 53c20c32..241f8419 100644 --- a/dipper-pgregistry/src/indexing_agreement.rs +++ b/dipper-pgregistry/src/indexing_agreement.rs @@ -200,12 +200,17 @@ pub enum Status { /// The liveness checker detected no indexing progress within the tolerance window. /// - /// Dipper canceled the agreement via `cancelIndexingAgreementByPayer` and will - /// trigger reassignment to find a replacement indexer. + /// Dipper cancelled the agreement on-chain, passing through `Cancelling` until the chain + /// confirmed it, and triggered reassignment to find a replacement indexer. /// /// This is a terminal state. AbandonedByIndexer = 8, + /// Dipper decided to end the agreement and is cancelling it on-chain, where it may + /// still be live. It becomes `CanceledByRequester`, or `AbandonedByIndexer` when its + /// indexer stopped serving it, announced as ended, only once the chain confirms the end. + Cancelling = 9, + /// A fallback for unknown status values. Unknown = i32::MAX, } @@ -221,6 +226,7 @@ impl std::fmt::Display for Status { Status::AcceptedOnChain => "ACCEPTED_ON_CHAIN", Status::Rejected => "REJECTED", Status::AbandonedByIndexer => "ABANDONED_BY_INDEXER", + Status::Cancelling => "CANCELLING", Status::Unknown => "UNKNOWN", }; f.write_str(status) diff --git a/dipper-pgregistry/src/lib.rs b/dipper-pgregistry/src/lib.rs index 20b40ea6..8e0ea869 100644 --- a/dipper-pgregistry/src/lib.rs +++ b/dipper-pgregistry/src/lib.rs @@ -20,9 +20,9 @@ pub use indexing_request::{ Status as IndexingRequestStatus, }; pub use postgres::{ - CancelKind, ChainListenerStateRow, NewAgreementParams, PendingAcceptedEvent, - PendingExpiredEvent, PendingTerminatedEvent, PgRegistry, ReconciliationAudit, - ReconciliationItem, ReconciliationOutcome, + CancelKind, CancellingAgreement, ChainListenerStateRow, NewAgreementParams, + PendingAcceptedEvent, PendingExpiredEvent, PendingTerminatedEvent, PgRegistry, + ReconciliationAudit, ReconciliationItem, ReconciliationOutcome, }; pub use result::{Error, Result}; diff --git a/dipper-pgregistry/src/postgres.rs b/dipper-pgregistry/src/postgres.rs index 14c85160..e9b7411e 100644 --- a/dipper-pgregistry/src/postgres.rs +++ b/dipper-pgregistry/src/postgres.rs @@ -187,6 +187,50 @@ impl sqlx::FromRow<'_, sqlx::postgres::PgRow> for PendingAcceptedEvent { } } +/// An agreement dipper is still cancelling on-chain. +#[derive(Debug, Clone)] +pub struct CancellingAgreement { + pub agreement: IndexingAgreement, + /// Whether dipper saw it accepted on-chain, so its end is announced. + pub accepted_on_chain: bool, + /// When a check first found it no longer live on-chain, if one has. + pub ended_seen_at: Option, + /// Whether it is being cancelled because its indexer stopped serving it. + pub abandoned: bool, +} + +impl sqlx::FromRow<'_, sqlx::postgres::PgRow> for CancellingAgreement { + fn from_row(row: &sqlx::postgres::PgRow) -> Result { + use sqlx::Row as _; + let accepted_at: Option = row.try_get("accepted_at")?; + Ok(Self { + agreement: IndexingAgreement::from_row(row)?, + accepted_on_chain: accepted_at.is_some(), + ended_seen_at: row.try_get("ended_seen_at")?, + abandoned: row.try_get("abandoned")?, + }) + } +} + +/// How long an ended agreement's `terminated` event waits for the transaction that ended it. +/// Dipper can mark an agreement ended before the chain listener records that transaction, so +/// the wait lets the event carry it; after this the event goes out without one. +const TERMINATED_TX_WAIT_MINUTES: i32 = 60; + +/// Statuses an on-chain cancel by dipper ends. +const CANCEL_BY_REQUESTER_FROM: &[IndexingAgreementStatus] = &[ + IndexingAgreementStatus::Created, + IndexingAgreementStatus::AcceptedOnChain, + IndexingAgreementStatus::Rejected, + IndexingAgreementStatus::Cancelling, +]; + +/// Statuses an on-chain cancel by the indexer ends. +const CANCEL_BY_INDEXER_FROM: &[IndexingAgreementStatus] = &[ + IndexingAgreementStatus::AcceptedOnChain, + IndexingAgreementStatus::Cancelling, +]; + /// A row that needs a `request.expired` lifecycle event emitted. Sourced from /// the agreement row alone; `request_expired_at` is the terms deadline (the true /// expiry instant), so the sweep needs no chain-time snapshot. @@ -699,9 +743,10 @@ impl PgRegistry { .collect()) } - /// Get declined `CanceledByIndexer`/`Expired`/`Rejected` indexers grouped by - /// deployment (deployment id -> indexer ids). Each rejection reason gets its own - /// exclusion window, as does an expiry that never had an offer transaction. + /// Get declined `CanceledByIndexer`/`Expired`/`Rejected` indexers, and those whose agreement + /// dipper ended `AbandonedByIndexer`, grouped by deployment (deployment id -> indexer ids). + /// Each rejection reason gets its own exclusion window, as does an expiry that never had an + /// offer transaction. pub async fn get_declined_indexers_by_deployment( &self, default_lookback_days: i32, @@ -722,7 +767,7 @@ impl PgRegistry { deployment_id, array_agg(DISTINCT indexer_id) as indexer_ids FROM dipper_reg_indexing_agreements - WHERE status IN ($1, $2, $3) + WHERE status IN ($1, $2, $3, $20) AND ( -- PRICE_TOO_LOW: shorter lookback (until next IISA refresh) (rejection_reason = $6 @@ -777,6 +822,7 @@ impl PgRegistry { .bind(uncertain_lookback_days) // $17 .bind(SENDER_NOT_TRUSTED) // $18 .bind(UNSPECIFIED) // $19 + .bind(IndexingAgreementStatus::AbandonedByIndexer) // $20 .fetch_all(&self.pool) .await?; @@ -873,69 +919,252 @@ impl PgRegistry { Ok(()) } - /// Persist the on-chain tx hash of the most recent `offer()` submission - /// for this agreement. Overwrites any prior value, so a resubmit after - /// mempool eviction records the live hash rather than the dropped one. - /// Observability-only: no status transition is performed here. - /// - /// Guarded on `status IN (Created, AcceptedOnChain)` so a delayed - /// receipt-confirmation cannot stamp `offer_tx_hash` onto a row that - /// has since transitioned to `Expired`, `Unresponsive`, `Rejected`, - /// or one of the cancel states. The caller treats any failure here - /// as non-fatal and just logs; a no-match result is also non-fatal - /// and silently skipped. + /// Record the hash of the latest `offer()` transaction, unless the agreement has ended. A + /// `Cancelling` row keeps its `updated_at`, which says when it was marked and paces its + /// cancel retry. Returns [`Error::NoRecordsUpdated`] when no row took the hash. pub async fn update_offer_tx_hash( &self, agreement_id: &IndexingAgreementId, tx_hash: &[u8; 32], ) -> Result<(), Error> { - sqlx::query( + let updated = sqlx::query( r#" UPDATE dipper_reg_indexing_agreements SET offer_tx_hash = $1, - updated_at = timezone('UTC', now()) - WHERE id = $2 AND status IN ($3, $4) + updated_at = CASE WHEN status = $5 THEN updated_at + ELSE timezone('UTC', now()) END + WHERE id = $2 AND status IN ($3, $4, $5) "#, ) .bind(&tx_hash[..]) .bind(agreement_id) .bind(IndexingAgreementStatus::Created) .bind(IndexingAgreementStatus::AcceptedOnChain) + .bind(IndexingAgreementStatus::Cancelling) .execute(&self.pool) .await?; + if updated.rows_affected() == 0 { + return Err(Error::NoRecordsUpdated); + } Ok(()) } + /// One being cancelled because its indexer stopped serving it ends `AbandonedByIndexer`. pub async fn mark_indexing_agreement_as_canceled_by_requester( &self, agreement_id: &IndexingAgreementId, ) -> Result<(), Error> { - let record: Option<(IndexingAgreementId,)> = sqlx::query_as( + self.set_status_from( + agreement_id, + IndexingAgreementStatus::CanceledByRequester, + CANCEL_BY_REQUESTER_FROM, + ) + .await + } + + /// Mark an agreement that may be live on-chain `Cancelling`, before dipper sends its + /// on-chain cancel. One marked `Expired` may have been accepted unseen by a lagging listener. + pub async fn mark_indexing_agreement_as_cancelling( + &self, + agreement_id: &IndexingAgreementId, + ) -> Result<(), Error> { + self.set_status_from( + agreement_id, + IndexingAgreementStatus::Cancelling, + &[ + IndexingAgreementStatus::Created, + IndexingAgreementStatus::AcceptedOnChain, + IndexingAgreementStatus::Rejected, + IndexingAgreementStatus::Expired, + ], + ) + .await + } + + /// Start ending an accepted agreement whose indexer stopped serving it: `Cancelling`, and + /// noted as abandoned, so the chain confirming dipper's cancel ends it `AbandonedByIndexer`. + pub async fn mark_indexing_agreement_as_abandoning( + &self, + agreement_id: &IndexingAgreementId, + ) -> Result<(), Error> { + let updated = sqlx::query( r#" UPDATE dipper_reg_indexing_agreements SET status = $1, + abandoned = true, + replacement_pending = true, updated_at = timezone('UTC', now()) - WHERE id = $2 AND status IN ($3, $4, $5) - RETURNING id + WHERE id = $2 AND status = $3 "#, ) - .bind(IndexingAgreementStatus::CanceledByRequester) + .bind(IndexingAgreementStatus::Cancelling) .bind(agreement_id) - .bind(IndexingAgreementStatus::Created) .bind(IndexingAgreementStatus::AcceptedOnChain) - .bind(IndexingAgreementStatus::Rejected) - .fetch_optional(&self.pool) + .execute(&self.pool) .await?; - - if record.is_none() { + if updated.rows_affected() == 0 { return Err(Error::NoRecordsUpdated); } + Ok(()) + } + /// Move an agreement dipper had already ended, cancelled or rejected, back to `Cancelling` + /// once the chain shows it live after all, with its cancel attempts started afresh. It counts + /// as checked, since no cancel is sent with it, so the retry takes it on its next sweep rather + /// than waiting for one to be mined. When the chain was read and showed it live (`seen_live`), + /// the end on record, and any announcement of it, no longer stands, so both are cleared for + /// the end still to come; an unread chain leaves them, as the agreement may have ended. + pub async fn reopen_indexing_agreement_cancel( + &self, + agreement_id: &IndexingAgreementId, + seen_live: bool, + ) -> Result<(), Error> { + let updated = sqlx::query( + r#" + UPDATE dipper_reg_indexing_agreements + SET + status = $1, + cancel_attempts = 0, + cancel_checked_at = timezone('UTC', now()), + ended_seen_at = NULL, + canceled_at = CASE WHEN $5::BOOLEAN THEN NULL ELSE canceled_at END, + canceled_by = CASE WHEN $5 THEN NULL ELSE canceled_by END, + canceled_tx = CASE WHEN $5 THEN NULL ELSE canceled_tx END, + terminated_event_emitted_at = + CASE WHEN $5 THEN NULL ELSE terminated_event_emitted_at END, + updated_at = timezone('UTC', now()) + WHERE id = $2 AND status IN ($3, $4) + "#, + ) + .bind(IndexingAgreementStatus::Cancelling) + .bind(agreement_id) + .bind(IndexingAgreementStatus::CanceledByRequester) + .bind(IndexingAgreementStatus::Rejected) + .bind(seen_live) + .execute(&self.pool) + .await?; + if updated.rows_affected() == 0 { + return Err(Error::NoRecordsUpdated); + } Ok(()) } + /// `Cancelling` agreements, those checked longest ago first, leaving out any never checked + /// that was marked in the last `min_age_minutes`, so the cancel sent with its mark can be + /// mined first; one whose cancel has failed `max_attempts` times only once an hour. One that may be paying + /// an indexer (accepted, or past the offer deadline, which only an accepted one outlives) + /// counts as checked an hour earlier, so it goes first without holding the rest back. One + /// never checked counts as checked when it was marked, so a burst of new ones can't jump it. + pub async fn get_cancelling_agreements( + &self, + batch_size: i64, + max_attempts: u32, + min_age_minutes: i32, + ) -> Result, Error> { + sqlx::query_as( + r#" + SELECT + id, + nonce_uuid, + created_at, + updated_at, + status, + indexing_request_id, + deployment_id, + indexer_id, + indexer_url, + terms, + last_block_height, + last_progress_at, + rejection_reason, + terms_version_hash, + accepted_at, + ended_seen_at, + abandoned + FROM dipper_reg_indexing_agreements + WHERE status = $1 + AND ( + cancel_attempts < $2 + OR cancel_checked_at < timezone('UTC', now()) - INTERVAL '1 hour' + ) + AND ( + cancel_checked_at IS NOT NULL + OR updated_at < timezone('UTC', now()) - make_interval(mins => $4) + ) + ORDER BY + COALESCE(cancel_checked_at, updated_at) - CASE + WHEN accepted_at IS NOT NULL + OR CAST(terms->>'deadline' AS bigint) < EXTRACT(EPOCH FROM now()) + THEN INTERVAL '1 hour' + ELSE INTERVAL '0 seconds' + END ASC, + updated_at ASC + LIMIT $3 + "#, + ) + .bind(IndexingAgreementStatus::Cancelling) + .bind(i32::try_from(max_attempts).unwrap_or(i32::MAX)) + .bind(batch_size) + .bind(min_age_minutes) + .fetch_all(&self.pool) + .await + .map_err(Into::into) + } + + /// Record a check of a `Cancelling` agreement that left it cancelling, adding + /// `failed_attempts` to its failed cancels and returning the new count. `ended` says whether + /// the check found it no longer live on-chain, or `None` when the chain couldn't tell; the + /// first time it is found ended is kept until it is found live again. + pub async fn record_cancel_check( + &self, + agreement_id: &IndexingAgreementId, + failed_attempts: u32, + ended: Option, + ) -> Result { + let record: Option<(i32,)> = sqlx::query_as( + r#" + UPDATE dipper_reg_indexing_agreements + SET + cancel_attempts = LEAST(cancel_attempts::BIGINT + $3, 2147483647)::INTEGER, + cancel_checked_at = timezone('UTC', now()), + ended_seen_at = CASE + WHEN $4::BOOLEAN IS NULL THEN ended_seen_at + WHEN $4 THEN COALESCE(ended_seen_at, timezone('UTC', now())) + ELSE NULL + END + WHERE id = $1 AND status = $2 + RETURNING cancel_attempts + "#, + ) + .bind(agreement_id) + .bind(IndexingAgreementStatus::Cancelling) + .bind(i64::from(failed_attempts)) + .bind(ended) + .fetch_optional(&self.pool) + .await?; + let (attempts,) = record.ok_or(Error::NoRecordsUpdated)?; + Ok(u32::try_from(attempts).unwrap_or_default()) + } + + /// Move an agreement to `new_status` if it is in one of `allowed_from`. + async fn set_status_from( + &self, + agreement_id: &IndexingAgreementId, + new_status: IndexingAgreementStatus, + allowed_from: &[IndexingAgreementStatus], + ) -> Result<(), Error> { + let mut tx = self.pool.begin().await?; + let updated = update_status_from(&mut tx, agreement_id, new_status, allowed_from).await?; + tx.commit().await?; + if updated { + Ok(()) + } else { + Err(Error::NoRecordsUpdated) + } + } + /// Atomically apply a reconciliation-driven state transition (accept /// and/or cancel) in a single database transaction so the /// chain_listener's Accept-then-Cancel-in-one-snapshot path does not @@ -986,15 +1215,11 @@ impl PgRegistry { let (new_status, allowed_from): (_, &[IndexingAgreementStatus]) = match kind { CancelKind::ByRequester => ( IndexingAgreementStatus::CanceledByRequester, - &[ - IndexingAgreementStatus::Created, - IndexingAgreementStatus::AcceptedOnChain, - IndexingAgreementStatus::Rejected, - ], + CANCEL_BY_REQUESTER_FROM, ), CancelKind::ByIndexer => ( IndexingAgreementStatus::CanceledByIndexer, - &[IndexingAgreementStatus::AcceptedOnChain], + CANCEL_BY_INDEXER_FROM, ), }; did_cancel = @@ -1084,15 +1309,11 @@ impl PgRegistry { let (new_status, allowed_from): (_, &[IndexingAgreementStatus]) = match cancel_kind { CancelKind::ByRequester => ( IndexingAgreementStatus::CanceledByRequester, - &[ - IndexingAgreementStatus::Created, - IndexingAgreementStatus::AcceptedOnChain, - IndexingAgreementStatus::Rejected, - ], + CANCEL_BY_REQUESTER_FROM, ), CancelKind::ByIndexer => ( IndexingAgreementStatus::CanceledByIndexer, - &[IndexingAgreementStatus::AcceptedOnChain], + CANCEL_BY_INDEXER_FROM, ), }; let did_cancel = @@ -1127,11 +1348,7 @@ impl PgRegistry { &mut tx, &cancel_by_requester, IndexingAgreementStatus::CanceledByRequester, - &[ - IndexingAgreementStatus::Created, - IndexingAgreementStatus::AcceptedOnChain, - IndexingAgreementStatus::Rejected, - ], + CANCEL_BY_REQUESTER_FROM, ) .await? { @@ -1142,7 +1359,7 @@ impl PgRegistry { &mut tx, &cancel_by_indexer, IndexingAgreementStatus::CanceledByIndexer, - &[IndexingAgreementStatus::AcceptedOnChain], + CANCEL_BY_INDEXER_FROM, ) .await? { @@ -1173,7 +1390,8 @@ impl PgRegistry { /// Fetch a batch of agreements awaiting a `terminated` event: in a /// terminal-cancel state, genuinely accepted on-chain (`accepted_at IS NOT /// NULL`, so a never-accepted local cancel is excluded), and not yet - /// emitted. Oldest-marked first so the backlog drains in order. + /// emitted, once the transaction that ended it is known or + /// [`TERMINATED_TX_WAIT_MINUTES`] have passed. Oldest-marked first. pub async fn get_agreements_pending_terminated_emission( &self, limit: i64, @@ -1185,6 +1403,8 @@ impl PgRegistry { WHERE status IN ($1, $2, $3) AND accepted_at IS NOT NULL AND terminated_event_emitted_at IS NULL + AND (canceled_tx IS NOT NULL + OR updated_at < timezone('UTC', now()) - make_interval(mins => $5)) ORDER BY updated_at ASC LIMIT $4 "#, @@ -1193,6 +1413,7 @@ impl PgRegistry { .bind(IndexingAgreementStatus::CanceledByIndexer) .bind(IndexingAgreementStatus::AbandonedByIndexer) .bind(limit) + .bind(TERMINATED_TX_WAIT_MINUTES) .fetch_all(&self.pool) .await?; Ok(rows) @@ -1341,6 +1562,8 @@ impl PgRegistry { /// emission sweep can populate the `terminated` event's tx/by/at fields. /// `COALESCE` keeps any value already observed on-chain. Best-effort /// enrichment: the event still emits (with fallbacks) if never recorded. + /// An end recorded before the agreement's accept, such as its offer's withdrawal before the + /// offer landed after all, can't be its end, so a later one replaces it and is announced. #[expect( clippy::cast_possible_wrap, reason = "predates this lint; fix when next touched" @@ -1355,13 +1578,67 @@ impl PgRegistry { sqlx::query( r#" UPDATE dipper_reg_indexing_agreements - SET canceled_at = COALESCE(canceled_at, $2), - canceled_by = COALESCE(canceled_by, $3), - canceled_tx = COALESCE(canceled_tx, $4) + SET canceled_at = CASE WHEN canceled_at < accepted_at AND $2 >= accepted_at + THEN $2 ELSE COALESCE(canceled_at, $2) END, + canceled_by = CASE WHEN canceled_at < accepted_at AND $2 >= accepted_at + THEN $3 ELSE COALESCE(canceled_by, $3) END, + canceled_tx = CASE WHEN canceled_at < accepted_at AND $2 >= accepted_at + THEN $4 ELSE COALESCE(canceled_tx, $4) END, + terminated_event_emitted_at = CASE + WHEN canceled_at < accepted_at AND $2 >= accepted_at THEN NULL + ELSE terminated_event_emitted_at + END + WHERE id = $1 + "#, + ) + .bind(agreement_id) + .bind(canceled_at as i64) + .bind(canceled_by) + .bind(canceled_tx) + .execute(&self.pool) + .await?; + Ok(()) + } + + /// Record an agreement's accept and its end together, as the chain shows them, in 1 write, + /// with the rules of [`Self::record_accepted_audit`] and [`Self::record_cancel_audit`]. An + /// end recorded before the accept is judged against the accept being recorded with it. + #[expect( + clippy::cast_possible_wrap, + reason = "chain timestamps are far below i64::MAX" + )] + pub async fn record_accept_and_cancel_audit( + &self, + agreement_id: &IndexingAgreementId, + accepted_at: u64, + accepted_tx: &str, + canceled_at: u64, + canceled_by: &str, + canceled_tx: Option<&str>, + ) -> Result<(), Error> { + sqlx::query( + r#" + UPDATE dipper_reg_indexing_agreements + SET accepted_at = COALESCE(accepted_at, $2), + accepted_tx = COALESCE(accepted_tx, $3), + canceled_at = CASE + WHEN canceled_at < COALESCE(accepted_at, $2) AND $4 >= COALESCE(accepted_at, $2) + THEN $4 ELSE COALESCE(canceled_at, $4) END, + canceled_by = CASE + WHEN canceled_at < COALESCE(accepted_at, $2) AND $4 >= COALESCE(accepted_at, $2) + THEN $5 ELSE COALESCE(canceled_by, $5) END, + canceled_tx = CASE + WHEN canceled_at < COALESCE(accepted_at, $2) AND $4 >= COALESCE(accepted_at, $2) + THEN $6 ELSE COALESCE(canceled_tx, $6) END, + terminated_event_emitted_at = CASE + WHEN canceled_at < COALESCE(accepted_at, $2) AND $4 >= COALESCE(accepted_at, $2) + THEN NULL ELSE terminated_event_emitted_at END WHERE id = $1 "#, ) .bind(agreement_id) + .bind(accepted_at as i64) + .bind(accepted_tx) .bind(canceled_at as i64) .bind(canceled_by) .bind(canceled_tx) @@ -1638,6 +1915,56 @@ impl PgRegistry { .map_err(Into::into) } + /// Agreements whose indexer stopped serving them that have ended, longest ended first, + /// whose replacement is yet to be queued. + pub async fn get_ended_agreements_awaiting_replacement( + &self, + batch_size: i64, + ) -> Result, Error> { + sqlx::query_as( + r#" + SELECT + id, + nonce_uuid, + created_at, + updated_at, + status, + indexing_request_id, + deployment_id, + indexer_id, + indexer_url, + terms, + last_block_height, + last_progress_at, + rejection_reason, + terms_version_hash + FROM dipper_reg_indexing_agreements + WHERE replacement_pending AND status <> $1 + ORDER BY updated_at ASC + LIMIT $2 + "#, + ) + .bind(IndexingAgreementStatus::Cancelling) + .bind(batch_size) + .fetch_all(&self.pool) + .await + .map_err(Into::into) + } + + /// Note that an agreement's replacement has been queued. + pub async fn mark_replacement_queued( + &self, + agreement_id: &IndexingAgreementId, + ) -> Result<(), Error> { + sqlx::query( + "UPDATE dipper_reg_indexing_agreements SET replacement_pending = false WHERE id = $1", + ) + .bind(agreement_id) + .execute(&self.pool) + .await?; + Ok(()) + } + /// Update the sync progress for an agreement. /// /// Called when the liveness checker observes the block height has changed @@ -1762,50 +2089,6 @@ impl PgRegistry { Ok(exists) } - /// Mark an agreement as `AbandonedByIndexer`. - /// - /// Transitions `AcceptedOnChain → AbandonedByIndexer`. Returns the full - /// agreement for use in the subsequent reassessment call. - /// - /// Returns [`NoRecordsUpdated`](Error::NoRecordsUpdated) if the agreement - /// doesn't exist or isn't in `AcceptedOnChain` status. - pub async fn mark_indexing_agreement_as_abandoned( - &self, - agreement_id: &IndexingAgreementId, - ) -> Result { - let record: Option = sqlx::query_as( - r#" - UPDATE dipper_reg_indexing_agreements - SET - status = $1, - updated_at = timezone('UTC', now()) - WHERE id = $2 AND status = $3 - RETURNING - id, - nonce_uuid, - created_at, - updated_at, - status, - indexing_request_id, - deployment_id, - indexer_id, - indexer_url, - terms, - last_block_height, - last_progress_at, - rejection_reason, - terms_version_hash - "#, - ) - .bind(IndexingAgreementStatus::AbandonedByIndexer) - .bind(agreement_id) - .bind(IndexingAgreementStatus::AcceptedOnChain) - .fetch_optional(&self.pool) - .await?; - - record.ok_or(Error::NoRecordsUpdated) - } - // ========================================================================= // Indexer denylist operations // ========================================================================= @@ -1834,8 +2117,8 @@ impl PgRegistry { /// Returns (agreement_id, indexer_id, deployment_id, base_rate_wei, /// entity_rate_wei) per active agreement for optimistic fee estimation. /// - /// Queries all `Created` or `AcceptedOnChain` agreements and extracts - /// both rate fields from the terms metadata. + /// Queries all `Created`, `AcceptedOnChain` or `Cancelling` agreements, the last + /// still paid until their cancel lands, and extracts both rate fields from the terms. pub async fn get_agreement_fee_rates( &self, ) -> Result, Error> { @@ -1847,11 +2130,12 @@ impl PgRegistry { r#" SELECT id, indexer_id, terms FROM dipper_reg_indexing_agreements - WHERE status IN ($1, $2) + WHERE status IN ($1, $2, $3) "#, ) .bind(IndexingAgreementStatus::Created) .bind(IndexingAgreementStatus::AcceptedOnChain) + .bind(IndexingAgreementStatus::Cancelling) .fetch_all(&self.pool) .await?; @@ -2123,7 +2407,8 @@ impl PgRegistry { /// Batched form of `update_status_from`: transitions all rows whose `id` /// is in `agreement_ids` and whose current status is in `allowed_from` to -/// `new_status`, in one statement. Returns the ids of the rows that +/// `new_status`, in one statement; a row noted abandoned that `new_status` would make +/// `CanceledByRequester` becomes `AbandonedByIndexer` instead. Returns the ids of the rows that /// actually flipped (matched the CAS guard) so callers can build per-id /// outcome maps. Empty input is a fast-path no-op. async fn batch_update_status_from( @@ -2136,20 +2421,25 @@ async fn batch_update_status_from( return Ok(Vec::new()); } let placeholders = (0..allowed_from.len()) - .map(|i| format!("${}", i + 3)) + .map(|i| format!("${}", i + 5)) .collect::>() .join(", "); + // An agreement dipper ended because its indexer stopped serving it ends as abandoned. let sql = format!( r#" UPDATE dipper_reg_indexing_agreements - SET status = $1, updated_at = timezone('UTC', now()) + SET + status = CASE WHEN abandoned AND $1 = $3 THEN $4 ELSE $1 END, + updated_at = timezone('UTC', now()) WHERE id = ANY($2) AND status IN ({placeholders}) RETURNING id "# ); let mut query = sqlx::query_as::<_, (IndexingAgreementId,)>(&sql) .bind(new_status) - .bind(agreement_ids); + .bind(agreement_ids) + .bind(IndexingAgreementStatus::CanceledByRequester) + .bind(IndexingAgreementStatus::AbandonedByIndexer); for status in allowed_from { query = query.bind(*status); } diff --git a/dipper-pgregistry/tests/it_registry_postgres.rs b/dipper-pgregistry/tests/it_registry_postgres.rs index 37cc9b4b..d5fd3cef 100644 --- a/dipper-pgregistry/tests/it_registry_postgres.rs +++ b/dipper-pgregistry/tests/it_registry_postgres.rs @@ -677,6 +677,52 @@ async fn get_declined_indexers_by_deployment_returns_rejected() { assert!(declined_5e.contains(&indexer_c)); } +#[tokio::test] +async fn get_declined_indexers_by_deployment_includes_an_indexer_that_abandoned_it() { + // Otherwise the reassessment that replaces it could pick the same indexer straight back. + let (db, _temp_db) = temp_registry_db().await; + run_fixture(&db, include_str!("fixtures/0002_indexing_agreements.sql")) + .await + .expect("Failed to run fixture"); + let registry = PgRegistry::new(db); + // Fixture 0002's accepted agreement. + let agreement_id = + IndexingAgreementId::from_bytes([0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 3]); + let deployment: DeploymentId = "QmUzRg2HHMpbgf6Q4VHKNDbtBEJnyp5JWCh2gUX9AV6jXv" + .parse() + .unwrap(); + let indexer = indexer_id!("d609e9fdd6ce53e5a26278c50486dd6791d4d705"); + registry + .mark_indexing_agreement_as_abandoning(&agreement_id) + .await + .expect("abandon"); + registry + .mark_indexing_agreement_as_canceled_by_requester(&agreement_id) + .await + .expect("ended abandoned"); + + let within = registry + .get_declined_indexers_by_deployment(30, 1, 5, 1) + .await + .expect("Failed to get declined indexers"); + let past = registry + .get_declined_indexers_by_deployment(0, 1, 5, 1) + .await + .expect("Failed to get declined indexers"); + + assert!( + within + .get(&deployment) + .is_some_and(|ids| ids.contains(&indexer)) + ); + assert!( + !past + .get(&deployment) + .is_some_and(|ids| ids.contains(&indexer)), + "only for the standard lookback" + ); +} + #[tokio::test] async fn get_declined_indexers_by_deployment_empty_when_no_declines() { //* Given @@ -1768,43 +1814,76 @@ async fn test_count_active_agreements_by_deployment() { ); } -#[tokio::test] -async fn test_mark_as_abandoned_transitions_status() { - //* Given +/// Start abandoning fixture 0002's accepted agreement, then end it the way `end` does. +#[expect( + clippy::expect_used, + reason = "a test helper fails its test on any error" +)] +async fn abandon_then_end(end: F) -> (Result<(), Error>, IndexingAgreementStatus) +where + F: AsyncFnOnce(&PgRegistry, &IndexingAgreementId), +{ let (db, _temp_db) = temp_registry_db().await; run_fixture(&db, include_str!("fixtures/0002_indexing_agreements.sql")) .await .expect("Failed to run fixture"); let registry = PgRegistry::new(db); - - // AcceptedOnChain agreement from fixture 0002 let agreement_id = IndexingAgreementId::from_bytes([0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 3]); - - //* When - let abandoned = registry - .mark_indexing_agreement_as_abandoned(&agreement_id) + registry + .mark_indexing_agreement_as_abandoning(&agreement_id) + .await + .expect("an accepted agreement can be abandoned"); + let listed = registry + .get_cancelling_agreements(100, 10, 0) .await - .expect("Failed to mark agreement as abandoned"); + .expect("cancelling query"); + assert!(listed[0].abandoned); - //* Then - assert_eq!( - abandoned.status, - IndexingAgreementStatus::AbandonedByIndexer, - "Status should be AbandonedByIndexer" - ); + end(®istry, &agreement_id).await; - // Second call must fail — agreement is no longer AcceptedOnChain - let err = registry - .mark_indexing_agreement_as_abandoned(&agreement_id) + let again = registry + .mark_indexing_agreement_as_abandoning(&agreement_id) + .await; + let ended = registry + .get_indexing_agreement_by_id(&agreement_id) .await - .expect_err("Expected error on second mark_as_abandoned call"); + .expect("agreement query") + .expect("agreement"); + (again, ended.status) +} + +#[tokio::test] +async fn an_abandoned_agreement_dipper_cancels_ends_abandoned() { + let (again, status) = abandon_then_end(async |registry, id| { + registry + .mark_indexing_agreement_as_canceled_by_requester(id) + .await + .expect("dipper's cancel confirmed"); + }) + .await; + + assert_eq!(status, IndexingAgreementStatus::AbandonedByIndexer); assert!( - matches!(err, Error::NoRecordsUpdated), - "Expected NoRecordsUpdated, got: {err:?}" + matches!(again, Err(Error::NoRecordsUpdated)), + "got {again:?}" ); } +#[tokio::test] +async fn an_abandoned_agreement_the_listener_sees_cancelled_ends_abandoned() { + let (_, status) = abandon_then_end(async |registry, id| { + let outcome = registry + .apply_reconciliation(id, false, Some(CancelKind::ByRequester)) + .await + .expect("reconciliation"); + assert!(outcome.did_cancel); + }) + .await; + + assert_eq!(status, IndexingAgreementStatus::AbandonedByIndexer); +} + // ============================================================================= // Rejection reason storage tests // ============================================================================= @@ -2931,6 +3010,86 @@ async fn apply_reconciliation_batch_handles_all_four_item_shapes() { ); } +#[tokio::test] +async fn recording_a_missed_accept_announces_the_agreement_once() { + // The chain listener records the accept and cancel of an agreement dipper had + // already cancelled locally, on every read of its cancelled snapshot. The first + // values must stick and, once announced, a later read must not announce again. + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let registry = PgRegistry::new(db); + let id = IndexingAgreementId::from_bytes([0xbb, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1]); + registry + .mark_indexing_agreement_as_canceled_by_requester(&id) + .await + .expect("dipper cancels it locally"); + + for (at, by) in [(1_700_000_002, "0xchain"), (1_800_000_000, "0xlater")] { + registry + .record_cancel_audit(&id, at, by, Some("0xcxltx")) + .await + .expect("cancel record"); + registry + .record_accepted_audit(&id, at - 1, "0xacc") + .await + .expect("accept record"); + } + + let accepted = registry + .get_agreements_pending_accepted_emission(100) + .await + .expect("accepted query"); + let accepted: Vec<_> = accepted.iter().filter(|p| p.agreement_id == id).collect(); + assert_eq!(accepted.len(), 1); + assert_eq!(accepted[0].accepted_at, 1_700_000_001); + let terminated = registry + .get_agreements_pending_terminated_emission(100) + .await + .expect("terminated query"); + let terminated: Vec<_> = terminated.iter().filter(|p| p.agreement_id == id).collect(); + assert_eq!(terminated.len(), 1); + assert_eq!(terminated[0].canceled_at, Some(1_700_000_002)); + assert_eq!(terminated[0].canceled_by.as_deref(), Some("0xchain")); + + registry + .mark_accepted_event_emitted(&id) + .await + .expect("mark accepted"); + registry + .mark_terminated_event_emitted(&id) + .await + .expect("mark terminated"); + registry + .record_cancel_audit(&id, 1_900_000_000, "0xagain", Some("0xcxltx")) + .await + .expect("cancel record"); + registry + .record_accepted_audit(&id, 1_899_999_999, "0xacc") + .await + .expect("accept record"); + assert!( + !registry + .get_agreements_pending_accepted_emission(100) + .await + .expect("accepted query") + .iter() + .any(|p| p.agreement_id == id) + ); + assert!( + !registry + .get_agreements_pending_terminated_emission(100) + .await + .expect("terminated query") + .iter() + .any(|p| p.agreement_id == id) + ); +} + #[tokio::test] async fn pending_emission_queries_cover_the_paired_accept_then_cancel() { // Regression guard: an agreement accepted and then cancelled in a single @@ -3264,3 +3423,633 @@ async fn count_created_agreements_by_indexer_counts_only_created() { ); assert_eq!(global, 3, "global counts only the 3 Created rows"); } + +fn fixture_agreement(prefix: u8) -> IndexingAgreementId { + let mut bytes = [0u8; 16]; + bytes[0] = prefix; + bytes[15] = 1; + IndexingAgreementId::from_bytes(bytes) +} + +#[tokio::test] +async fn cancelling_agreements_are_listed_until_their_cancel_fails_too_often() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let created = fixture_agreement(0xaa); + let registry = PgRegistry::new(db.clone()); + let accepted = + IndexingAgreementId::from_bytes([0xaa, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2]); + let ended = + IndexingAgreementId::from_bytes([0xaa, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 3]); + let expired = fixture_agreement(0xcc); + + for id in [created, accepted, expired] { + registry + .mark_indexing_agreement_as_cancelling(&id) + .await + .expect("an agreement that may be live can be marked cancelling"); + } + let result = registry.mark_indexing_agreement_as_cancelling(&ended).await; + assert!( + matches!(result, Err(Error::NoRecordsUpdated)), + "got {result:?}" + ); + registry + .mark_indexing_agreement_as_canceled_by_requester(&expired) + .await + .expect("leave 2 cancelling"); + registry + .record_accepted_audit(&accepted, 1_700_000_000, "0xacc") + .await + .expect("accept record"); + + let just_marked = registry + .get_cancelling_agreements(100, 2, 5) + .await + .expect("cancelling query"); + assert!( + just_marked.is_empty(), + "one just marked waits for the cancel sent with the mark to be mined" + ); + + let listed = registry + .get_cancelling_agreements(100, 2, 0) + .await + .expect("cancelling query"); + let mut seen: Vec<_> = listed + .iter() + .map(|row| { + ( + row.agreement.id, + row.accepted_on_chain, + row.agreement.status, + ) + }) + .collect(); + seen.sort_by_key(|(id, ..)| *id); + assert_eq!( + seen, + vec![ + (created, false, IndexingAgreementStatus::Cancelling), + (accepted, true, IndexingAgreementStatus::Cancelling), + ] + ); + + for attempts in [1, 2] { + let counted = registry.record_cancel_check(&created, 1, None).await; + assert_eq!(counted.unwrap(), attempts); + } + let listed = registry + .get_cancelling_agreements(100, 2, 0) + .await + .expect("cancelling query"); + let ids: Vec<_> = listed.iter().map(|row| row.agreement.id).collect(); + assert_eq!( + ids, + vec![accepted], + "one that failed too often waits an hour between checks" + ); + + sqlx::query( + "UPDATE dipper_reg_indexing_agreements \ + SET cancel_checked_at = cancel_checked_at - INTERVAL '2 hours' WHERE id = $1", + ) + .bind(created) + .execute(&db) + .await + .expect("Failed to age the check"); + let listed = registry + .get_cancelling_agreements(100, 2, 0) + .await + .expect("cancelling query"); + assert!( + listed.iter().any(|row| row.agreement.id == created), + "and is tried again after it" + ); + + let not_cancelling = registry + .record_cancel_check(&fixture_agreement(0xbb), 1, None) + .await; + assert!(matches!(not_cancelling, Err(Error::NoRecordsUpdated))); +} + +#[tokio::test] +async fn cancelling_agreements_that_may_be_paying_go_first() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let offer = fixture_agreement(0xaa); + // An offer still open to acceptance, so only the accepted agreement can be paying. + sqlx::query( + "UPDATE dipper_reg_indexing_agreements \ + SET terms = jsonb_set(terms::jsonb, '{deadline}', to_jsonb(4102444800::bigint)) \ + WHERE id = $1", + ) + .bind(offer) + .execute(&db) + .await + .expect("Failed to update deadline"); + let registry = PgRegistry::new(db.clone()); + let accepted = + IndexingAgreementId::from_bytes([0xaa, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2]); + registry + .record_accepted_audit(&accepted, 1_700_000_000, "0xacc") + .await + .expect("accept record"); + for id in [accepted, offer] { + registry + .mark_indexing_agreement_as_cancelling(&id) + .await + .expect("mark cancelling"); + } + let order = async || -> Vec { + let listed = registry + .get_cancelling_agreements(100, 2, 0) + .await + .expect("cancelling query"); + listed.iter().map(|row| row.agreement.id).collect() + }; + + registry + .record_cancel_check(&accepted, 0, None) + .await + .unwrap(); + assert_eq!( + order().await, + vec![accepted, offer], + "one accepted on-chain may be paying its indexer, so it goes ahead of an offer newly \ + marked, which counts as checked when it was marked" + ); + + registry.record_cancel_check(&offer, 0, None).await.unwrap(); + assert_eq!(order().await, vec![accepted, offer]); + + sqlx::query( + "UPDATE dipper_reg_indexing_agreements \ + SET cancel_checked_at = cancel_checked_at - INTERVAL '2 hours' WHERE id = $1", + ) + .bind(offer) + .execute(&db) + .await + .expect("Failed to age the check"); + assert_eq!( + order().await, + vec![offer, accepted], + "an offer unchecked for over an hour isn't held back for ever" + ); +} + +#[tokio::test] +async fn a_cancelling_agreement_keeps_when_it_was_first_found_ended() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let registry = PgRegistry::new(db); + let id = fixture_agreement(0xaa); + registry + .mark_indexing_agreement_as_cancelling(&id) + .await + .expect("mark cancelling"); + let ended_seen_at = async || { + registry + .get_cancelling_agreements(100, 10, 0) + .await + .expect("cancelling query")[0] + .ended_seen_at + }; + + registry + .record_cancel_check(&id, 0, Some(true)) + .await + .unwrap(); + let first = ended_seen_at().await.expect("found ended"); + registry + .record_cancel_check(&id, 0, Some(true)) + .await + .unwrap(); + registry.record_cancel_check(&id, 0, None).await.unwrap(); + assert_eq!( + ended_seen_at().await, + Some(first), + "kept from the first time" + ); + + registry + .record_cancel_check(&id, 0, Some(false)) + .await + .unwrap(); + assert_eq!(ended_seen_at().await, None, "found live again"); +} + +#[tokio::test] +async fn an_ended_agreement_found_live_on_chain_goes_back_to_cancelling() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let registry = PgRegistry::new(db.clone()); + let ended = fixture_agreement(0xaa); + let accepted = + IndexingAgreementId::from_bytes([0xaa, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2]); + registry + .mark_indexing_agreement_as_cancelling(&ended) + .await + .expect("mark cancelling"); + // The withdrawal of its offer, recorded as its end before the offer landed after all. + registry + .record_cancel_audit(&ended, 1_700_000_000, "0xpayer", Some("0xwithdrawal")) + .await + .expect("cancel record"); + assert_eq!( + registry.record_cancel_check(&ended, 2, None).await.unwrap(), + 2 + ); + registry + .mark_indexing_agreement_as_canceled_by_requester(&ended) + .await + .expect("mark ended"); + + registry + .reopen_indexing_agreement_cancel(&ended, true) + .await + .expect("an ended agreement can be reopened"); + let (canceled_tx,): (Option,) = + sqlx::query_as("SELECT canceled_tx FROM dipper_reg_indexing_agreements WHERE id = $1") + .bind(ended) + .fetch_one(&db) + .await + .expect("cancel record query"); + assert_eq!( + canceled_tx, None, + "the end on record no longer stands once the chain shows it live" + ); + + // Reopening sends no cancel, so there is none to wait on being mined. + let listed = registry + .get_cancelling_agreements(100, 1, 5) + .await + .expect("cancelling query"); + let ids: Vec<_> = listed.iter().map(|row| row.agreement.id).collect(); + assert_eq!(ids, vec![ended], "its cancel attempts start afresh"); + let still_wanted = registry + .reopen_indexing_agreement_cancel(&accepted, true) + .await; + assert!( + matches!(still_wanted, Err(Error::NoRecordsUpdated)), + "got {still_wanted:?}" + ); +} + +#[tokio::test] +async fn an_end_recorded_before_the_accept_gives_way_to_the_real_one() { + // An offer withdrawn, then accepted when it landed after all: the withdrawal can't be the + // end of an agreement accepted later, however the agreement came back to be cancelled. + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let registry = PgRegistry::new(db.clone()); + let id = fixture_agreement(0xaa); + let canceled_tx = async || -> Option { + let (tx,): (Option,) = + sqlx::query_as("SELECT canceled_tx FROM dipper_reg_indexing_agreements WHERE id = $1") + .bind(id) + .fetch_one(&db) + .await + .expect("cancel record query"); + tx + }; + registry + .record_cancel_audit(&id, 1_000, "0xpayer", Some("0xwithdrawal")) + .await + .expect("cancel record"); + registry + .record_accepted_audit(&id, 2_000, "0xaccept") + .await + .expect("accept record"); + + registry + .record_cancel_audit(&id, 3_000, "0xpayer", Some("0xcancel")) + .await + .expect("cancel record"); + assert_eq!(canceled_tx().await.as_deref(), Some("0xcancel")); + + registry + .record_cancel_audit(&id, 4_000, "0xpayer", Some("0xlater")) + .await + .expect("cancel record"); + assert_eq!( + canceled_tx().await.as_deref(), + Some("0xcancel"), + "a real end, after the accept, is kept" + ); +} + +#[tokio::test] +async fn a_cancelling_agreement_stays_live_and_unannounced_until_it_ends() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let registry = PgRegistry::new(db); + let cancelling = + IndexingAgreementId::from_bytes([0xaa, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2]); + let by_indexer = fixture_agreement(0xbb); + registry + .mark_indexing_agreement_as_canceled_by_requester(&fixture_agreement(0xaa)) + .await + .expect("cancel the other live agreement"); + for id in [cancelling, by_indexer] { + registry + .mark_indexing_agreement_as_cancelling(&id) + .await + .expect("mark cancelling"); + } + registry + .record_accepted_audit(&cancelling, 1_700_000_000, "0xacc") + .await + .expect("accept record"); + registry + .record_cancel_audit(&cancelling, 1_700_000_001, "0xmgr", Some("0xcxl")) + .await + .expect("cancel record"); + + assert!( + !registry.exists_active_agreements().await.unwrap(), + "agreements being cancelled don't keep the listener polling fast; their retry \ + runs on its own timer" + ); + let fee_rates = registry + .get_agreement_fee_rates() + .await + .expect("fee rates query"); + assert!( + fee_rates.iter().any(|(id, ..)| *id == cancelling), + "its fees still count until the cancel lands" + ); + let terminated = registry + .get_agreements_pending_terminated_emission(100) + .await + .expect("terminated query"); + assert!( + terminated.iter().all(|p| p.agreement_id != cancelling), + "not announced as ended while still cancelling" + ); + + let outcome = registry + .apply_reconciliation(&by_indexer, false, Some(CancelKind::ByIndexer)) + .await + .expect("indexer's cancel read from the chain"); + assert!(outcome.did_cancel); + registry + .mark_indexing_agreement_as_canceled_by_requester(&cancelling) + .await + .expect("dipper's cancel confirmed"); + + let terminated = registry + .get_agreements_pending_terminated_emission(100) + .await + .expect("terminated query"); + assert!(terminated.iter().any(|p| p.agreement_id == cancelling)); +} + +/// Reassess can mark an agreement `Cancelling` while its offer is mining, and the offer can +/// still land, so its hash is worth keeping. Once the agreement has ended it isn't. +#[tokio::test] +async fn an_offer_mined_after_the_cancel_began_keeps_its_hash() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let registry = PgRegistry::new(db.clone()); + let id = fixture_agreement(0xaa); + registry + .mark_indexing_agreement_as_cancelling(&id) + .await + .expect("mark cancelling"); + let stored = async || { + sqlx::query_as::<_, (Option>, time::OffsetDateTime)>( + "SELECT offer_tx_hash, updated_at FROM dipper_reg_indexing_agreements WHERE id = $1", + ) + .bind(id) + .fetch_one(&db) + .await + .expect("read the agreement") + }; + let (_, marked_at) = stored().await; + + registry + .update_offer_tx_hash(&id, &[0x11; 32]) + .await + .expect("a cancelling agreement takes the hash"); + assert_eq!( + stored().await, + (Some(vec![0x11; 32]), marked_at), + "hash stored, and the time it was marked kept" + ); + + sqlx::query("UPDATE dipper_reg_indexing_agreements SET status = 5 WHERE id = $1") + .bind(id) + .execute(&db) + .await + .expect("expire the agreement"); + let err = registry + .update_offer_tx_hash(&id, &[0x22; 32]) + .await + .expect_err("an ended agreement doesn't take the hash"); + assert!(matches!(err, Error::NoRecordsUpdated)); +} + +/// An agreement whose indexer stopped serving it is replaced only once it can't be paid, and +/// only once, whatever ended it. +#[tokio::test] +async fn an_abandoned_agreement_awaits_replacement_once_ended_until_queued() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let id = fixture_agreement(0xaa); + sqlx::query("UPDATE dipper_reg_indexing_agreements SET status = $1 WHERE id = $2") + .bind(IndexingAgreementStatus::AcceptedOnChain) + .bind(id) + .execute(&db) + .await + .expect("accept the agreement"); + let registry = PgRegistry::new(db); + registry + .mark_indexing_agreement_as_abandoning(&id) + .await + .expect("mark abandoning"); + let awaiting = async || { + registry + .get_ended_agreements_awaiting_replacement(10) + .await + .expect("awaiting query") + .into_iter() + .map(|agreement| agreement.id) + .collect::>() + }; + assert!( + awaiting().await.is_empty(), + "still cancelling, so maybe paid" + ); + + registry + .mark_indexing_agreement_as_canceled_by_requester(&id) + .await + .expect("end it"); + assert_eq!(awaiting().await, vec![id]); + + registry + .mark_replacement_queued(&id) + .await + .expect("note it queued"); + assert!(awaiting().await.is_empty()); +} + +/// The accept and end of an agreement replayed from the chain go in together: an end already +/// known stays, unless it came before the accept, and then the replayed end is announced. +#[tokio::test] +async fn an_accept_and_end_from_the_chain_are_recorded_in_1_write() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let id = fixture_agreement(0xaa); + let registry = PgRegistry::new(db.clone()); + let record = async |accepted_at: u64, canceled_at: u64, tx: &str| { + registry + .record_accept_and_cancel_audit( + &id, + accepted_at, + "0xacc", + canceled_at, + "payer", + Some(tx), + ) + .await + .expect("record"); + }; + let recorded = async || { + sqlx::query_as::<_, (Option, Option, Option, bool)>( + "SELECT accepted_at, canceled_at, canceled_tx, terminated_event_emitted_at IS NULL \ + FROM dipper_reg_indexing_agreements WHERE id = $1", + ) + .bind(id) + .fetch_one(&db) + .await + .expect("read") + }; + + // Its offer was withdrawn at 50, already announced, before an accept at 100 came to light. + sqlx::query( + "UPDATE dipper_reg_indexing_agreements SET canceled_at = 50, canceled_tx = '0xwd', \ + terminated_event_emitted_at = now() WHERE id = $1", + ) + .bind(id) + .execute(&db) + .await + .expect("withdraw"); + record(100, 200, "0xend").await; + assert_eq!( + recorded().await, + (Some(100), Some(200), Some("0xend".to_owned()), true), + "the end before the accept gives way, to be announced again" + ); + + record(300, 400, "0xlater").await; + assert_eq!( + recorded().await, + (Some(100), Some(200), Some("0xend".to_owned()), true), + "what is already known stays" + ); +} + +/// An ended agreement's `terminated` event waits for the transaction that ended it, which the +/// chain listener can record after dipper marks it ended, but not for ever. +#[tokio::test] +async fn an_end_is_announced_once_its_transaction_is_known_or_after_an_hour() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let id = fixture_agreement(0xaa); + let registry = PgRegistry::new(db.clone()); + registry + .mark_indexing_agreement_as_canceled_by_requester(&id) + .await + .expect("dipper ends it"); + registry + .record_accepted_audit(&id, 1_700_000_000, "0xacc") + .await + .expect("its accept is known"); + let pending = async || { + registry + .get_agreements_pending_terminated_emission(100) + .await + .expect("terminated query") + .iter() + .any(|p| p.agreement_id == id) + }; + assert!(!pending().await, "waits for its transaction"); + + sqlx::query( + "UPDATE dipper_reg_indexing_agreements \ + SET updated_at = timezone('UTC', now()) - interval '61 minutes' WHERE id = $1", + ) + .bind(id) + .execute(&db) + .await + .expect("age it"); + assert!(pending().await, "goes out without one after an hour"); + + sqlx::query( + "UPDATE dipper_reg_indexing_agreements SET updated_at = timezone('UTC', now()) WHERE id = $1", + ) + .bind(id) + .execute(&db) + .await + .expect("make it recent again"); + registry + .record_cancel_audit(&id, 1_700_000_100, "0xmgr", Some("0xend")) + .await + .expect("its transaction is recorded"); + assert!( + pending().await, + "goes out as soon as its transaction is known" + ); +} diff --git a/dipper-rpc/src/admin/indexing_agreements.rs b/dipper-rpc/src/admin/indexing_agreements.rs index a08c7e9a..7be13742 100644 --- a/dipper-rpc/src/admin/indexing_agreements.rs +++ b/dipper-rpc/src/admin/indexing_agreements.rs @@ -122,6 +122,9 @@ pub enum Status { /// This is a terminal state. AbandonedByIndexer, + /// Dipper is cancelling the agreement on-chain; it may still be live there. + Cancelling, + /// A fallback for unknown status values. Unknown, } @@ -140,6 +143,7 @@ impl serde::Serialize for Status { Status::AcceptedOnChain => "ACCEPTED_ON_CHAIN", Status::Rejected => "REJECTED", Status::AbandonedByIndexer => "ABANDONED_BY_INDEXER", + Status::Cancelling => "CANCELLING", Status::Unknown => "UNKNOWN", }; serializer.serialize_str(status) @@ -161,6 +165,7 @@ impl<'de> serde::Deserialize<'de> for Status { "ACCEPTED_ON_CHAIN" => Status::AcceptedOnChain, "REJECTED" => Status::Rejected, "ABANDONED_BY_INDEXER" => Status::AbandonedByIndexer, + "CANCELLING" => Status::Cancelling, _ => Status::Unknown, }; Ok(status) diff --git a/k8s/configmap-example.yaml b/k8s/configmap-example.yaml index 2d369960..88222ea5 100644 --- a/k8s/configmap-example.yaml +++ b/k8s/configmap-example.yaml @@ -118,5 +118,10 @@ data: "enabled": true, "listen_addr": "0.0.0.0:8546", "threshold": 840 + }, + "alerts": { + "slack_webhook_url": "REPLACE_ME_OR_REMOVE_FOR_NO_ALERTS", + "events": ["agreement_cancel_stuck", "rpc_blocks_refused", "nonce_gap_fill_failed"], + "throttle": 900 } }