From dc0a0d355f2f8fed2868286c763746503729238c Mon Sep 17 00:00:00 2001 From: MoonBoi9001 Date: Fri, 2 Oct 2026 15:52:16 +0300 Subject: [PATCH 01/26] chore(review): start a branch that collects the agreement-cancel fixes From 649ae2378e4f8bdeae48916e5aa49c82fa2e97e0 Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Fri, 2 Oct 2026 15:53:24 +0300 Subject: [PATCH 02/26] fix: withdraw the on-chain offer of an agreement dipper cancels (#708) * fix(reassess): revoke the pending offer when cancelling an agreement Dipper puts an offer on-chain before the indexer accepts it, but cancelled such agreements only in its database, so the indexer could still accept and be paid with nothing tracking it. The cancel now also goes on-chain, which revokes the offer, or ends the agreement if just accepted. * fix(chain): report a pending offer as live after a cancel A cancel that mined but changed nothing was counted as done when the agreement was only offered, because the post-cancel check only knew about accepted agreements. It now also reports an offer still waiting for the indexer, so a failed revoke is retried instead of marked cancelled. * docs(reassess): explain how a failed revoke of an offer recovers A failed cancel can now leave an agreement that was only offered, not just an accepted one, and the comment on the failure log said only accepted agreements were left behind. --- bin/dipper-service/src/cancel_dispatch.rs | 5 +- bin/dipper-service/src/chain_client.rs | 4 +- bin/dipper-service/src/chain_client/client.rs | 43 ++++++++-- .../handlers/reassess_indexing_request.rs | 86 ++++++++++++++----- 4 files changed, 103 insertions(+), 35 deletions(-) diff --git a/bin/dipper-service/src/cancel_dispatch.rs b/bin/dipper-service/src/cancel_dispatch.rs index 08032e4a..7452dbd6 100644 --- a/bin/dipper-service/src/cancel_dispatch.rs +++ b/bin/dipper-service/src/cancel_dispatch.rs @@ -10,8 +10,9 @@ use crate::{ }; /// Pass both ACTIVE and PENDING; local status lags the chain, so let the -/// collector no-op the absent scope. Never SCOPE_SIGNED (=4): acceptance is -/// offer-based and dipper never retracts a pending offer, so it isn't needed. +/// collector no-op the absent scope. PENDING revokes an offer not yet accepted. +/// Never SCOPE_SIGNED (=4): acceptance is offer-based, so revoking the stored +/// offer is enough. const SCOPE_ACTIVE: u16 = 1; const SCOPE_PENDING: u16 = 2; const SCOPE_BOTH: u16 = SCOPE_ACTIVE | SCOPE_PENDING; diff --git a/bin/dipper-service/src/chain_client.rs b/bin/dipper-service/src/chain_client.rs index dd4e4da3..5c734b26 100644 --- a/bin/dipper-service/src/chain_client.rs +++ b/bin/dipper-service/src/chain_client.rs @@ -121,8 +121,8 @@ pub trait ChainClient { ) -> Result, ChainClientError>; /// Read whether the agreement is still live on-chain (terms accepted and no - /// cancellation notice given) via the RecurringCollector's - /// `getAgreementDetails(id, VERSION_CURRENT)`. + /// cancellation notice given, or an offer still waiting to be accepted) via + /// the RecurringCollector's `getAgreementDetails(id, VERSION_CURRENT)`. async fn agreement_still_active( &self, agreement_id: &[u8; 16], diff --git a/bin/dipper-service/src/chain_client/client.rs b/bin/dipper-service/src/chain_client/client.rs index 67fc7efe..e759edb4 100644 --- a/bin/dipper-service/src/chain_client/client.rs +++ b/bin/dipper-service/src/chain_client/client.rs @@ -57,12 +57,24 @@ const RECEIPT_POLL_INTERVAL: Duration = Duration::from_millis(500); /// pre-acceptance) terms. `getAgreementDetails(id, 0)` reports their state. const VERSION_CURRENT: u64 = 0; -/// `AgreementDetails.state` flags from `IAgreementCollector.sol` (ACCEPTED=2, -/// NOTICE_GIVEN=4). `getAgreementDetails` keeps ACCEPTED set on a canceled -/// agreement and ORs in NOTICE_GIVEN, so a cancel must clear it, not just lack it. +/// `AgreementDetails.state` flags from `IAgreementCollector.sol` (REGISTERED=1, +/// ACCEPTED=2, NOTICE_GIVEN=4). `getAgreementDetails` keeps ACCEPTED set on a +/// canceled agreement and ORs in NOTICE_GIVEN, so a cancel must clear it, not +/// just lack it. REGISTERED without ACCEPTED is an offer still waiting. +const STATE_REGISTERED: u16 = 1; const STATE_ACCEPTED: u16 = 2; const STATE_NOTICE_GIVEN: u16 = 4; +/// Live iff the terms are accepted and no cancellation notice exists, or an +/// offer is still stored for the indexer to accept. A cancel sets NOTICE_GIVEN +/// while ACCEPTED stays set, so the notice bit tells a live agreement from a +/// cancelled one; a revoked offer reads as an empty state. +fn still_live(state: u16) -> bool { + let accepted = state & STATE_ACCEPTED != 0; + let pending_offer = state & STATE_REGISTERED != 0 && !accepted; + pending_offer || (accepted && state & STATE_NOTICE_GIVEN == 0) +} + /// Error patterns that indicate a nonce-related issue. /// /// These errors can be resolved by refreshing the nonce and retrying. @@ -805,11 +817,7 @@ impl ChainClient for AlloyChainClient { )) })?; - // Live iff the terms are accepted and no cancellation notice exists. - // A cancel sets NOTICE_GIVEN while ACCEPTED stays set, so checking the - // notice bit is what tells a still-live agreement from a cancelled one. - let state = details.state; - Ok(state & STATE_ACCEPTED != 0 && state & STATE_NOTICE_GIVEN == 0) + Ok(still_live(details.state)) } async fn reconcile_provider( @@ -1009,6 +1017,25 @@ mod tests { use super::*; + /// A cancel must leave nothing the indexer can still be paid through: neither + /// an accepted agreement without a cancellation notice, nor an offer still + /// stored and waiting to be accepted. The collector reports a revoked offer, + /// or an id it never saw, as an empty state. + #[test] + fn still_live_covers_accepted_agreements_and_pending_offers() { + const SETTLED: u16 = 8; + const BY_PAYER: u16 = 16; + assert!(still_live(STATE_REGISTERED | STATE_ACCEPTED)); + assert!( + still_live(STATE_REGISTERED), + "a pending offer can still be accepted" + ); + assert!(!still_live( + STATE_REGISTERED | STATE_ACCEPTED | STATE_NOTICE_GIVEN | BY_PAYER | SETTLED + )); + assert!(!still_live(0), "revoked or never offered"); + } + /// Answers a send with a fixed transaction hash, echoing the request id so alloy's /// transport accepts the response. Any other call is a mistake in the test rather than /// something to answer with a hash, so say so instead of returning nonsense. diff --git a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs index fdd5fb81..b1414080 100644 --- a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs +++ b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs @@ -641,13 +641,17 @@ where let mut directly_cancelled = 0u32; let mut cancel_failures = 0u32; for old_agreement in old_iter { - // Skip agreements that haven't been accepted on-chain yet -- there is - // nothing on the contract to cancel. The local row goes straight to - // CanceledByRequester so the indexer never picks it up. - let needs_on_chain_cancel = matches!( + // A Created agreement's offer may already be on-chain, where the indexer + // can still accept it, so it is cancelled on-chain too: the cancel revokes + // a pending offer and does nothing when none was made. Rows without a + // stored terms hash can't be cancelled on-chain and are only marked locally. + let was_accepted = matches!( old_agreement.status, crate::registry::IndexingAgreementStatus::AcceptedOnChain ); + let needs_on_chain_cancel = was_accepted + || (old_agreement.status == crate::registry::IndexingAgreementStatus::Created + && old_agreement.terms_version_hash.is_some()); let mut on_chain_cancel_tx: Option = None; if needs_on_chain_cancel { @@ -711,10 +715,10 @@ where // Record the cancel audit for the accepted-on-chain agreements dipper just // cancelled, so the chain_listener's `terminated` sweep announces them - // durably. Never-accepted agreements (`!needs_on_chain_cancel`) were never - // live on-chain: they are not sweep-eligible (`accepted_at IS NULL`) and + // durably. Never-accepted agreements (`!was_accepted`) were never live + // on-chain: they are not sweep-eligible (`accepted_at IS NULL`) and // correctly emit nothing. - if needs_on_chain_cancel { + if was_accepted { let manager = ctx.agreement_conf.recurring_agreement_manager().to_string(); if let Err(err) = ctx .registry @@ -746,11 +750,15 @@ where } if cancel_failures > 0 { - // Two recovery paths cover any agreements left AcceptedOnChain here: + // Agreements whose cancel failed keep their status (AcceptedOnChain or + // Created), and two recovery paths cover them: // // - Shrink-to-zero (request now Canceled): the chain_listener's - // `sweep_orphan_canceled_agreements` retries on every sweep tick - // (default ~5 min at fast poll, ~5 h at slow poll). + // `sweep_orphan_canceled_agreements` retries AcceptedOnChain rows on + // every sweep tick (default ~5 min at fast poll, ~5 h at slow poll). + // A Created row is not retried: its offer stays open until its + // deadline, and if the indexer accepts it the listener marks it + // AcceptedOnChain, which brings it into that sweep. // - Shrink-not-zero (request still Open with too many agreements): // the periodic reassignment service re-queues reassessment at its // configured cadence (default 24 h). @@ -1043,9 +1051,12 @@ mod lifecycle_event_tests { // ---- Mock: chain client -------------------------------------------------- - /// Always reports a successful cancel that the post-cancel read confirms. - #[derive(Default)] - struct MockChainClient; + /// Always reports a successful cancel that the post-cancel read confirms, and + /// records the id of every agreement it was asked to cancel. + #[derive(Default, Clone)] + struct MockChainClient { + cancelled: Arc>>, + } #[async_trait] impl ChainClient for MockChainClient { @@ -1065,10 +1076,11 @@ mod lifecycle_event_tests { async fn cancel_via_manager( &self, _collector: Address, - _agreement_id: &[u8; 16], + agreement_id: &[u8; 16], _version_hash: B256, _options: u16, ) -> std::result::Result, ChainClientError> { + self.cancelled.lock().unwrap().push(*agreement_id); Ok(Some(B256::repeat_byte(0xcd))) } @@ -1573,7 +1585,7 @@ mod lifecycle_event_tests { selected: vec![selected(new_idx)], }, MockQueue::default(), - MockChainClient, + MockChainClient::default(), events.clone(), snapshot, ); @@ -1626,7 +1638,7 @@ mod lifecycle_event_tests { selected: vec![selected(new_idx)], }, MockQueue::default(), - MockChainClient, + MockChainClient::default(), events.clone(), snapshot, ); @@ -1673,7 +1685,7 @@ mod lifecycle_event_tests { selected: vec![selected(idx)], }, MockQueue::default(), - MockChainClient, + MockChainClient::default(), events.clone(), indexer_urls::Snapshot::new(), ); @@ -1709,7 +1721,7 @@ mod lifecycle_event_tests { selected: vec![selected(idx)], }, MockQueue::default(), - MockChainClient, + MockChainClient::default(), events.clone(), indexer_urls::Snapshot::new(), ); @@ -1734,7 +1746,7 @@ mod lifecycle_event_tests { MockRegistry::default(), // no current agreements, latch false MockIisa { selected: vec![] }, // zero available MockQueue::default(), - MockChainClient, + MockChainClient::default(), events.clone(), indexer_urls::Snapshot::new(), ); @@ -1777,7 +1789,7 @@ mod lifecycle_event_tests { registry, MockIisa { selected: vec![] }, // IISA returns nothing -> coverage drops MockQueue::default(), - MockChainClient, + MockChainClient::default(), events.clone(), indexer_urls::Snapshot::new(), ); @@ -1832,7 +1844,7 @@ mod lifecycle_event_tests { selected: vec![selected(new_idx)], }, queue.clone(), - MockChainClient, + MockChainClient::default(), events.clone(), snapshot, ); @@ -1871,7 +1883,7 @@ mod lifecycle_event_tests { // Both old agreements were never accepted on-chain (Created). One add // pairs with the first old agreement; the second, unpaired old agreement // reaches the cancel loop but, being never-accepted - // (`!needs_on_chain_cancel`), must NOT emit `terminated`. The add still + // (`!was_accepted`), must NOT emit `terminated`. The add still // emits `proposed`. Net: exactly one event, a `proposed`. let new_idx = indexer_id(0x44); let old_paired = indexer_id(0x55); @@ -1896,7 +1908,7 @@ mod lifecycle_event_tests { selected: vec![selected(new_idx)], }, MockQueue::default(), - MockChainClient, + MockChainClient::default(), events.clone(), snapshot, ); @@ -1911,6 +1923,34 @@ mod lifecycle_event_tests { ); assert!(matches!(captured[0], CapturedEvent::Proposed { .. })); } + + #[tokio::test] + async fn cancelling_an_unaccepted_agreement_revokes_its_offer_on_chain() { + // A Created agreement's offer can already be on-chain, where the indexer + // can still accept it. Cancelling the request must revoke that offer, not + // only mark the local row cancelled. + let leaving = agreement(indexer_id(0x11), IndexingAgreementStatus::Created); + let leaving_id = *leaving.id.as_bytes(); + let chain_client = MockChainClient::default(); + let registry = MockRegistry { + active_agreements: vec![leaving], + accepted_count: 0, + chain_state_lookup_fails: false, + shortfall_active: std::sync::Mutex::new(false), + }; + let ctx = build_ctx( + registry, + MockIisa { selected: vec![] }, + MockQueue::default(), + chain_client.clone(), + CapturingEventsProducer::new(), + indexer_urls::Snapshot::new(), + ); + + handle(ctx, &test_message(0)).await.expect("handler ok"); + + assert_eq!(*chain_client.cancelled.lock().unwrap(), vec![leaving_id]); + } } /// Olds reserved from cancellation: one per add-cancel pairing lost to a From a910e270b1d16ffa4857f9f2b469bab6c3b1ecad Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Fri, 2 Oct 2026 15:53:25 +0300 Subject: [PATCH 03/26] fix: stop an offer from landing after a reassessment cancels it (#710) * fix(worker): stop an offer and a reassessment from overlapping An offer job could check an agreement was still wanted, then send its offer after a reassessment had already cancelled it on-chain, leaving the offer open. Offers now take the reassessment lock shared from that check until they land, and reassessments take it exclusively. * fix(worker): let a reassessment queue behind offers already being sent A reassessment only retried every second, so a steady stream of offers could keep it out for as long as the stream lasted. It now waits up to 30 s for offers in flight to land, and no new offer starts meanwhile. Only one reassessment waits; any other still defers at once. * fix(worker): withdraw an offer whose agreement was cancelled in flight The chain listener cancels a replaced agreement without the reassessment lock, so its cancel can land just before that agreement's offer and find nothing to withdraw. The offer job now re-reads the agreement once its offer lands and withdraws the offer if dipper cancelled it meanwhile. * fix(worker): release the offer lock before withdrawing a cancelled offer Once an offer has landed any later cancel follows it, so holding the lock through a withdraw only kept a waiting reassessment out for another send and receipt wait. * fix(worker): wait for offers only after a reassessment knows it will run A reassessment that the in-flight offer limit would defer still waited up to 30 s for offers to land first, blocking new ones for nothing. It now checks that limit before waiting, logs a wait that times out, and the lock test now checks both locks are released after giving up. * fix(worker): hold offers back only while cancelling unaccepted ones A reassessment held every offer back from before its IISA call to its last cancel, so busy reassessments could keep offers waiting past their deadline. Offers now wait only while it cancels an agreement whose offer may be in flight; if that wait times out, the cancel is retried later. * fix(worker): retry withdrawing an offer of a cancelled agreement A failed withdraw only logged a warning, and a retried offer job skipped a cancelled agreement without checking for an offer an earlier attempt had left on-chain. Both now withdraw any offer still stored, and a failed withdraw retries the job. --- bin/dipper-service/src/main.rs | 2 +- bin/dipper-service/src/worker.rs | 2 + bin/dipper-service/src/worker/context.rs | 10 +- .../handlers/reassess_indexing_request.rs | 157 ++++++- .../src/worker/handlers/submit_offer.rs | 425 ++++++++++++++++-- .../src/worker/reassess_lock.rs | 156 +++++++ bin/dipper-service/src/worker/service.rs | 8 +- 7 files changed, 705 insertions(+), 55 deletions(-) create mode 100644 bin/dipper-service/src/worker/reassess_lock.rs diff --git a/bin/dipper-service/src/main.rs b/bin/dipper-service/src/main.rs index 8b22c649..2eda5367 100644 --- a/bin/dipper-service/src/main.rs +++ b/bin/dipper-service/src/main.rs @@ -431,7 +431,7 @@ pub async fn main() -> anyhow::Result<()> { } // A single global reassess lock, shared across all worker loops. - let reassess_lock = Arc::new(tokio::sync::Mutex::new(())); + let reassess_lock = worker::ReassessLock::default(); let unresponsive_breaker = Arc::new(worker::UnresponsiveBreaker::new()); let dips_accepting_cache = worker::DipsAcceptingCache::new(std::time::Duration::from_secs( agreement_conf.dips_accepting_cache_ttl_seconds(), diff --git a/bin/dipper-service/src/worker.rs b/bin/dipper-service/src/worker.rs index 9ef02e12..453895dc 100644 --- a/bin/dipper-service/src/worker.rs +++ b/bin/dipper-service/src/worker.rs @@ -2,10 +2,12 @@ mod context; mod handlers; mod messages; pub mod queue; +mod reassess_lock; mod result; pub mod service; mod service_queue; mod unresponsive_breaker; pub use context::Ctx; +pub use reassess_lock::ReassessLock; pub use unresponsive_breaker::{DipsAcceptingCache, UnresponsiveBreaker}; diff --git a/bin/dipper-service/src/worker/context.rs b/bin/dipper-service/src/worker/context.rs index 4b606026..d24cb65e 100644 --- a/bin/dipper-service/src/worker/context.rs +++ b/bin/dipper-service/src/worker/context.rs @@ -4,8 +4,9 @@ use dipper_core::state::FromState; use dipper_producer::events::SubgraphIndexingAgreementEventsProducer; use graph_networks_registry::NetworksRegistry; use thegraph_core::alloy::primitives::ChainId; -use tokio::sync::{Mutex, Notify}; +use tokio::sync::Notify; +pub use super::reassess_lock::ReassessLock; use super::{ handlers::{ CancelRejectedAgreementOnChainCtx, ReassessIndexingRequestCtx, @@ -19,11 +20,6 @@ use crate::{ signing::eip712::Eip712Signer, }; -/// A single process-wide async mutex. Only one reassessment runs at a time -/// across every worker loop, so two loops can't diff the same baseline and both -/// create agreements. In-process only (dipper is single-replica). -pub type ReassessLock = Arc>; - /// Generates a `FromState>` impl mapping InnerCtx fields onto a /// handler context type. Syntax: `impl_from_state!(Target { mappings })`, /// where a mapping is `field` (same name) or `target: source` (renamed). @@ -240,4 +236,6 @@ impl_from_state!(CancelRejectedAgreementOnChainCtx { impl_from_state!(SubmitOfferCtx { registry, chain_client, + agreement_conf, + reassess_lock, }); diff --git a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs index b1414080..5027872c 100644 --- a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs +++ b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs @@ -69,8 +69,9 @@ pub struct Ctx { /// is true. `None` disables the bypass path even if the flag is set /// (handler falls back to wall clock with a warning). pub chain_listener_chain_id: Option, - /// Global reassess lock; only one reassessment runs at a time across all - /// worker loops (see `crate::worker::context::ReassessLock`). + /// Global reassess lock: only one reassessment runs at a time across all + /// worker loops, and its cancels of unaccepted agreements wait for offers in + /// flight (see `crate::worker::context::ReassessLock`). pub reassess_lock: crate::worker::context::ReassessLock, /// Mass-unresponsive circuit breaker (see its module). pub unresponsive_breaker: Arc, @@ -128,9 +129,8 @@ where // Only one reassessment runs globally at a time; if another loop holds the // lock this pass would diff the same baseline, so defer ~1s rather than park // this loop. Deferral isn't a failure: no backoff, no attempt count. - let _reassess_guard = match ctx.reassess_lock.try_lock() { - Ok(guard) => guard, - Err(_) => return Err(JobError::Deferred(Duration::from_secs(1))), + let Some(reassessment) = ctx.reassess_lock.start_reassessment() else { + return Err(JobError::Deferred(Duration::from_secs(1))); }; // Get current active agreements for this indexing request. Fetched once here @@ -640,7 +640,24 @@ where // transient chain-client failure does the right thing. let mut directly_cancelled = 0u32; let mut cancel_failures = 0u32; - for old_agreement in old_iter { + // A cancel of an agreement whose offer may be in flight waits for that offer + // to land, so the cancel follows it on-chain and withdraws it. Only these + // cancels hold new offers back; the rest of the pass doesn't. + let unpaired: Vec<_> = old_iter.collect(); + let offers_held_back = if unpaired.iter().any(|a| offer_may_be_in_flight(a)) { + reassessment.hold_back_offers().await + } else { + None + }; + for old_agreement in unpaired { + if offer_may_be_in_flight(old_agreement) && offers_held_back.is_none() { + tracing::warn!( + agreement_id = %old_agreement.id, + "Offers in flight did not land; will retry this cancel on next reassessment" + ); + cancel_failures += 1; + continue; + } // A Created agreement's offer may already be on-chain, where the indexer // can still accept it, so it is cancelled on-chain too: the cancel revokes // a pending offer and does nothing when none was made. Rows without a @@ -1056,6 +1073,9 @@ mod lifecycle_event_tests { #[derive(Default, Clone)] struct MockChainClient { cancelled: Arc>>, + /// When set, each cancel records whether new offers were held back. + reassess_lock: Option, + offers_held_back_at_cancel: Arc>>, } #[async_trait] @@ -1081,6 +1101,12 @@ mod lifecycle_event_tests { _options: u16, ) -> std::result::Result, ChainClientError> { self.cancelled.lock().unwrap().push(*agreement_id); + if let Some(lock) = &self.reassess_lock { + self.offers_held_back_at_cancel + .lock() + .unwrap() + .push(lock.offer().is_none()); + } Ok(Some(B256::repeat_byte(0xcd))) } @@ -1498,7 +1524,7 @@ mod lifecycle_event_tests { chain_listener_notify: Arc::new(tokio::sync::Notify::new()), bypass_chain_clock_defenses: false, chain_listener_chain_id: None, - reassess_lock: Arc::new(tokio::sync::Mutex::new(())), + reassess_lock: crate::worker::ReassessLock::default(), unresponsive_breaker: Arc::new(crate::worker::UnresponsiveBreaker::new()), dips_accepting_cache: crate::worker::DipsAcceptingCache::new( std::time::Duration::from_secs(300), @@ -1924,6 +1950,116 @@ mod lifecycle_event_tests { assert!(matches!(captured[0], CapturedEvent::Proposed { .. })); } + #[tokio::test(start_paused = true)] + async fn a_saturated_pass_defers_without_holding_up_offers() { + // A pass that will only defer must not first wait out offers in flight, + // which would block new offers for nothing. + let mut ctx = build_ctx( + MockRegistry::default(), + MockIisa { selected: vec![] }, + MockQueue::default(), + MockChainClient::default(), + CapturingEventsProducer::new(), + indexer_urls::Snapshot::new(), + ); + ctx.agreement_conf = Arc::new(IndexingAgreementConfig { + max_in_flight_offers_total: Some(0), + ..test_agreement_conf() + }); + let _offer = ctx.reassess_lock.offer().expect("lock is free"); + + let started = tokio::time::Instant::now(); + let result = handle(ctx, &test_message(1)).await; + + assert!( + matches!(result, Err(crate::worker::result::JobError::Deferred(_))), + "got {result:?}" + ); + assert_eq!(started.elapsed(), std::time::Duration::ZERO); + } + + /// Ctx whose only active agreement, in `status`, leaves the target group, + /// with the chain mock recording whether offers were held back at each cancel. + fn ctx_cancelling_one( + status: IndexingAgreementStatus, + ) -> ( + Ctx, + IndexingAgreement, + ) { + let leaving = agreement(indexer_id(0x11), status); + let mut ctx = build_ctx( + MockRegistry { + active_agreements: vec![leaving.clone()], + ..MockRegistry::default() + }, + MockIisa { selected: vec![] }, + MockQueue::default(), + MockChainClient::default(), + CapturingEventsProducer::new(), + indexer_urls::Snapshot::new(), + ); + ctx.chain_client.reassess_lock = Some(ctx.reassess_lock.clone()); + (ctx, leaving) + } + + #[tokio::test] + async fn cancelling_an_unaccepted_agreement_holds_back_new_offers() { + // Its offer may be in flight: a cancel sent ahead of it would find nothing + // to withdraw, and the offer would land after it and stay open. + let (ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::Created); + let chain = ctx.chain_client.clone(); + + let result = handle(ctx, &test_message(0)).await; + + assert!(result.is_ok(), "got {result:?}"); + assert_eq!( + *chain.cancelled.lock().unwrap(), + vec![*leaving.id.as_bytes()] + ); + assert_eq!( + *chain.offers_held_back_at_cancel.lock().unwrap(), + vec![true] + ); + } + + #[tokio::test(start_paused = true)] + async fn cancelling_only_accepted_agreements_leaves_offers_running() { + // An accepted agreement has no offer in flight, so nothing to wait for. + let (ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::AcceptedOnChain); + let chain = ctx.chain_client.clone(); + let _offer = ctx.reassess_lock.offer().expect("lock is free"); + + let started = tokio::time::Instant::now(); + let result = handle(ctx, &test_message(0)).await; + + assert!(result.is_ok(), "got {result:?}"); + assert_eq!(started.elapsed(), std::time::Duration::ZERO); + assert_eq!( + *chain.cancelled.lock().unwrap(), + vec![*leaving.id.as_bytes()] + ); + assert_eq!( + *chain.offers_held_back_at_cancel.lock().unwrap(), + vec![false] + ); + } + + #[tokio::test(start_paused = true)] + async fn skips_cancelling_an_unaccepted_agreement_while_an_offer_will_not_land() { + // After 30 s it gives up on that cancel rather than send it ahead of the + // offer, leaving it for the next reassessment. + let (ctx, _leaving) = ctx_cancelling_one(IndexingAgreementStatus::Created); + let chain = ctx.chain_client.clone(); + let _stuck_offer = ctx.reassess_lock.offer().expect("lock is free"); + + let started = tokio::time::Instant::now(); + let result = handle(ctx, &test_message(0)).await; + + assert!(result.is_ok(), "got {result:?}"); + assert_eq!(started.elapsed(), std::time::Duration::from_secs(30)); + assert!(chain.cancelled.lock().unwrap().is_empty()); + } + #[tokio::test] async fn cancelling_an_unaccepted_agreement_revokes_its_offer_on_chain() { // A Created agreement's offer can already be on-chain, where the indexer @@ -1953,6 +2089,13 @@ mod lifecycle_event_tests { } } +/// Whether an offer for this agreement could still be on its way on-chain: it +/// isn't accepted yet and has the terms hash an on-chain cancel needs. +fn offer_may_be_in_flight(agreement: &crate::registry::IndexingAgreement) -> bool { + agreement.status == crate::registry::IndexingAgreementStatus::Created + && agreement.terms_version_hash.is_some() +} + /// Olds reserved from cancellation: one per add-cancel pairing lost to a /// withheld addition. Olds beyond the pairing count are true surplus (IISA /// dropped them outright) and cancel immediately regardless of pacing. diff --git a/bin/dipper-service/src/worker/handlers/submit_offer.rs b/bin/dipper-service/src/worker/handlers/submit_offer.rs index 7eb0b263..b3b39d46 100644 --- a/bin/dipper-service/src/worker/handlers/submit_offer.rs +++ b/bin/dipper-service/src/worker/handlers/submit_offer.rs @@ -16,28 +16,39 @@ //! `rcaOffers` mapping on `RecurringCollector` sits in an ERC-7201 namespaced storage //! struct with no public getter, so the chain cannot cheaply be asked what already landed. -use std::time::Duration; +use std::{sync::Arc, time::Duration}; use dipper_core::ids::{IndexingAgreementId, IndexingRequestId}; use thegraph_core::{DeploymentId, alloy::primitives::ChainId}; use url::Url; use crate::{ + cancel_dispatch::cancel_agreement_on_chain, chain_client::{ChainClient, ChainClientError, decode_revert_reason}, + config::IndexingAgreementConfig, indexer_rpc_client::into_sol_rca, - registry::{AgreementRegistry, IndexingAgreementStatus}, - worker::result::{JobError, JobResult}, + registry::{AgreementRegistry, IndexingAgreement, IndexingAgreementStatus}, + worker::{ + context::ReassessLock, + result::{JobError, JobResult}, + }, }; /// Backoff base for a tx the RPC accepted and then dropped from the mempool. pub const DROPPED_TX_RETRY_BASE: Duration = Duration::from_secs(5); +/// Retry shortly while a reassessment runs or waits; not counted as a failure. +const DEFER_WHILE_REASSESSING: JobError = JobError::Deferred(Duration::from_secs(1)); + /// Backoff base for a transient submission failure: RPC, gas or nonce. pub const TRANSIENT_RETRY_BASE: Duration = Duration::from_secs(30); pub struct Ctx { pub registry: R, pub chain_client: T, + pub agreement_conf: Arc, + /// Taken shared while the offer is checked and sent (see `ReassessLock`). + pub reassess_lock: ReassessLock, } /// Submit an RCA offer on-chain. @@ -68,30 +79,18 @@ where R: AgreementRegistry, T: ChainClient, { - // Fetch the agreement. Skip silently if it's already been transitioned - // out of Created (e.g. expired by the reassignment service). - let agreement = match ctx - .registry - .get_indexing_agreement_by_id(agreement_id) - .await - .map_err(|err| JobError::Fatal(err.into()))? - { - None => { - tracing::error!( - agreement_id = %agreement_id, - "Agreement not found in registry at submit_offer" - ); - return Ok(()); - } - Some(a) if a.status != IndexingAgreementStatus::Created => { - tracing::warn!( - agreement_id = %agreement_id, - status = %a.status, - "Agreement not in Created status, skipping offer submission" - ); - return Ok(()); + // Held from the status check until the offer lands: a reassessment's cancel + // then either follows this offer (later nonce, same wallet) and withdraws it, + // or finished first and the check below sees the agreement cancelled. + let reassess_guard = ctx.reassess_lock.offer().ok_or(DEFER_WHILE_REASSESSING)?; + + let agreement = match next_step(&ctx.registry, agreement_id).await? { + NextStep::Offer(agreement) => agreement, + NextStep::Withdraw(agreement) => { + drop(reassess_guard); + return withdraw_offer_if_stored(&ctx, &agreement).await; } - Some(a) => a, + NextStep::Skip => return Ok(()), }; // Rebuild the on-chain RCA struct from the stored terms. The bytes must be @@ -189,15 +188,133 @@ where } } + // Landed, so any later cancel follows it; stop holding up reassessments. + drop(reassess_guard); + if let Some(agreement) = cancelled_meanwhile(&ctx.registry, agreement_id).await { + return withdraw_offer_if_stored(&ctx, &agreement).await; + } + // Offer is confirmed on-chain (or was already there). The indexer-agent will // pick up the pending_rca_proposals row and call acceptIndexingAgreement. No // further enqueue needed; chain_listener detects the acceptance event. Ok(()) } +/// What this run of the job does with its agreement. +enum NextStep { + /// Still wanted: send the offer. + Offer(IndexingAgreement), + /// Dipper cancelled it: withdraw any offer an earlier attempt left on-chain. + Withdraw(IndexingAgreement), + /// Gone, expired or otherwise past offering. + Skip, +} + +async fn next_step( + registry: &R, + agreement_id: &IndexingAgreementId, +) -> JobResult { + let agreement = registry + .get_indexing_agreement_by_id(agreement_id) + .await + .map_err(|err| JobError::Fatal(err.into()))?; + Ok(match agreement { + None => { + tracing::error!( + agreement_id = %agreement_id, + "Agreement not found in registry at submit_offer" + ); + NextStep::Skip + } + Some(a) if a.status == IndexingAgreementStatus::Created => NextStep::Offer(a), + Some(a) if a.status == IndexingAgreementStatus::CanceledByRequester => { + NextStep::Withdraw(a) + } + Some(a) => { + tracing::warn!( + agreement_id = %agreement_id, + status = %a.status, + "Agreement not in Created status, skipping offer submission" + ); + NextStep::Skip + } + }) +} + +/// Withdraw the agreement's offer if one is on-chain: dipper cancelled it while +/// this job's offer was in flight (the chain listener cancels replaced agreements +/// without the reassess lock) or before this retry. A failure retries the job, +/// which comes back here through its status check. +async fn withdraw_offer_if_stored( + ctx: &Ctx, + agreement: &IndexingAgreement, +) -> JobResult<()> { + let stored = ctx + .chain_client + .agreement_still_active(agreement.id.as_bytes()) + .await + .map_err(|err| retry_withdraw(agreement, err))?; + if !stored { + return Ok(()); + } + match cancel_agreement_on_chain(&ctx.chain_client, agreement, &ctx.agreement_conf).await { + Ok(tx_hash) => { + tracing::info!( + agreement_id = %agreement.id, + tx_hash = ?tx_hash, + "Withdrew the offer of an agreement dipper had cancelled" + ); + Ok(()) + } + Err(err @ ChainClientError::MissingTermsVersionHash { .. }) => { + tracing::error!( + agreement_id = %agreement.id, + error = %err, + "Cannot withdraw the offer of a cancelled agreement; it stays open until its deadline" + ); + Err(JobError::Fatal(err.into())) + } + Err(err) => Err(retry_withdraw(agreement, err)), + } +} + +fn retry_withdraw(agreement: &IndexingAgreement, err: ChainClientError) -> JobError { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to withdraw the offer of a cancelled agreement, will retry" + ); + JobError::Retryable(err.into(), TRANSIENT_RETRY_BASE) +} + +/// The agreement, if it was cancelled after this job's status check. +async fn cancelled_meanwhile( + registry: &R, + agreement_id: &IndexingAgreementId, +) -> Option { + match registry.get_indexing_agreement_by_id(agreement_id).await { + Ok(Some(agreement)) if agreement.status == IndexingAgreementStatus::CanceledByRequester => { + Some(agreement) + } + Ok(_) => None, + Err(err) => { + tracing::warn!( + agreement_id = %agreement_id, + error = %err, + "Failed to re-read agreement after its offer landed; a cancel made meanwhile \ + would leave the offer open until its deadline" + ); + None + } + } +} + #[cfg(test)] mod tests { - use std::sync::Mutex; + use std::sync::{ + Arc, Mutex, + atomic::{AtomicBool, Ordering}, + }; use async_trait::async_trait; use thegraph_core::{ @@ -214,8 +331,11 @@ mod tests { }, }; + /// Shared with the chain mock so a test can change the row mid-send. + type SharedAgreement = Arc>>; + struct MockRegistry { - agreement: Option, + agreement: SharedAgreement, } #[async_trait] @@ -224,7 +344,7 @@ mod tests { &self, _id: &IndexingAgreementId, ) -> crate::registry::Result> { - Ok(self.agreement.clone()) + Ok(self.agreement.lock().unwrap().clone()) } async fn update_offer_tx_hash( &self, @@ -236,9 +356,21 @@ mod tests { } /// Yields the configured result once; a second call means the handler - /// retried inside one run, which must never happen. + /// retried inside one run, and a call with no result configured means the + /// handler sent when it must not. Records whether a reassessment could have + /// taken `reassess_lock` while the offer was being sent. struct MockChainClient { offer_result: Mutex, ChainClientError>>>, + reassess_lock: ReassessLock, + reassessment_could_start_mid_send: Arc>>, + /// When set, the agreement is cancelled locally while the offer is sent, + /// as the chain listener does when a replacement is accepted. + cancel_mid_send: Option, + cancelled: Arc>>, + /// Whether the agreement's offer (or the agreement) is live on-chain: set + /// by a mined offer, cleared by a cancel. + on_chain: Arc, + fail_cancel: bool, } #[async_trait] @@ -247,20 +379,37 @@ mod tests { &self, _rca: &dipper_rpc::indexer::indexer_client::sol::RecurringCollectionAgreement, ) -> Result, ChainClientError> { - self.offer_result + *self.reassessment_could_start_mid_send.lock().unwrap() = + Some(self.reassess_lock.reassessment_could_start_now()); + if let Some(agreement) = &self.cancel_mid_send + && let Some(row) = agreement.lock().unwrap().as_mut() + { + row.status = IndexingAgreementStatus::CanceledByRequester; + } + let result = self + .offer_result .lock() .unwrap() .take() - .expect("offer_via_manager called more than once") + .expect("offer_via_manager called more than once, or when it must not send"); + if matches!(result, Ok(Some(_))) { + self.on_chain.store(true, Ordering::SeqCst); + } + result } async fn cancel_via_manager( &self, _collector: Address, - _agreement_id: &[u8; 16], + agreement_id: &[u8; 16], _version_hash: B256, _options: u16, ) -> Result, ChainClientError> { - unimplemented!() + if self.fail_cancel { + return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); + } + self.cancelled.lock().unwrap().push(*agreement_id); + self.on_chain.store(false, Ordering::SeqCst); + Ok(Some(B256::repeat_byte(0xcd))) } async fn reconcile_provider( &self, @@ -280,7 +429,7 @@ mod tests { &self, _agreement_id: &[u8; 16], ) -> Result { - unimplemented!() + Ok(self.on_chain.load(Ordering::SeqCst)) } async fn latest_block_timestamp(&self) -> Result { unimplemented!() @@ -348,17 +497,219 @@ mod tests { fn ctx_with_offer_result( agreement: IndexingAgreement, offer_result: Result, ChainClientError>, + ) -> Ctx { + ctx_with_lock(agreement, Some(offer_result), ReassessLock::default()) + } + + /// `offer_result: None` makes any send panic. + fn ctx_with_lock( + agreement: IndexingAgreement, + offer_result: Option, ChainClientError>>, + reassess_lock: ReassessLock, ) -> Ctx { Ctx { registry: MockRegistry { - agreement: Some(agreement), + agreement: Arc::new(Mutex::new(Some(agreement))), }, chain_client: MockChainClient { - offer_result: Mutex::new(Some(offer_result)), + offer_result: Mutex::new(offer_result), + reassess_lock: reassess_lock.clone(), + reassessment_could_start_mid_send: Arc::default(), + cancel_mid_send: None, + cancelled: Arc::default(), + on_chain: Arc::default(), + fail_cancel: false, }, + agreement_conf: Arc::new(test_agreement_conf()), + reassess_lock, } } + fn test_agreement_conf() -> crate::config::IndexingAgreementConfig { + crate::config::IndexingAgreementConfig { + data_service: Address::ZERO, + recurring_collector: Address::ZERO, + recurring_agreement_manager: Address::ZERO, + max_agreement_grt_per_30_days: 0.0, + max_seconds_per_collection: 0, + min_seconds_per_collection: 0, + duration_seconds: 0, + deadline_seconds: 0, + max_grt_per_30_days: std::collections::BTreeMap::new(), + max_grt_per_billion_entities_per_30_days: 0.0, + declined_indexer_lookback_days: 0, + price_rejection_lookback_days: 0, + transient_rejection_lookback_minutes: 0, + uncertain_rejection_lookback_days: 0, + unresponsive_indexer_lookback_days: 0, + mass_unresponsive_trip_fraction: 0.5, + mass_unresponsive_reset_fraction: 0.25, + dips_accepting_snapshot_max_age_hours: 48, + dips_accepting_cache_ttl_seconds: 300, + max_in_flight_offers_per_indexer: None, + max_in_flight_offers_total: None, + } + } + + #[tokio::test] + async fn withdraws_its_offer_when_the_agreement_was_cancelled_while_it_was_sent() { + //* Arrange - the agreement is cancelled locally while the offer is in flight, + // so the cancel found nothing to withdraw and the offer would stay open + let mut agreement = make_test_agreement(); + agreement.terms_version_hash = Some(vec![7u8; 32]); + let agreement_id = agreement.id; + let message = make_message(agreement_id); + let mut ctx = ctx_with_offer_result(agreement, Ok(Some(B256::repeat_byte(0xab)))); + ctx.chain_client.cancel_mid_send = Some(ctx.registry.agreement.clone()); + let cancelled = ctx.chain_client.cancelled.clone(); + + //* Act + let result = handle(ctx, &message).await; + + //* Assert + assert!(result.is_ok(), "got {result:?}"); + assert_eq!(*cancelled.lock().unwrap(), vec![*agreement_id.as_bytes()]); + } + + #[tokio::test] + async fn keeps_its_offer_when_the_agreement_is_still_wanted() { + //* Arrange + let agreement = make_test_agreement(); + let message = make_message(agreement.id); + let ctx = ctx_with_offer_result(agreement, Ok(Some(B256::repeat_byte(0xab)))); + let cancelled = ctx.chain_client.cancelled.clone(); + + //* Act + let result = handle(ctx, &message).await; + + //* Assert + assert!(result.is_ok(), "got {result:?}"); + assert!(cancelled.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn waits_without_sending_while_a_reassessment_holds_the_lock() { + //* Arrange - a reassessment holds the lock; no offer result is configured, + // so a send would panic + let agreement = make_test_agreement(); + let message = make_message(agreement.id); + let lock = ReassessLock::default(); + let _reassessment = lock.reassessment().await.expect("lock is free"); + let ctx = ctx_with_lock(agreement, None, lock); + + //* Act + let result = handle(ctx, &message).await; + + //* Assert - deferred, not failed, so it runs once the reassessment ends + assert!( + matches!(result, Err(JobError::Deferred(delay)) if delay == Duration::from_secs(1)), + "an offer must wait while a reassessment runs, got {result:?}" + ); + } + + #[tokio::test] + async fn no_reassessment_can_start_while_the_offer_is_sent() { + //* Arrange + let agreement = make_test_agreement(); + let message = make_message(agreement.id); + let ctx = ctx_with_offer_result(agreement, Ok(Some(B256::repeat_byte(0xab)))); + let lock = ctx.reassess_lock.clone(); + let could_start = ctx.chain_client.reassessment_could_start_mid_send.clone(); + + //* Act + let result = handle(ctx, &message).await; + + //* Assert - the lock was held during the send and is free once the job ends + assert!(result.is_ok(), "got {result:?}"); + assert_eq!( + *could_start.lock().unwrap(), + Some(false), + "a reassessment must not be able to start while the offer is sent" + ); + assert!( + lock.reassessment_could_start_now(), + "the job must release the lock when it ends" + ); + } + + #[tokio::test] + async fn sends_alongside_another_offer() { + //* Arrange - another offer job holds the lock shared + let agreement = make_test_agreement(); + let message = make_message(agreement.id); + let lock = ReassessLock::default(); + let _other_offer = lock.offer().expect("lock is free"); + let ctx = ctx_with_lock(agreement, Some(Ok(Some(B256::repeat_byte(0xab)))), lock); + + //* Act + let result = handle(ctx, &message).await; + + //* Assert + assert!( + result.is_ok(), + "offers must not wait for each other, got {result:?}" + ); + } + + #[tokio::test] + async fn skips_an_agreement_a_reassessment_already_cancelled() { + //* Arrange - the reassessment ran first and cancelled the agreement; no offer + // result is configured, so a send would panic + let mut agreement = make_test_agreement(); + agreement.status = IndexingAgreementStatus::CanceledByRequester; + let message = make_message(agreement.id); + let ctx = ctx_with_lock(agreement, None, ReassessLock::default()); + let cancelled = ctx.chain_client.cancelled.clone(); + + //* Act + let result = handle(ctx, &message).await; + + //* Assert - and with no offer on-chain, nothing to withdraw + assert!(result.is_ok(), "got {result:?}"); + assert!(cancelled.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn withdraws_a_stored_offer_of_an_agreement_already_cancelled() { + //* Arrange - an earlier attempt sent the offer, then the agreement was + // cancelled before this retry; no offer result, so a send would panic + let mut agreement = make_test_agreement(); + agreement.status = IndexingAgreementStatus::CanceledByRequester; + agreement.terms_version_hash = Some(vec![7u8; 32]); + let agreement_id = agreement.id; + let message = make_message(agreement_id); + let ctx = ctx_with_lock(agreement, None, ReassessLock::default()); + ctx.chain_client.on_chain.store(true, Ordering::SeqCst); + let cancelled = ctx.chain_client.cancelled.clone(); + + //* Act + let result = handle(ctx, &message).await; + + //* Assert + assert!(result.is_ok(), "got {result:?}"); + assert_eq!(*cancelled.lock().unwrap(), vec![*agreement_id.as_bytes()]); + } + + #[tokio::test] + async fn retries_a_withdraw_that_fails() { + //* Arrange - cancelled while the offer was in flight, and the withdraw fails + let mut agreement = make_test_agreement(); + agreement.terms_version_hash = Some(vec![7u8; 32]); + let message = make_message(agreement.id); + let mut ctx = ctx_with_offer_result(agreement, Ok(Some(B256::repeat_byte(0xab)))); + ctx.chain_client.cancel_mid_send = Some(ctx.registry.agreement.clone()); + ctx.chain_client.fail_cancel = true; + + //* Act + let result = handle(ctx, &message).await; + + //* Assert - a retry finds the row cancelled and withdraws through the check + assert!( + matches!(result, Err(JobError::Retryable(_, _))), + "got {result:?}" + ); + } + #[tokio::test] async fn contract_revert_fails_the_job_instead_of_retrying() { //* Arrange - the offer reverts with the observed window selector diff --git a/bin/dipper-service/src/worker/reassess_lock.rs b/bin/dipper-service/src/worker/reassess_lock.rs new file mode 100644 index 00000000..3fddb696 --- /dev/null +++ b/bin/dipper-service/src/worker/reassess_lock.rs @@ -0,0 +1,156 @@ +//! Keeps a reassessment's cancels and offers apart: an offer sent just after such +//! a cancel would land after it and stay open. Offers hold this shared from their +//! status check until they land. In-process only (dipper is single-replica). + +use std::{sync::Arc, time::Duration}; + +use tokio::sync::{Mutex, OwnedMutexGuard, OwnedRwLockReadGuard, OwnedRwLockWriteGuard, RwLock}; + +/// Longest a reassessment waits for offers already being sent to land before it +/// gives up on its cancels. An offer holds the lock through its receipt wait +/// (15 s) and its turn at the chain client's send, so a slow RPC can outlast this. +const OFFER_DRAIN_WAIT: Duration = Duration::from_secs(30); + +#[derive(Clone, Default)] +pub struct ReassessLock { + /// Only one reassessment runs at a time across every worker loop, so two + /// loops can't diff the same baseline and both create agreements. + reassessment: Arc>, + /// Offers hold it shared; a reassessment holds it exclusively while it + /// cancels agreements whose offers may be in flight. + offers: Arc>, +} + +/// The running reassessment. +pub struct Reassessment { + _running: OwnedMutexGuard<()>, + offers: Arc>, +} + +impl ReassessLock { + /// Start a reassessment, or `None` (defer) if another one is running. + pub fn start_reassessment(&self) -> Option { + Some(Reassessment { + _running: self.reassessment.clone().try_lock_owned().ok()?, + offers: self.offers.clone(), + }) + } + + /// A reassessment holding offers back, as one does while it cancels. + #[cfg(test)] + pub async fn reassessment(&self) -> Option<(Reassessment, OwnedRwLockWriteGuard<()>)> { + let reassessment = self.start_reassessment()?; + let offers_held_back = reassessment.hold_back_offers().await?; + Some((reassessment, offers_held_back)) + } + + /// Start an offer submission, or `None` (defer) while a reassessment is + /// cancelling or waiting to. Offers never wait for each other. + pub fn offer(&self) -> Option> { + self.offers.clone().try_read_owned().ok() + } + + /// Whether a reassessment could start and take the lock right now. + #[cfg(test)] + pub fn reassessment_could_start_now(&self) -> bool { + self.reassessment.try_lock().is_ok() && self.offers.try_write().is_ok() + } +} + +impl Reassessment { + /// Wait for offers in flight to land and keep new ones from starting until + /// the guard drops, or `None` after `OFFER_DRAIN_WAIT`. Waiting already holds + /// new offers back, so a stream of them can't keep it out. + pub async fn hold_back_offers(&self) -> Option> { + let held = tokio::time::timeout(OFFER_DRAIN_WAIT, self.offers.clone().write_owned()).await; + if held.is_err() { + tracing::warn!( + wait_secs = OFFER_DRAIN_WAIT.as_secs(), + "Offers in flight did not land in time; skipping cancels that could race them" + ); + } + held.ok() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn offers_run_alongside_each_other() { + let lock = ReassessLock::default(); + let _first = lock.offer().expect("first offer starts"); + assert!(lock.offer().is_some(), "a second offer must not wait"); + } + + #[tokio::test] + async fn offers_run_while_a_reassessment_is_not_cancelling() { + let lock = ReassessLock::default(); + let _running = lock.start_reassessment().expect("lock is free"); + assert!(lock.offer().is_some()); + } + + #[tokio::test] + async fn no_offer_starts_while_a_reassessment_holds_them_back() { + let lock = ReassessLock::default(); + let _reassessment = lock.reassessment().await.expect("lock is free"); + assert!(lock.offer().is_none()); + } + + #[tokio::test(start_paused = true)] + async fn a_second_reassessment_defers_without_waiting() { + let lock = ReassessLock::default(); + let _running = lock.reassessment().await.expect("lock is free"); + + let started = tokio::time::Instant::now(); + assert!(lock.reassessment().await.is_none()); + assert_eq!( + started.elapsed(), + Duration::ZERO, + "it must not park its loop" + ); + } + + #[tokio::test] + async fn a_reassessment_waits_for_an_offer_in_flight_and_blocks_new_ones() { + let lock = ReassessLock::default(); + let in_flight = lock.offer().expect("offer starts"); + + let waiting = tokio::spawn({ + let lock = lock.clone(); + async move { lock.reassessment().await.is_some() } + }); + // Let the reassessment queue behind the offer in flight. + while lock.reassessment.try_lock().is_ok() { + tokio::task::yield_now().await; + } + tokio::task::yield_now().await; + + assert!( + lock.offer().is_none(), + "a new offer must not start ahead of the waiting reassessment" + ); + drop(in_flight); + assert!( + waiting.await.unwrap(), + "the reassessment runs once the offer lands" + ); + } + + #[tokio::test(start_paused = true)] + async fn holding_back_offers_gives_up_when_they_do_not_land_in_time() { + let lock = ReassessLock::default(); + let _stuck = lock.offer().expect("offer starts"); + + let started = tokio::time::Instant::now(); + assert!(lock.reassessment().await.is_none()); + assert_eq!(started.elapsed(), OFFER_DRAIN_WAIT); + + drop(_stuck); + assert!( + lock.reassessment_could_start_now(), + "giving up must release both locks" + ); + } +} diff --git a/bin/dipper-service/src/worker/service.rs b/bin/dipper-service/src/worker/service.rs index 7605945f..659ca90c 100644 --- a/bin/dipper-service/src/worker/service.rs +++ b/bin/dipper-service/src/worker/service.rs @@ -619,14 +619,14 @@ where } } Err(JobError::Deferred(delay)) => { - // Couldn't run now (another reassessment holds the global lock); - // re-queue at a flat delay without counting a failed attempt. - // Logged at info so sustained contention is visible per job id. + // Couldn't run now (the global reassess lock is busy, or pacing + // held it back); re-queue at a flat delay without counting a failed + // attempt. Logged at info so sustained contention is visible per job id. let scheduled_for = OffsetDateTime::now_utc() + delay; tracing::info!( job = %job.id(), delay_secs = %delay.as_secs(), - "Deferring job; another reassessment holds the global lock, will retry" + "Deferring job; it can't run yet, will retry" ); if let Err(err) = job.reschedule(scheduled_for).await { tracing::error!(error=?err, "Failed to reschedule deferred job"); From 8e54fd8040f22ccbf8e156bad31de8abd643308c Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Fri, 2 Oct 2026 15:53:25 +0300 Subject: [PATCH 04/26] fix: end agreements accepted after dipper cancelled them (#711) * fix(listener): cancel an agreement accepted after dipper cancelled it An indexer could accept an offer dipper had already cancelled, and the listener ignored that, so the agreement stayed live. The listener now queues an on-chain cancel; the job checks the chain first and logs 1 ERROR line with a fixed event field for each agreement it ends. * fix(worker): give the on-chain cancel job about 40 minutes of retries The job that cancels an unwanted live agreement gave up after 2 retries, about 90 seconds, which a short RPC outage outlasts, and giving up leaves the indexer paid. It now retries 10 times. * fix(worker): alert on every failed cancel and run one cancel per agreement A failed cancel of a live agreement only logged a warning, so a job that ran out of retries went unnoticed; each failure is now an alert line. Two jobs for one agreement could both cancel it and both alert; the second now leaves it to the first. Tests check the alert lines. * fix(reassess): mark unaccepted agreements cancelled if the cancel fails Left unaccepted after a failed on-chain cancel, its offer job would still send the offer and the indexer could accept an agreement dipper wanted gone. Marked cancelled, the job withdraws any offer instead, and the chain listener cancels the agreement if the indexer accepts one anyway. --- .../src/network/service/chain_listener.rs | 75 +++ .../cancel_rejected_agreement_on_chain.rs | 469 ++++++++++++++++-- .../handlers/reassess_indexing_request.rs | 86 +++- .../src/worker/service_queue.rs | 32 +- 4 files changed, 584 insertions(+), 78 deletions(-) diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index 2082646b..5ea95eb4 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -876,6 +876,8 @@ where } } + queue_cancel_if_cancelled_but_accepted(snapshot, &agreement, worker_queue).await?; + // Both transitions are applied atomically downstream so the // Accept-then-Cancel-in-one-snapshot path can't leak an intermediate // AcceptedOnChain to concurrent readers. @@ -933,6 +935,31 @@ where } } +/// Safety net for an agreement dipper cancelled whose offer the indexer accepted +/// anyway, such as one that landed after dipper's cancel. Nothing else would end +/// it: reconciliation ignores an accept on a cancelled row. The job reads the +/// chain before acting, so a stale snapshot of an agreement already ended is free. +async fn queue_cancel_if_cancelled_but_accepted( + snapshot: &AgreementStateSnapshot, + agreement: &IndexingAgreement, + worker_queue: &W, +) -> anyhow::Result<()> { + if agreement.status == IndexingAgreementStatus::CanceledByRequester + && snapshot.state.reached_accepted() + && !snapshot.state.is_canceled() + { + tracing::warn!( + agreement_id = %agreement.id, + indexer = %snapshot.indexer, + "Cancelled agreement accepted on-chain, queuing cancellation" + ); + worker_queue + .cancel_rejected_agreement_on_chain(agreement.id, JobPriority::Background) + .await?; + } + Ok(()) +} + /// Log the transition that landed and, on fresh accepts, fan out the /// linked pending cancellations. #[expect( @@ -2893,6 +2920,54 @@ mod tests { assert!(!worker_queue.was_cancellation_queued(&agreement_id)); } + #[tokio::test] + async fn test_reconcile_queues_cancel_for_cancelled_agreement_accepted_on_chain() { + // Dipper cancelled the agreement locally, but the indexer accepted its offer + // (for example one that landed after dipper's cancel). Nothing else would end + // it, so the listener queues an on-chain cancel and leaves the row cancelled. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let worker_queue = MockWorkerQueue::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); + + let snapshot = make_snapshot(agreement_id, AgreementState::Accepted, Address::ZERO); + let result = reconcile_agreement( + &snapshot, + ®istry, + &worker_queue, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok()); + assert!(worker_queue.was_cancellation_queued(&agreement_id)); + assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); + } + + #[tokio::test] + async fn test_reconcile_cancelled_agreement_already_cancelled_on_chain_queues_nothing() { + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let worker_queue = MockWorkerQueue::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); + + let snapshot = make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); + let result = reconcile_agreement( + &snapshot, + ®istry, + &worker_queue, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok()); + assert!(!worker_queue.was_cancellation_queued(&agreement_id)); + } + #[tokio::test] async fn test_reconcile_ignores_already_canceled() { let registry = MockRegistry::new(); diff --git a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs index d11e8e72..8905c54e 100644 --- a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs +++ b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs @@ -1,10 +1,12 @@ -//! Cancel a rejected agreement that was accepted on-chain -//! -//! When an indexer rejects an agreement off-chain but later accepts it on-chain, -//! the chain listener detects this and queues this job to cancel the agreement -//! via the RecurringAgreementManager. - -use std::{sync::Arc, time::Duration}; +//! Cancel on-chain, via the RecurringAgreementManager, an agreement dipper doesn't +//! want that was accepted anyway: one the indexer rejected off-chain, or one dipper +//! had already cancelled. The chain listener queues it. + +use std::{ + collections::HashSet, + sync::{Arc, LazyLock, Mutex, PoisonError}, + time::Duration, +}; use dipper_core::ids::IndexingAgreementId; @@ -12,7 +14,7 @@ use crate::{ cancel_dispatch::cancel_agreement_on_chain, chain_client::{ChainClient, ChainClientError}, config::IndexingAgreementConfig, - registry::{AgreementRegistry, IndexingAgreementStatus}, + registry::{AgreementRegistry, IndexingAgreement, IndexingAgreementStatus}, worker::result::{JobError, JobResult}, }; @@ -22,17 +24,15 @@ pub struct Ctx { pub agreement_conf: Arc, } -/// Cancel a rejected agreement on-chain. +/// Cancel on-chain an agreement dipper rejected or cancelled that was accepted anyway. #[derive(Debug, serde::Serialize, serde::Deserialize)] pub struct Message { pub agreement_id: IndexingAgreementId, } -/// Cancel a rejected agreement on-chain. -/// -/// This is called when an indexer rejected the proposal off-chain but then accepted -/// on-chain anyway. We cancel the agreement via `cancelIndexingAgreementByPayer` to -/// ensure the indexer doesn't receive payment for work we didn't want. +/// Cancel on-chain an agreement dipper rejected or cancelled that was accepted +/// anyway, via `cancelIndexingAgreementByPayer`, so the indexer isn't paid for +/// work dipper didn't want. #[expect( clippy::cognitive_complexity, reason = "predates this lint; fix when next touched" @@ -60,15 +60,20 @@ where } }; - // Verify the agreement is in Rejected status (off-chain rejection that got accepted on-chain) - // The chain listener should only queue this job for Rejected agreements - if agreement.status != IndexingAgreementStatus::Rejected { - tracing::warn!( - agreement_id = %agreement_id, - status = %agreement.status, - "Agreement not in Rejected status, skipping on-chain cancellation" - ); - return Ok(()); + // The chain listener queues this job only for these two statuses. + match agreement.status { + IndexingAgreementStatus::Rejected => {} + IndexingAgreementStatus::CanceledByRequester => { + return cancel_live_agreement_dipper_cancelled(&ctx, &agreement).await; + } + status => { + tracing::warn!( + agreement_id = %agreement_id, + status = %status, + "Agreement neither Rejected nor CanceledByRequester, skipping on-chain cancellation" + ); + return Ok(()); + } } tracing::info!( @@ -116,12 +121,9 @@ where } }; - // When the row was actually flipped to terminal, record the cancel audit so - // the chain_listener's `terminated` sweep announces it durably. The accept - // was recorded when the rejected-then-accepted anomaly was first detected, so - // the row is sweep-eligible. If the mark failed, the row stays `Rejected` and - // the chain_listener observes the on-chain cancel and flips it itself, then - // the same sweep emits -- so nothing is lost either way. + // Once the row is terminal, the cancel audit lets the `terminated` sweep + // announce it (the accept was recorded when the listener queued this job). + // If the mark failed, the listener sees the on-chain cancel and flips it. if mark_cancellation_complete(&ctx.registry, agreement_id).await { let manager = ctx.agreement_conf.recurring_agreement_manager().to_string(); if let Err(err) = ctx @@ -145,16 +147,110 @@ where Ok(()) } -/// Flip the local row to CanceledByRequester after either a fresh on-chain -/// cancel or the discovery that the agreement was already canceled on-chain. -/// Failures here are logged but not fatal — the on-chain side is already in -/// the right state, so the next reconciliation pass can re-attempt the DB -/// update without risking a duplicate transaction. -/// -/// Returns `true` when the row was marked terminal. The caller emits the -/// `terminated` event only on `true`: if the mark failed the row stays -/// `Rejected` (non-terminal), so the chain_listener will observe the on-chain -/// cancel and emit `terminated` itself — emitting here too would duplicate. +/// Agreements a job in this process is cancelling right now. The listener can +/// queue one twice before the chain shows it ended, and two jobs at once would +/// both cancel it and both alert. +static CANCELLING: LazyLock>> = LazyLock::new(Mutex::default); + +/// A job's claim on cancelling one agreement, released when the job ends. +struct Cancelling(IndexingAgreementId); + +impl Cancelling { + fn claim(agreement_id: IndexingAgreementId) -> Option { + // Unlock before a claim exists: dropping one locks the set again. + let claimed = CANCELLING + .lock() + .unwrap_or_else(PoisonError::into_inner) + .insert(agreement_id); + claimed.then(|| Self(agreement_id)) + } +} + +impl Drop for Cancelling { + fn drop(&mut self) { + let mut cancelling = CANCELLING.lock().unwrap_or_else(PoisonError::into_inner); + cancelling.remove(&self.0); + } +} + +/// Cancel on-chain an agreement dipper had already cancelled that the indexer +/// accepted anyway; the row is already terminal. The chain is read first, so a +/// stale snapshot of one dipper has since ended raises no alert. +async fn cancel_live_agreement_dipper_cancelled( + ctx: &Ctx, + agreement: &IndexingAgreement, +) -> JobResult<()> +where + T: ChainClient, +{ + let Some(_claim) = Cancelling::claim(agreement.id) else { + tracing::info!( + agreement_id = %agreement.id, + "Another job is already cancelling this agreement" + ); + return Ok(()); + }; + if !live_on_chain(&ctx.chain_client, agreement).await? { + tracing::info!( + agreement_id = %agreement.id, + "Cancelled agreement is no longer live on-chain; nothing to cancel" + ); + return Ok(()); + } + + match cancel_agreement_on_chain(&ctx.chain_client, agreement, &ctx.agreement_conf).await { + Ok(tx_hash) => { + let tx = tx_hash.map_or_else(|| "none".to_owned(), |hash| hash.to_string()); + log_caught_live_agreement(agreement, "cancelled", &tx); + Ok(()) + } + Err(err @ ChainClientError::MissingTermsVersionHash { .. }) => { + log_caught_live_agreement(agreement, "cancel_impossible", &err.to_string()); + Err(JobError::Fatal(err.into())) + } + Err(err) => { + log_caught_live_agreement(agreement, "cancel_failed", &err.to_string()); + Err(JobError::Retryable(err.into(), Duration::from_secs(30))) + } + } +} + +/// ERROR line with stable `event` and `outcome` for alerting: `cancelled` once per +/// agreement ended, `cancel_failed` per failed attempt (so running out of retries +/// is never silent), `cancel_impossible` when it never can be. +fn log_caught_live_agreement(agreement: &IndexingAgreement, outcome: &str, detail: &str) { + tracing::error!( + event = "cancelled_agreement_live_on_chain", + outcome, + agreement_id = %agreement.id, + indexer_id = %agreement.indexer.id, + indexing_request_id = %agreement.indexing_request_id, + detail, + "Agreement dipper had cancelled is live on-chain" + ); +} + +/// Read the chain for whether the agreement is still live; a failed read retries. +async fn live_on_chain( + chain_client: &T, + agreement: &IndexingAgreement, +) -> JobResult { + chain_client + .agreement_still_active(agreement.id.as_bytes()) + .await + .map_err(|err| { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to read whether a cancelled agreement is live on-chain, will retry" + ); + JobError::Retryable(err.into(), Duration::from_secs(30)) + }) +} + +/// Flip the row to CanceledByRequester once the chain shows it cancelled. A failure +/// is logged, not fatal: the chain is already right and the listener retries the +/// DB update. Returns whether the row is now terminal, gating the cancel audit. async fn mark_cancellation_complete(registry: &R, agreement_id: &IndexingAgreementId) -> bool where R: AgreementRegistry + Sync, @@ -186,7 +282,10 @@ where #[cfg(test)] mod tests { - use std::sync::Mutex; + use std::sync::{ + Mutex, + atomic::{AtomicBool, Ordering}, + }; use async_trait::async_trait; use dipper_core::ids::IndexingRequestId; @@ -211,10 +310,9 @@ mod tests { // Mock implementations // ========================================================================= - /// Registry that returns a single configurable agreement and records the - /// terminal-cancel transition + cancel-audit calls the handler drives. - /// `Clone` shares the tracked state (Arc), so a test can clone one into the - /// `Ctx` and still assert on the original after `handle` consumes the ctx. + /// Returns one configurable agreement and records the terminal-cancel and + /// cancel-audit calls. Clones share state, so a test can still assert on its + /// copy after `handle` consumes the ctx. #[derive(Clone)] struct MockRegistry { agreement: Arc>>, @@ -455,10 +553,24 @@ mod tests { } /// Chain client whose manager cancel always mines (returns a tx hash) and - /// whose post-cancel liveness read reports the agreement is no longer active, - /// so the cancel is confirmed. - #[derive(Default)] - struct MockChainClient; + /// ends the agreement, so the post-cancel liveness read confirms it. `live` + /// is whether the agreement is live on-chain before any cancel lands. + #[derive(Default, Clone)] + struct MockChainClient { + live: Arc, + cancelled: Arc>>, + fail_liveness_read: bool, + fail_cancel: bool, + } + + impl MockChainClient { + fn live() -> Self { + Self { + live: Arc::new(AtomicBool::new(true)), + ..Self::default() + } + } + } #[async_trait] impl ChainClient for MockChainClient { @@ -478,10 +590,15 @@ mod tests { async fn cancel_via_manager( &self, _collector: Address, - _agreement_id: &[u8; 16], + agreement_id: &[u8; 16], _version_hash: B256, _options: u16, ) -> Result, ChainClientError> { + if self.fail_cancel { + return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); + } + self.cancelled.lock().unwrap().push(*agreement_id); + self.live.store(false, Ordering::SeqCst); Ok(Some(B256::ZERO)) } @@ -504,7 +621,10 @@ mod tests { &self, _agreement_id: &[u8; 16], ) -> Result { - Ok(false) + if self.fail_liveness_read { + return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); + } + Ok(self.live.load(Ordering::SeqCst)) } } @@ -585,13 +705,253 @@ mod tests { // ========================================================================= fn ctx_for(registry: MockRegistry) -> Ctx { + ctx_with_chain(registry, MockChainClient::default()) + } + + fn ctx_with_chain( + registry: MockRegistry, + chain_client: MockChainClient, + ) -> Ctx { Ctx { registry, - chain_client: MockChainClient, + chain_client, agreement_conf: test_agreement_conf(), } } + /// Records the `outcome` of every ERROR line carrying the alert's `event`. + #[derive(Clone, Default)] + struct AlertLines(Arc>>); + + impl AlertLines { + fn outcomes(&self) -> Vec { + self.0.lock().unwrap().clone() + } + } + + impl tracing_subscriber::Layer for AlertLines { + fn on_event( + &self, + event: &tracing::Event<'_>, + _ctx: tracing_subscriber::layer::Context<'_, S>, + ) { + #[derive(Default)] + struct Fields { + event: Option, + outcome: Option, + } + impl tracing::field::Visit for Fields { + fn record_str(&mut self, field: &tracing::field::Field, value: &str) { + match field.name() { + "event" => self.event = Some(value.to_owned()), + "outcome" => self.outcome = Some(value.to_owned()), + _ => {} + } + } + fn record_debug( + &mut self, + _field: &tracing::field::Field, + _value: &dyn std::fmt::Debug, + ) { + } + } + if *event.metadata().level() != tracing::Level::ERROR { + return; + } + let mut fields = Fields::default(); + event.record(&mut fields); + if fields.event.as_deref() == Some("cancelled_agreement_live_on_chain") { + self.0 + .lock() + .unwrap() + .push(fields.outcome.unwrap_or_default()); + } + } + } + + /// Run `handle` with alert lines captured; tokio tests run on one thread. + async fn handle_capturing_alerts( + ctx: Ctx, + agreement_id: IndexingAgreementId, + ) -> (JobResult<()>, Vec) { + use tracing_subscriber::layer::SubscriberExt; + let alerts = AlertLines::default(); + let _guard = + tracing::subscriber::set_default(tracing_subscriber::registry().with(alerts.clone())); + let result = handle(ctx, &Message { agreement_id }).await; + (result, alerts.outcomes()) + } + + #[tokio::test] + async fn logs_one_alert_line_for_a_live_agreement_it_ends() { + let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); + let agreement_id = agreement.id; + let ctx = ctx_with_chain(MockRegistry::new(agreement), MockChainClient::live()); + + let (result, alerts) = handle_capturing_alerts(ctx, agreement_id).await; + + assert!(result.is_ok(), "got {result:?}"); + assert_eq!(alerts, vec!["cancelled"]); + } + + #[tokio::test] + async fn logs_no_alert_line_for_an_agreement_that_already_ended() { + let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); + let agreement_id = agreement.id; + let ctx = ctx_with_chain(MockRegistry::new(agreement), MockChainClient::default()); + + let (result, alerts) = handle_capturing_alerts(ctx, agreement_id).await; + + assert!(result.is_ok(), "got {result:?}"); + assert!( + alerts.is_empty(), + "a stale snapshot must not alert: {alerts:?}" + ); + } + + #[tokio::test] + async fn logs_an_alert_line_for_each_failed_attempt_at_a_live_agreement() { + // The job can run out of retries; each failure is visible, so a live + // agreement dipper couldn't end is never silent. + let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); + let agreement_id = agreement.id; + let chain = MockChainClient { + fail_cancel: true, + ..MockChainClient::live() + }; + let ctx = ctx_with_chain(MockRegistry::new(agreement), chain); + + let (result, alerts) = handle_capturing_alerts(ctx, agreement_id).await; + + assert!( + matches!(result, Err(JobError::Retryable(_, _))), + "got {result:?}" + ); + assert_eq!(alerts, vec!["cancel_failed"]); + } + + #[tokio::test] + async fn fails_with_an_alert_line_when_a_live_agreement_cannot_be_cancelled() { + let mut agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); + agreement.terms_version_hash = None; + let agreement_id = agreement.id; + let chain = MockChainClient::live(); + let ctx = ctx_with_chain(MockRegistry::new(agreement), chain.clone()); + + let (result, alerts) = handle_capturing_alerts(ctx, agreement_id).await; + + assert!(matches!(result, Err(JobError::Fatal(_))), "got {result:?}"); + assert_eq!(alerts, vec!["cancel_impossible"]); + assert!(chain.cancelled.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn a_second_job_for_the_same_agreement_leaves_it_to_the_first() { + // The listener can queue an agreement twice before the chain shows it + // ended; two jobs at once would both cancel it and both alert. + let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); + let agreement_id = agreement.id; + let chain = MockChainClient::live(); + let ctx = ctx_with_chain(MockRegistry::new(agreement), chain.clone()); + let _first_job = Cancelling::claim(agreement_id).expect("not yet claimed"); + + let (result, alerts) = handle_capturing_alerts(ctx, agreement_id).await; + + assert!(result.is_ok(), "got {result:?}"); + assert!(alerts.is_empty(), "{alerts:?}"); + assert!(chain.cancelled.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn cancels_a_live_agreement_dipper_had_cancelled() { + // Dipper cancelled the agreement locally, but the indexer accepted its offer + // anyway. The job ends it on-chain and leaves the already-terminal row alone. + let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); + let agreement_id = agreement.id; + let registry = MockRegistry::new(agreement); + let chain = MockChainClient::live(); + + let result = handle( + ctx_with_chain(registry.clone(), chain.clone()), + &Message { agreement_id }, + ) + .await; + + assert!(result.is_ok(), "got {result:?}"); + assert_eq!( + *chain.cancelled.lock().unwrap(), + vec![*agreement_id.as_bytes()] + ); + assert!( + registry.marked_canceled.lock().unwrap().is_empty(), + "the row is already cancelled; it must not be marked again" + ); + } + + #[tokio::test] + async fn leaves_alone_an_agreement_dipper_cancelled_that_already_ended() { + // A stale snapshot can report an accept after dipper's own cancel already + // ended the agreement. Reading the chain first avoids a pointless cancel. + let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); + let agreement_id = agreement.id; + let chain = MockChainClient::default(); + + let result = handle( + ctx_with_chain(MockRegistry::new(agreement), chain.clone()), + &Message { agreement_id }, + ) + .await; + + assert!(result.is_ok(), "got {result:?}"); + assert!( + chain.cancelled.lock().unwrap().is_empty(), + "nothing to cancel" + ); + } + + #[tokio::test] + async fn retries_without_cancelling_when_the_chain_cannot_be_read() { + let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); + let agreement_id = agreement.id; + let chain = MockChainClient { + fail_liveness_read: true, + ..MockChainClient::live() + }; + + let result = handle( + ctx_with_chain(MockRegistry::new(agreement), chain.clone()), + &Message { agreement_id }, + ) + .await; + + assert!( + matches!(result, Err(JobError::Retryable(_, _))), + "got {result:?}" + ); + assert!(chain.cancelled.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn retries_when_cancelling_a_live_agreement_fails() { + let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); + let agreement_id = agreement.id; + let chain = MockChainClient { + fail_cancel: true, + ..MockChainClient::live() + }; + + let result = handle( + ctx_with_chain(MockRegistry::new(agreement), chain), + &Message { agreement_id }, + ) + .await; + + assert!( + matches!(result, Err(JobError::Retryable(_, _))), + "got {result:?}" + ); + } + #[tokio::test] async fn rejected_agreement_records_cancel_audit_once() { // The handler no longer emits `terminated` directly: it records the cancel @@ -610,10 +970,9 @@ mod tests { #[tokio::test] async fn failed_local_mark_records_no_cancel_audit() { - // On-chain cancel succeeds but the local DB mark fails, leaving the row - // non-terminal. The handler must NOT record cancel audit -- the - // chain_listener will observe the on-chain cancel, flip the row, and the - // sweep emits from there. + // The on-chain cancel succeeds but the DB mark fails, so the row stays + // non-terminal and no cancel audit is recorded: the listener sees the + // on-chain cancel, flips the row, and the sweep emits from there. let agreement = make_agreement(IndexingAgreementStatus::Rejected); let agreement_id = agreement.id; let registry = MockRegistry::with_mark_failure(agreement); diff --git a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs index 5027872c..e48850d7 100644 --- a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs +++ b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs @@ -650,14 +650,6 @@ where None }; for old_agreement in unpaired { - if offer_may_be_in_flight(old_agreement) && offers_held_back.is_none() { - tracing::warn!( - agreement_id = %old_agreement.id, - "Offers in flight did not land; will retry this cancel on next reassessment" - ); - cancel_failures += 1; - continue; - } // A Created agreement's offer may already be on-chain, where the indexer // can still accept it, so it is cancelled on-chain too: the cancel revokes // a pending offer and does nothing when none was made. Rows without a @@ -669,9 +661,21 @@ where let needs_on_chain_cancel = was_accepted || (old_agreement.status == crate::registry::IndexingAgreementStatus::Created && old_agreement.terms_version_hash.is_some()); + // An unaccepted agreement is marked cancelled even when its on-chain cancel + // can't go out or fails: its offer job then withdraws any offer instead of + // sending one, and the chain listener cancels it if the indexer accepts. + let offers_still_landing = + offer_may_be_in_flight(old_agreement) && offers_held_back.is_none(); + if offers_still_landing { + tracing::warn!( + agreement_id = %old_agreement.id, + "Offers in flight did not land; marking the agreement cancelled without an on-chain cancel" + ); + cancel_failures += 1; + } let mut on_chain_cancel_tx: Option = None; - if needs_on_chain_cancel { + if needs_on_chain_cancel && !offers_still_landing { match crate::cancel_dispatch::cancel_agreement_on_chain( &ctx.chain_client, old_agreement, @@ -699,10 +703,14 @@ where tracing::warn!( error = %err, agreement_id = %old_agreement.id, - "On-chain cancel failed; will retry on next reassessment" + was_accepted, + "On-chain cancel failed; an accepted agreement is retried later, an \ + unaccepted one is marked cancelled anyway" ); cancel_failures += 1; - continue; + if was_accepted { + continue; + } } } } @@ -767,15 +775,13 @@ where } if cancel_failures > 0 { - // Agreements whose cancel failed keep their status (AcceptedOnChain or - // Created), and two recovery paths cover them: + // An unaccepted agreement whose cancel failed was still marked cancelled + // (see above). An accepted one keeps its status, and two recovery paths + // cover it: // // - Shrink-to-zero (request now Canceled): the chain_listener's // `sweep_orphan_canceled_agreements` retries AcceptedOnChain rows on // every sweep tick (default ~5 min at fast poll, ~5 h at slow poll). - // A Created row is not retried: its offer stays open until its - // deadline, and if the indexer accepts it the listener marks it - // AcceptedOnChain, which brings it into that sweep. // - Shrink-not-zero (request still Open with too many agreements): // the periodic reassignment service re-queues reassessment at its // configured cadence (default 24 h). @@ -1076,6 +1082,7 @@ mod lifecycle_event_tests { /// When set, each cancel records whether new offers were held back. reassess_lock: Option, offers_held_back_at_cancel: Arc>>, + fail_cancel: bool, } #[async_trait] @@ -1100,6 +1107,9 @@ mod lifecycle_event_tests { _version_hash: B256, _options: u16, ) -> std::result::Result, ChainClientError> { + if self.fail_cancel { + return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); + } self.cancelled.lock().unwrap().push(*agreement_id); if let Some(lock) = &self.reassess_lock { self.offers_held_back_at_cancel @@ -1150,6 +1160,8 @@ mod lifecycle_event_tests { /// `set_indexing_request_shortfall_active` flips it and reports whether it /// changed, so tests can exercise the transition-based emit. shortfall_active: std::sync::Mutex, + /// Ids marked CanceledByRequester locally. + marked_cancelled: Arc>>, } #[async_trait] @@ -1296,8 +1308,9 @@ mod lifecycle_event_tests { // Cancel path: pre-mark the local row terminal. async fn mark_indexing_agreement_as_canceled_by_requester( &self, - _id: &IndexingAgreementId, + id: &IndexingAgreementId, ) -> RegistryResult<()> { + self.marked_cancelled.lock().unwrap().push(*id); Ok(()) } async fn apply_reconciliation( @@ -1704,6 +1717,7 @@ mod lifecycle_event_tests { chain_state_lookup_fails: false, // Already in shortfall. shortfall_active: std::sync::Mutex::new(true), + marked_cancelled: Arc::default(), }; let ctx = build_ctx( registry, @@ -1740,6 +1754,7 @@ mod lifecycle_event_tests { accepted_count: 1, chain_state_lookup_fails: false, shortfall_active: std::sync::Mutex::new(false), + marked_cancelled: Arc::default(), }; let ctx = build_ctx( registry, @@ -1810,6 +1825,7 @@ mod lifecycle_event_tests { accepted_count: 0, chain_state_lookup_fails: false, shortfall_active: std::sync::Mutex::new(false), + marked_cancelled: Arc::default(), }; let ctx = build_ctx( registry, @@ -1863,6 +1879,7 @@ mod lifecycle_event_tests { accepted_count: 0, chain_state_lookup_fails: false, shortfall_active: std::sync::Mutex::new(false), + marked_cancelled: Arc::default(), }; let ctx = build_ctx( registry, @@ -1927,6 +1944,7 @@ mod lifecycle_event_tests { accepted_count: 0, chain_state_lookup_fails: false, shortfall_active: std::sync::Mutex::new(false), + marked_cancelled: Arc::default(), }; let ctx = build_ctx( registry, @@ -2048,8 +2066,9 @@ mod lifecycle_event_tests { async fn skips_cancelling_an_unaccepted_agreement_while_an_offer_will_not_land() { // After 30 s it gives up on that cancel rather than send it ahead of the // offer, leaving it for the next reassessment. - let (ctx, _leaving) = ctx_cancelling_one(IndexingAgreementStatus::Created); + let (ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::Created); let chain = ctx.chain_client.clone(); + let marked = ctx.registry.marked_cancelled.clone(); let _stuck_offer = ctx.reassess_lock.offer().expect("lock is free"); let started = tokio::time::Instant::now(); @@ -2058,6 +2077,36 @@ mod lifecycle_event_tests { assert!(result.is_ok(), "got {result:?}"); assert_eq!(started.elapsed(), std::time::Duration::from_secs(30)); assert!(chain.cancelled.lock().unwrap().is_empty()); + // Marked cancelled anyway: the offer job withdraws its offer once it lands. + assert_eq!(*marked.lock().unwrap(), vec![leaving.id]); + } + + #[tokio::test] + async fn an_unaccepted_agreement_whose_cancel_fails_is_marked_cancelled_anyway() { + // Left unaccepted, its offer job would still send the offer. Marked + // cancelled, the job withdraws any offer instead, and the chain listener + // cancels the agreement if the indexer accepts one. + let (mut ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::Created); + ctx.chain_client.fail_cancel = true; + let marked = ctx.registry.marked_cancelled.clone(); + + let result = handle(ctx, &test_message(0)).await; + + assert!(result.is_ok(), "got {result:?}"); + assert_eq!(*marked.lock().unwrap(), vec![leaving.id]); + } + + #[tokio::test] + async fn an_accepted_agreement_whose_cancel_fails_stays_for_a_retry() { + // Marking it cancelled would leave it live with nothing to end it. + let (mut ctx, _leaving) = ctx_cancelling_one(IndexingAgreementStatus::AcceptedOnChain); + ctx.chain_client.fail_cancel = true; + let marked = ctx.registry.marked_cancelled.clone(); + + let result = handle(ctx, &test_message(0)).await; + + assert!(result.is_ok(), "got {result:?}"); + assert!(marked.lock().unwrap().is_empty()); } #[tokio::test] @@ -2073,6 +2122,7 @@ mod lifecycle_event_tests { accepted_count: 0, chain_state_lookup_fails: false, shortfall_active: std::sync::Mutex::new(false), + marked_cancelled: Arc::default(), }; let ctx = build_ctx( registry, diff --git a/bin/dipper-service/src/worker/service_queue.rs b/bin/dipper-service/src/worker/service_queue.rs index 064928d8..2a0195cd 100644 --- a/bin/dipper-service/src/worker/service_queue.rs +++ b/bin/dipper-service/src/worker/service_queue.rs @@ -12,6 +12,11 @@ use super::{ queue::{JobId, JobPriority, Queue}, }; +/// Retries for the job that cancels on-chain an agreement dipper doesn't want but +/// that went live anyway: about 40 minutes of attempts at its 30 s backoff base, +/// since giving up leaves the indexer paid. +const CANCEL_ON_CHAIN_MAX_RETRIES: u32 = 10; + #[async_trait] pub trait WorkerQueue { async fn send_indexing_agreement_proposal( @@ -130,11 +135,12 @@ where priority: JobPriority, ) -> anyhow::Result { self.queue - .push( + .push_with_max_retries( Message::CancelRejectedAgreementOnChain(CancelRejectedAgreementOnChain { agreement_id, }), priority, + CANCEL_ON_CHAIN_MAX_RETRIES, ) .await } @@ -246,10 +252,10 @@ mod tests { assert_eq!(*queue.queue.pushes.lock().unwrap(), vec![Some(4)]); } - /// Only the offer submission has a deadline to spend its retries against, - /// so every other job keeps the queue-wide budget. + /// A proposal has no deadline of its own to spend retries against, so it + /// keeps the queue-wide budget. #[tokio::test] - async fn other_jobs_keep_the_queue_default_retry_budget() { + async fn a_proposal_keeps_the_queue_default_retry_budget() { //* Arrange let queue = handle(4); @@ -265,6 +271,19 @@ mod tests { ) .await .unwrap(); + + //* Assert + assert_eq!(*queue.queue.pushes.lock().unwrap(), vec![None]); + } + + /// Giving up on cancelling a live agreement dipper doesn't want leaves the + /// indexer paid, so that job keeps trying well past the queue default. + #[tokio::test] + async fn an_on_chain_cancel_carries_its_longer_retry_budget() { + //* Arrange + let queue = handle(4); + + //* Act queue .cancel_rejected_agreement_on_chain( IndexingAgreementId::from_bytes([0; 16]), @@ -274,6 +293,9 @@ mod tests { .unwrap(); //* Assert - assert_eq!(*queue.queue.pushes.lock().unwrap(), vec![None, None]); + assert_eq!( + *queue.queue.pushes.lock().unwrap(), + vec![Some(CANCEL_ON_CHAIN_MAX_RETRIES)] + ); } } From c07ba0dc0e4152e27bafef1ceb8c193fe05919f5 Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Fri, 2 Oct 2026 15:53:25 +0300 Subject: [PATCH 05/26] fix: announce agreements that went live after being cancelled (#712) * fix(listener): record the accept of an agreement ended after cancel An agreement dipper had marked cancelled could still be accepted and then ended on-chain, but its accept was never recorded, so neither its accepted nor its terminated event reached Kafka. The listener now records both from the chain once it shows the agreement accepted and then cancelled. * fix(listener): never announce agreements accepted before events existed Agreements accepted before lifecycle events existed have no recorded accept, and cancelling one would have announced it. Only an agreement dipper cancelled while its offer could still be accepted (within an hour of the deadline) can have an accept dipper missed. * fix(listener): stop a never-accepted cancel record hiding the real end Cancelling a replaced agreement recorded the cancel even when it was never accepted. If the indexer accepted it after all, that early record won over the chain's, so the terminated event could report an end before the accept. Only an accepted agreement gets the record now. * test(listener): prove a missed accept is announced once and only whole Adds a database test that repeat records keep the first values and announce nothing once sent, and a test that a failed cancel record leaves the accept unrecorded. The failure log no longer promises a retry that only comes if the listener reads the agreement again. --- .../src/network/service/chain_listener.rs | 257 +++++++++++++++++- .../tests/it_registry_postgres.rs | 80 ++++++ 2 files changed, 330 insertions(+), 7 deletions(-) diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index 5ea95eb4..5b06f855 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -928,13 +928,65 @@ where tracing::debug!( agreement_id = %agreement.id, status = %agreement.status, - "Agreement already canceled, ignoring snapshot" + "Agreement already canceled, recording any accept dipper missed" ); + record_accept_and_cancel_from_chain(snapshot, &agreement, registry).await; } Ok(None) } } +/// How long after an offer's deadline dipper's cancel can still have raced an +/// accept the listener hadn't caught up with. +const LISTENER_LAG_SLACK_SECS: i64 = 3_600; + +/// Record the accept and cancel of an agreement dipper had already marked +/// cancelled that went live on-chain first, so its accepted and terminated +/// events go out. Cancel first: the terminated sweep waits only for the accept. +/// Existing values win, so an agreement dipper already recorded is unchanged. +async fn record_accept_and_cancel_from_chain( + snapshot: &AgreementStateSnapshot, + agreement: &IndexingAgreement, + registry: &R, +) { + if snapshot.accepted_at == 0 || !cancelled_while_offer_open(agreement) { + return; + } + let canceled_by = snapshot.canceled_by.to_string(); + let recorded = match registry + .record_cancel_audit( + &agreement.id, + snapshot.canceled_at, + &canceled_by, + Some(&snapshot.canceled_tx), + ) + .await + { + Ok(()) => { + registry + .record_accepted_audit(&agreement.id, snapshot.accepted_at, &snapshot.accepted_tx) + .await + } + Err(err) => Err(err), + }; + if let Err(err) = recorded { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "failed to record the on-chain accept and cancel of a cancelled agreement; \ + its events go out only if the listener reads this agreement again" + ); + } +} + +/// Whether dipper cancelled the agreement while its offer could still be accepted, +/// so an accept on-chain may have slipped past it. One accepted before lifecycle +/// events existed was cancelled long after its deadline and is never announced. +fn cancelled_while_offer_open(agreement: &IndexingAgreement) -> bool { + let deadline = i64::try_from(agreement.terms.deadline).unwrap_or(i64::MAX); + agreement.updated_at.unix_timestamp() <= deadline.saturating_add(LISTENER_LAG_SLACK_SECS) +} + /// Safety net for an agreement dipper cancelled whose offer the indexer accepted /// anyway, such as one that landed after dipper's cancel. Nothing else would end /// it: reconciliation ignores an accept on a cancelled row. The job reads the @@ -1183,10 +1235,11 @@ where ); // Record the cancel audit so `sweep_pending_terminated_events` can emit - // the `terminated` durably. Crucially, the sweep only emits for rows that - // were genuinely accepted on-chain (`accepted_at IS NOT NULL`): a - // proposed-but-never-accepted replacement records audit here but is never - // swept, so it produces no spurious `terminated`. + // the `terminated` durably. Only for an accepted agreement: one that never + // was gets the chain's own cancel data if the indexer accepts it after all. + if old_agreement.status != IndexingAgreementStatus::AcceptedOnChain { + continue; + } let manager = config.recurring_agreement_manager().to_string(); if let Err(err) = registry .record_cancel_audit( @@ -1902,6 +1955,10 @@ mod tests { /// Ids passed to `record_cancel_audit` -- the signal a cancel path drives /// the terminated event (the sweep emits from this audit). recorded_cancel_audit: Vec, + /// Every audit write in order, as ("cancel" | "accept", id). + audit_writes: Vec<(&'static str, IndexingAgreementId)>, + /// When true, `record_cancel_audit` fails. + fail_cancel_audit: bool, pending_cancellations: std::collections::HashMap< IndexingAgreementId, Vec, @@ -1968,6 +2025,14 @@ mod tests { self.state.lock().unwrap().canceled_request_ids.insert(id); } + /// Set when the offer could last be accepted, as seconds from now. + fn set_agreement_deadline_from_now(&self, agreement_id: IndexingAgreementId, secs: i64) { + if let Some(a) = self.state.lock().unwrap().agreements.get_mut(&agreement_id) { + let deadline = OffsetDateTime::now_utc().unix_timestamp() + secs; + a.terms.deadline = u64::try_from(deadline).unwrap(); + } + } + fn set_agreement_request_id( &self, agreement_id: IndexingAgreementId, @@ -2002,6 +2067,10 @@ mod tests { .contains(id) } + fn audit_writes(&self) -> Vec<(&'static str, IndexingAgreementId)> { + self.state.lock().unwrap().audit_writes.clone() + } + fn was_cancel_audit_recorded(&self, id: &IndexingAgreementId) -> bool { self.state .lock() @@ -2163,12 +2232,27 @@ mod tests { _canceled_at: u64, _canceled_by: &str, _canceled_tx: Option<&str>, + ) -> RegistryResult<()> { + let mut state = self.state.lock().unwrap(); + if state.fail_cancel_audit { + return Err(crate::registry::Error::NoRecordsUpdated); + } + state.recorded_cancel_audit.push(*agreement_id); + state.audit_writes.push(("cancel", *agreement_id)); + Ok(()) + } + + async fn record_accepted_audit( + &self, + agreement_id: &IndexingAgreementId, + _accepted_at: u64, + _accepted_tx: &str, ) -> RegistryResult<()> { self.state .lock() .unwrap() - .recorded_cancel_audit - .push(*agreement_id); + .audit_writes + .push(("accept", *agreement_id)); Ok(()) } @@ -2968,6 +3052,138 @@ mod tests { assert!(!worker_queue.was_cancellation_queued(&agreement_id)); } + #[tokio::test] + async fn test_reconcile_records_accept_and_cancel_of_cancelled_agreement_that_went_live() { + // Dipper had marked the agreement cancelled, but it was accepted on-chain + // before being ended there. Recording both lets the accepted and terminated + // events go out; the cancel goes first because the terminated sweep only + // waits for the accept. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let worker_queue = MockWorkerQueue::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); + // Dipper cancelled it while its offer could still be accepted. + registry.set_agreement_deadline_from_now(agreement_id, 600); + + let snapshot = make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); + let result = reconcile_agreement( + &snapshot, + ®istry, + &worker_queue, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok()); + assert_eq!( + registry.audit_writes(), + vec![("cancel", agreement_id), ("accept", agreement_id)] + ); + } + + #[tokio::test] + async fn test_reconcile_records_no_accept_when_the_cancel_record_fails() { + // An accept recorded without its cancel would let the terminated event go + // out with fallback cancel fields. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let worker_queue = MockWorkerQueue::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); + registry.set_agreement_deadline_from_now(agreement_id, 600); + registry.state.lock().unwrap().fail_cancel_audit = true; + + let snapshot = make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); + let result = reconcile_agreement( + &snapshot, + ®istry, + &worker_queue, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok(), "a failed record must not fail the snapshot"); + assert!(registry.audit_writes().is_empty()); + } + + #[tokio::test] + async fn test_reconcile_does_not_announce_an_agreement_accepted_before_events_existed() { + // Agreements accepted before lifecycle events existed have no recorded + // accept and are never announced. Dipper cancelled this one long after its + // offer's deadline, so it can't be an accept dipper missed. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let worker_queue = MockWorkerQueue::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); + registry.set_agreement_deadline_from_now(agreement_id, -2 * 86_400); + + let snapshot = make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); + let result = reconcile_agreement( + &snapshot, + ®istry, + &worker_queue, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok()); + assert!(registry.audit_writes().is_empty()); + } + + #[tokio::test] + async fn test_reconcile_records_nothing_for_cancelled_agreement_never_accepted() { + // A withdrawn offer was never accepted, so there is nothing to announce. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let worker_queue = MockWorkerQueue::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); + + let mut snapshot = + make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); + snapshot.accepted_at = 0; + let result = reconcile_agreement( + &snapshot, + ®istry, + &worker_queue, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok()); + assert!(registry.audit_writes().is_empty()); + } + + #[tokio::test] + async fn test_reconcile_records_nothing_while_cancelled_agreement_is_still_live() { + // Recording the accept now would let the terminated event go out before the + // agreement has actually ended on-chain. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let worker_queue = MockWorkerQueue::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); + + let snapshot = make_snapshot(agreement_id, AgreementState::Accepted, Address::ZERO); + let result = reconcile_agreement( + &snapshot, + ®istry, + &worker_queue, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok()); + assert!(registry.audit_writes().is_empty()); + } + #[tokio::test] async fn test_reconcile_ignores_already_canceled() { let registry = MockRegistry::new(); @@ -3144,6 +3360,33 @@ mod tests { ); } + #[tokio::test] + async fn test_pending_cancellations_records_no_audit_for_a_never_accepted_agreement() { + // A cancel record on an agreement that was never accepted would later win + // over the chain's own, if the indexer accepted it after all and it was + // then ended: the terminated event would report an end before the accept. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let new_id = IndexingAgreementId::from_bytes(rand::random()); + let old_id = IndexingAgreementId::from_bytes(rand::random()); + + registry.add_agreement(new_id, IndexingAgreementStatus::AcceptedOnChain); + registry.add_agreement(old_id, IndexingAgreementStatus::Created); + registry.add_pending_cancellation(new_id, old_id); + + let result = execute_pending_cancellations( + &new_id, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok()); + assert!(registry.was_marked_canceled_by_requester(&old_id)); + assert!(!registry.was_cancel_audit_recorded(&old_id)); + } + #[tokio::test] async fn test_pending_cancellations_failed_cancel_records_no_audit() { let registry = MockRegistry::new(); diff --git a/dipper-pgregistry/tests/it_registry_postgres.rs b/dipper-pgregistry/tests/it_registry_postgres.rs index 37cc9b4b..cd6f3e7a 100644 --- a/dipper-pgregistry/tests/it_registry_postgres.rs +++ b/dipper-pgregistry/tests/it_registry_postgres.rs @@ -2931,6 +2931,86 @@ async fn apply_reconciliation_batch_handles_all_four_item_shapes() { ); } +#[tokio::test] +async fn recording_a_missed_accept_announces_the_agreement_once() { + // The chain listener records the accept and cancel of an agreement dipper had + // already cancelled locally, on every read of its cancelled snapshot. The first + // values must stick and, once announced, a later read must not announce again. + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let registry = PgRegistry::new(db); + let id = IndexingAgreementId::from_bytes([0xbb, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1]); + registry + .mark_indexing_agreement_as_canceled_by_requester(&id) + .await + .expect("dipper cancels it locally"); + + for (at, by) in [(1_700_000_002, "0xchain"), (1_800_000_000, "0xlater")] { + registry + .record_cancel_audit(&id, at, by, Some("0xcxltx")) + .await + .expect("cancel record"); + registry + .record_accepted_audit(&id, at - 1, "0xacc") + .await + .expect("accept record"); + } + + let accepted = registry + .get_agreements_pending_accepted_emission(100) + .await + .expect("accepted query"); + let accepted: Vec<_> = accepted.iter().filter(|p| p.agreement_id == id).collect(); + assert_eq!(accepted.len(), 1); + assert_eq!(accepted[0].accepted_at, 1_700_000_001); + let terminated = registry + .get_agreements_pending_terminated_emission(100) + .await + .expect("terminated query"); + let terminated: Vec<_> = terminated.iter().filter(|p| p.agreement_id == id).collect(); + assert_eq!(terminated.len(), 1); + assert_eq!(terminated[0].canceled_at, Some(1_700_000_002)); + assert_eq!(terminated[0].canceled_by.as_deref(), Some("0xchain")); + + registry + .mark_accepted_event_emitted(&id) + .await + .expect("mark accepted"); + registry + .mark_terminated_event_emitted(&id) + .await + .expect("mark terminated"); + registry + .record_cancel_audit(&id, 1_900_000_000, "0xagain", Some("0xcxltx")) + .await + .expect("cancel record"); + registry + .record_accepted_audit(&id, 1_899_999_999, "0xacc") + .await + .expect("accept record"); + assert!( + !registry + .get_agreements_pending_accepted_emission(100) + .await + .expect("accepted query") + .iter() + .any(|p| p.agreement_id == id) + ); + assert!( + !registry + .get_agreements_pending_terminated_emission(100) + .await + .expect("terminated query") + .iter() + .any(|p| p.agreement_id == id) + ); +} + #[tokio::test] async fn pending_emission_queries_cover_the_paired_accept_then_cancel() { // Regression guard: an agreement accepted and then cancelled in a single From 8441bf48bb7432a3883845a6386bfb0db203273e Mon Sep 17 00:00:00 2001 From: MoonBoi9001 Date: Fri, 2 Oct 2026 17:35:25 +0300 Subject: [PATCH 06/26] fix(worker): mark an unaccepted agreement cancelled before its cancel Offers and cancels share 1 wallet, so marking an unaccepted agreement first means its offer lands before the cancel and is withdrawn by it, or after, when the offer job sees the mark. This replaces a lock that paused offers during reassessment, and failed cancels are retried. --- bin/dipper-service/src/main.rs | 2 +- .../src/network/service/chain_listener.rs | 124 +++++++-- bin/dipper-service/src/worker.rs | 2 - bin/dipper-service/src/worker/context.rs | 9 +- .../cancel_rejected_agreement_on_chain.rs | 12 +- .../handlers/reassess_indexing_request.rs | 238 +++++++----------- .../src/worker/handlers/submit_offer.rs | 106 +------- .../src/worker/reassess_lock.rs | 156 ------------ 8 files changed, 210 insertions(+), 439 deletions(-) delete mode 100644 bin/dipper-service/src/worker/reassess_lock.rs diff --git a/bin/dipper-service/src/main.rs b/bin/dipper-service/src/main.rs index 2eda5367..8b22c649 100644 --- a/bin/dipper-service/src/main.rs +++ b/bin/dipper-service/src/main.rs @@ -431,7 +431,7 @@ pub async fn main() -> anyhow::Result<()> { } // A single global reassess lock, shared across all worker loops. - let reassess_lock = worker::ReassessLock::default(); + let reassess_lock = Arc::new(tokio::sync::Mutex::new(())); let unresponsive_breaker = Arc::new(worker::UnresponsiveBreaker::new()); let dips_accepting_cache = worker::DipsAcceptingCache::new(std::time::Duration::from_secs( agreement_conf.dips_accepting_cache_ttl_seconds(), diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index 5b06f855..f5ce990d 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -1116,8 +1116,8 @@ where /// /// Called from the Created -> AcceptedOnChain and Expired -> AcceptedOnChain /// transitions. For each pending cancellation, fires -/// `cancelIndexingAgreementByPayer` against the RecurringCollector contract, -/// then flips the dipper DB row to `CanceledByRequester`. Each pending row is +/// `cancelIndexingAgreementByPayer` against the RecurringCollector contract and +/// flips the dipper DB row to `CanceledByRequester`, an unaccepted one first. Each pending row is /// deleted individually after both steps succeed; transient failures retain /// the record so the next reconcile pass can retry. #[expect( @@ -1163,6 +1163,28 @@ where Some(a) => a, }; + // An unaccepted one is marked first. An offer for it still in flight then + // lands before this cancel, which withdraws it, or sees the mark and withdraws itself. + let unaccepted = old_agreement.status == IndexingAgreementStatus::Created; + if unaccepted { + match registry + .mark_indexing_agreement_as_canceled_by_requester(&cancellation.old_agreement_id) + .await + { + Ok(()) | Err(crate::registry::Error::NoRecordsUpdated) => {} + Err(err) => { + tracing::error!( + old_agreement_id = %cancellation.old_agreement_id, + error = %err, + "Failed to mark replaced agreement cancelled before its on-chain cancel, \ + retaining pending row" + ); + transient_failures += 1; + continue; + } + } + } + let mut on_chain_cancel_tx: Option = None; match crate::cancel_dispatch::cancel_agreement_on_chain( chain_client, @@ -1198,29 +1220,31 @@ where } } - match registry - .mark_indexing_agreement_as_canceled_by_requester(&cancellation.old_agreement_id) - .await - { - Ok(()) => {} - Err(crate::registry::Error::NoRecordsUpdated) => { - tracing::debug!( - old_agreement_id = %cancellation.old_agreement_id, - "Old agreement already in terminal state, skipping local cancel flip" - ); - registry - .delete_pending_cancellation(*agreement_id, cancellation.old_agreement_id) - .await?; - continue; - } - Err(err) => { - tracing::error!( - old_agreement_id = %cancellation.old_agreement_id, - error = %err, - "On-chain cancel succeeded but DB update failed, retaining pending row" - ); - transient_failures += 1; - continue; + if !unaccepted { + match registry + .mark_indexing_agreement_as_canceled_by_requester(&cancellation.old_agreement_id) + .await + { + Ok(()) => {} + Err(crate::registry::Error::NoRecordsUpdated) => { + tracing::debug!( + old_agreement_id = %cancellation.old_agreement_id, + "Old agreement already in terminal state, skipping local cancel flip" + ); + registry + .delete_pending_cancellation(*agreement_id, cancellation.old_agreement_id) + .await?; + continue; + } + Err(err) => { + tracing::error!( + old_agreement_id = %cancellation.old_agreement_id, + error = %err, + "On-chain cancel succeeded but DB update failed, retaining pending row" + ); + transient_failures += 1; + continue; + } } } @@ -2567,6 +2591,9 @@ mod tests { struct MockChainClient { cancels: Arc>>, already_canceled: Arc>>, + /// When set, each cancel records whether its agreement was already marked. + registry: Option, + marked_at_cancel: Arc>>, } impl MockChainClient { @@ -2614,6 +2641,11 @@ mod tests { // Record manager-routed cancels through the same recorder so the // existing assertions hold. self.cancels.lock().unwrap().push(*agreement_id); + if let Some(registry) = &self.registry { + let id = IndexingAgreementId::from_bytes(*agreement_id); + let was_marked = registry.was_marked_canceled_by_requester(&id); + self.marked_at_cancel.lock().unwrap().push(was_marked); + } // A manager-routed cancel has no "already canceled" result: the // contract silently no-ops a stale cancel and the tx still succeeds, // so the real cancel_via_manager never returns Ok(None). @@ -3387,6 +3419,48 @@ mod tests { assert!(!registry.was_cancel_audit_recorded(&old_id)); } + /// Runs the pending cancellation of one old agreement in `status`, returning + /// whether it was already marked cancelled when its on-chain cancel went out. + async fn marked_at_pending_cancel(status: IndexingAgreementStatus) -> Vec { + let registry = MockRegistry::new(); + let chain_client = MockChainClient { + registry: Some(registry.clone()), + ..MockChainClient::default() + }; + let new_id = IndexingAgreementId::from_bytes(rand::random()); + let old_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(new_id, IndexingAgreementStatus::AcceptedOnChain); + registry.add_agreement(old_id, status); + registry.add_pending_cancellation(new_id, old_id); + + let result = execute_pending_cancellations( + &new_id, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok(), "got {result:?}"); + assert!(registry.was_marked_canceled_by_requester(&old_id)); + assert!(registry.was_pending_cancellation_deleted(&new_id, &old_id)); + chain_client.marked_at_cancel.lock().unwrap().clone() + } + + #[tokio::test] + async fn test_pending_cancellations_mark_an_unaccepted_agreement_before_its_cancel() { + // Its offer may be in flight. An offer landing after the cancel can't be + // withdrawn by it, so its job must find the agreement already marked. + let marked = marked_at_pending_cancel(IndexingAgreementStatus::Created).await; + assert_eq!(marked, vec![true]); + } + + #[tokio::test] + async fn test_pending_cancellations_mark_an_accepted_agreement_after_its_cancel() { + let marked = marked_at_pending_cancel(IndexingAgreementStatus::AcceptedOnChain).await; + assert_eq!(marked, vec![false]); + } + #[tokio::test] async fn test_pending_cancellations_failed_cancel_records_no_audit() { let registry = MockRegistry::new(); diff --git a/bin/dipper-service/src/worker.rs b/bin/dipper-service/src/worker.rs index 453895dc..9ef02e12 100644 --- a/bin/dipper-service/src/worker.rs +++ b/bin/dipper-service/src/worker.rs @@ -2,12 +2,10 @@ mod context; mod handlers; mod messages; pub mod queue; -mod reassess_lock; mod result; pub mod service; mod service_queue; mod unresponsive_breaker; pub use context::Ctx; -pub use reassess_lock::ReassessLock; pub use unresponsive_breaker::{DipsAcceptingCache, UnresponsiveBreaker}; diff --git a/bin/dipper-service/src/worker/context.rs b/bin/dipper-service/src/worker/context.rs index d24cb65e..5e1b7e60 100644 --- a/bin/dipper-service/src/worker/context.rs +++ b/bin/dipper-service/src/worker/context.rs @@ -4,9 +4,8 @@ use dipper_core::state::FromState; use dipper_producer::events::SubgraphIndexingAgreementEventsProducer; use graph_networks_registry::NetworksRegistry; use thegraph_core::alloy::primitives::ChainId; -use tokio::sync::Notify; +use tokio::sync::{Mutex, Notify}; -pub use super::reassess_lock::ReassessLock; use super::{ handlers::{ CancelRejectedAgreementOnChainCtx, ReassessIndexingRequestCtx, @@ -20,6 +19,11 @@ use crate::{ signing::eip712::Eip712Signer, }; +/// A single process-wide async mutex. Only one reassessment runs at a time +/// across every worker loop, so two loops can't diff the same baseline and both +/// create agreements. In-process only (dipper is single-replica). +pub type ReassessLock = Arc>; + /// Generates a `FromState>` impl mapping InnerCtx fields onto a /// handler context type. Syntax: `impl_from_state!(Target { mappings })`, /// where a mapping is `field` (same name) or `target: source` (renamed). @@ -237,5 +241,4 @@ impl_from_state!(SubmitOfferCtx { registry, chain_client, agreement_conf, - reassess_lock, }); diff --git a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs index 8905c54e..c135defb 100644 --- a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs +++ b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs @@ -1,6 +1,6 @@ //! Cancel on-chain, via the RecurringAgreementManager, an agreement dipper doesn't -//! want that was accepted anyway: one the indexer rejected off-chain, or one dipper -//! had already cancelled. The chain listener queues it. +//! want that is still live: one the indexer rejected off-chain, or one dipper had already +//! cancelled. The chain listener queues it, and a reassessment does when its own cancel fails. use std::{ collections::HashSet, @@ -60,7 +60,7 @@ where } }; - // The chain listener queues this job only for these two statuses. + // This job is only queued for these 2 statuses. match agreement.status { IndexingAgreementStatus::Rejected => {} IndexingAgreementStatus::CanceledByRequester => { @@ -173,9 +173,9 @@ impl Drop for Cancelling { } } -/// Cancel on-chain an agreement dipper had already cancelled that the indexer -/// accepted anyway; the row is already terminal. The chain is read first, so a -/// stale snapshot of one dipper has since ended raises no alert. +/// Cancel on-chain an agreement dipper had already cancelled that is still live: accepted +/// anyway, or an offer a failed cancel left open. The row is already terminal. The chain is +/// read first, so a stale snapshot of one dipper has since ended raises no alert. async fn cancel_live_agreement_dipper_cancelled( ctx: &Ctx, agreement: &IndexingAgreement, diff --git a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs index e48850d7..0fb5bfa7 100644 --- a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs +++ b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs @@ -69,9 +69,8 @@ pub struct Ctx { /// is true. `None` disables the bypass path even if the flag is set /// (handler falls back to wall clock with a warning). pub chain_listener_chain_id: Option, - /// Global reassess lock: only one reassessment runs at a time across all - /// worker loops, and its cancels of unaccepted agreements wait for offers in - /// flight (see `crate::worker::context::ReassessLock`). + /// Global reassess lock; only one reassessment runs at a time across all + /// worker loops (see `crate::worker::context::ReassessLock`). pub reassess_lock: crate::worker::context::ReassessLock, /// Mass-unresponsive circuit breaker (see its module). pub unresponsive_breaker: Arc, @@ -129,8 +128,9 @@ where // Only one reassessment runs globally at a time; if another loop holds the // lock this pass would diff the same baseline, so defer ~1s rather than park // this loop. Deferral isn't a failure: no backoff, no attempt count. - let Some(reassessment) = ctx.reassess_lock.start_reassessment() else { - return Err(JobError::Deferred(Duration::from_secs(1))); + let _reassess_guard = match ctx.reassess_lock.try_lock() { + Ok(guard) => guard, + Err(_) => return Err(JobError::Deferred(Duration::from_secs(1))), }; // Get current active agreements for this indexing request. Fetched once here @@ -633,49 +633,30 @@ where ); } - // Cancel old agreements that have no replacement to pair with. - // These indexers are leaving the target group with nothing taking - // their place. Fire the on-chain cancel first; only mark the local row - // CanceledByRequester after the chain tx is accepted, so retry on a - // transient chain-client failure does the right thing. + // Cancel old agreements that have no replacement to pair with: these indexers + // leave the target group with nothing taking their place. An accepted one is + // cancelled on-chain before it is marked, so a failed cancel leaves it for a retry. let mut directly_cancelled = 0u32; let mut cancel_failures = 0u32; - // A cancel of an agreement whose offer may be in flight waits for that offer - // to land, so the cancel follows it on-chain and withdraws it. Only these - // cancels hold new offers back; the rest of the pass doesn't. - let unpaired: Vec<_> = old_iter.collect(); - let offers_held_back = if unpaired.iter().any(|a| offer_may_be_in_flight(a)) { - reassessment.hold_back_offers().await - } else { - None - }; - for old_agreement in unpaired { - // A Created agreement's offer may already be on-chain, where the indexer - // can still accept it, so it is cancelled on-chain too: the cancel revokes - // a pending offer and does nothing when none was made. Rows without a - // stored terms hash can't be cancelled on-chain and are only marked locally. + for old_agreement in old_iter { let was_accepted = matches!( old_agreement.status, crate::registry::IndexingAgreementStatus::AcceptedOnChain ); + // A Created agreement's offer may already be on-chain, so it is cancelled + // on-chain too; without a stored terms hash it can only be marked locally. let needs_on_chain_cancel = was_accepted || (old_agreement.status == crate::registry::IndexingAgreementStatus::Created && old_agreement.terms_version_hash.is_some()); - // An unaccepted agreement is marked cancelled even when its on-chain cancel - // can't go out or fails: its offer job then withdraws any offer instead of - // sending one, and the chain listener cancels it if the indexer accepts. - let offers_still_landing = - offer_may_be_in_flight(old_agreement) && offers_held_back.is_none(); - if offers_still_landing { - tracing::warn!( - agreement_id = %old_agreement.id, - "Offers in flight did not land; marking the agreement cancelled without an on-chain cancel" - ); + // An unaccepted one is marked first. An offer in flight then lands before the + // cancel, which withdraws it, or after, when its job sees the mark and withdraws it. + if !was_accepted && !mark_unpaired_cancelled(&ctx.registry, old_agreement).await { cancel_failures += 1; + continue; } let mut on_chain_cancel_tx: Option = None; - if needs_on_chain_cancel && !offers_still_landing { + if needs_on_chain_cancel { match crate::cancel_dispatch::cancel_agreement_on_chain( &ctx.chain_client, old_agreement, @@ -705,26 +686,32 @@ where agreement_id = %old_agreement.id, was_accepted, "On-chain cancel failed; an accepted agreement is retried later, an \ - unaccepted one is marked cancelled anyway" + unaccepted one by a queued cancel job" ); cancel_failures += 1; if was_accepted { continue; } + if let Err(err) = ctx + .queue + .cancel_rejected_agreement_on_chain( + old_agreement.id, + JobPriority::Background, + ) + .await + { + tracing::error!( + error = %err, + agreement_id = %old_agreement.id, + "Failed to queue a retry of the on-chain cancel; an offer already \ + on-chain stays open" + ); + } } } } - if let Err(err) = ctx - .registry - .mark_indexing_agreement_as_canceled_by_requester(&old_agreement.id) - .await - { - tracing::error!( - error=%err, - agreement_id=%old_agreement.id, - "Failed to mark unpaired old agreement as canceled in local DB" - ); + if was_accepted && !mark_unpaired_cancelled(&ctx.registry, old_agreement).await { cancel_failures += 1; continue; } @@ -775,22 +762,16 @@ where } if cancel_failures > 0 { - // An unaccepted agreement whose cancel failed was still marked cancelled - // (see above). An accepted one keeps its status, and two recovery paths - // cover it: - // - // - Shrink-to-zero (request now Canceled): the chain_listener's - // `sweep_orphan_canceled_agreements` retries AcceptedOnChain rows on - // every sweep tick (default ~5 min at fast poll, ~5 h at slow poll). - // - Shrink-not-zero (request still Open with too many agreements): - // the periodic reassignment service re-queues reassessment at its - // configured cadence (default 24 h). + // An unaccepted agreement's failed cancel was queued as its own job above. An + // accepted one stays AcceptedOnChain: the orphan-cancel sweep retries it once the + // request is Canceled, the periodic reassignment service (24 h) while it is Open. tracing::warn!( indexing_request_id=%indexing_request_id, failures=cancel_failures, "some agreement cancels failed during reassessment; retry will fire \ - via the orphan-cancel sweep (Canceled requests) or the periodic \ - reassignment service (Open requests over-target)" + via a queued cancel job (unaccepted agreements), the orphan-cancel \ + sweep (Canceled requests) or the periodic reassignment service \ + (Open requests over-target)" ); } @@ -1018,11 +999,12 @@ mod lifecycle_event_tests { // ---- Mock: worker queue -------------------------------------------------- - /// Records every `send_indexing_agreement_proposal` call's indexer URL. - /// Clone shares the buffer so a caller can inspect proposals after `handle`. + /// Records every `send_indexing_agreement_proposal` call's indexer URL and every + /// queued on-chain cancel. Clone shares the buffers for inspection after `handle`. #[derive(Default, Clone)] struct MockQueue { proposals: Arc>>, + cancels_queued: Arc>>, } #[async_trait] @@ -1053,10 +1035,11 @@ mod lifecycle_event_tests { async fn cancel_rejected_agreement_on_chain( &self, - _agreement_id: IndexingAgreementId, + agreement_id: IndexingAgreementId, _priority: crate::worker::queue::JobPriority, ) -> anyhow::Result { - unimplemented!("not exercised by reassess handler") + self.cancels_queued.lock().unwrap().push(agreement_id); + Ok(crate::worker::queue::JobId::default()) } async fn submit_offer( @@ -1079,9 +1062,9 @@ mod lifecycle_event_tests { #[derive(Default, Clone)] struct MockChainClient { cancelled: Arc>>, - /// When set, each cancel records whether new offers were held back. - reassess_lock: Option, - offers_held_back_at_cancel: Arc>>, + /// When set, each cancel records whether its agreement was already marked. + marked_cancelled: Option>>>, + marked_at_cancel: Arc>>, fail_cancel: bool, } @@ -1111,11 +1094,10 @@ mod lifecycle_event_tests { return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); } self.cancelled.lock().unwrap().push(*agreement_id); - if let Some(lock) = &self.reassess_lock { - self.offers_held_back_at_cancel - .lock() - .unwrap() - .push(lock.offer().is_none()); + if let Some(marked) = &self.marked_cancelled { + let id = IndexingAgreementId::from_bytes(*agreement_id); + let was_marked = marked.lock().unwrap().contains(&id); + self.marked_at_cancel.lock().unwrap().push(was_marked); } Ok(Some(B256::repeat_byte(0xcd))) } @@ -1537,7 +1519,7 @@ mod lifecycle_event_tests { chain_listener_notify: Arc::new(tokio::sync::Notify::new()), bypass_chain_clock_defenses: false, chain_listener_chain_id: None, - reassess_lock: crate::worker::ReassessLock::default(), + reassess_lock: Arc::new(tokio::sync::Mutex::new(())), unresponsive_breaker: Arc::new(crate::worker::UnresponsiveBreaker::new()), dips_accepting_cache: crate::worker::DipsAcceptingCache::new( std::time::Duration::from_secs(300), @@ -1968,36 +1950,8 @@ mod lifecycle_event_tests { assert!(matches!(captured[0], CapturedEvent::Proposed { .. })); } - #[tokio::test(start_paused = true)] - async fn a_saturated_pass_defers_without_holding_up_offers() { - // A pass that will only defer must not first wait out offers in flight, - // which would block new offers for nothing. - let mut ctx = build_ctx( - MockRegistry::default(), - MockIisa { selected: vec![] }, - MockQueue::default(), - MockChainClient::default(), - CapturingEventsProducer::new(), - indexer_urls::Snapshot::new(), - ); - ctx.agreement_conf = Arc::new(IndexingAgreementConfig { - max_in_flight_offers_total: Some(0), - ..test_agreement_conf() - }); - let _offer = ctx.reassess_lock.offer().expect("lock is free"); - - let started = tokio::time::Instant::now(); - let result = handle(ctx, &test_message(1)).await; - - assert!( - matches!(result, Err(crate::worker::result::JobError::Deferred(_))), - "got {result:?}" - ); - assert_eq!(started.elapsed(), std::time::Duration::ZERO); - } - - /// Ctx whose only active agreement, in `status`, leaves the target group, - /// with the chain mock recording whether offers were held back at each cancel. + /// Ctx whose only active agreement, in `status`, leaves the target group, with + /// the chain mock recording whether the row was already marked at each cancel. fn ctx_cancelling_one( status: IndexingAgreementStatus, ) -> ( @@ -2016,14 +1970,14 @@ mod lifecycle_event_tests { CapturingEventsProducer::new(), indexer_urls::Snapshot::new(), ); - ctx.chain_client.reassess_lock = Some(ctx.reassess_lock.clone()); + ctx.chain_client.marked_cancelled = Some(ctx.registry.marked_cancelled.clone()); (ctx, leaving) } #[tokio::test] - async fn cancelling_an_unaccepted_agreement_holds_back_new_offers() { - // Its offer may be in flight: a cancel sent ahead of it would find nothing - // to withdraw, and the offer would land after it and stay open. + async fn marks_an_unaccepted_agreement_cancelled_before_cancelling_it_on_chain() { + // Its offer may be in flight. An offer landing after the cancel can't be + // withdrawn by it, so its job must find the agreement already marked. let (ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::Created); let chain = ctx.chain_client.clone(); @@ -2034,66 +1988,36 @@ mod lifecycle_event_tests { *chain.cancelled.lock().unwrap(), vec![*leaving.id.as_bytes()] ); - assert_eq!( - *chain.offers_held_back_at_cancel.lock().unwrap(), - vec![true] - ); + assert_eq!(*chain.marked_at_cancel.lock().unwrap(), vec![true]); } - #[tokio::test(start_paused = true)] - async fn cancelling_only_accepted_agreements_leaves_offers_running() { - // An accepted agreement has no offer in flight, so nothing to wait for. + #[tokio::test] + async fn cancels_an_accepted_agreement_on_chain_before_marking_it() { let (ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::AcceptedOnChain); let chain = ctx.chain_client.clone(); - let _offer = ctx.reassess_lock.offer().expect("lock is free"); - - let started = tokio::time::Instant::now(); - let result = handle(ctx, &test_message(0)).await; - - assert!(result.is_ok(), "got {result:?}"); - assert_eq!(started.elapsed(), std::time::Duration::ZERO); - assert_eq!( - *chain.cancelled.lock().unwrap(), - vec![*leaving.id.as_bytes()] - ); - assert_eq!( - *chain.offers_held_back_at_cancel.lock().unwrap(), - vec![false] - ); - } - - #[tokio::test(start_paused = true)] - async fn skips_cancelling_an_unaccepted_agreement_while_an_offer_will_not_land() { - // After 30 s it gives up on that cancel rather than send it ahead of the - // offer, leaving it for the next reassessment. - let (ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::Created); - let chain = ctx.chain_client.clone(); let marked = ctx.registry.marked_cancelled.clone(); - let _stuck_offer = ctx.reassess_lock.offer().expect("lock is free"); - let started = tokio::time::Instant::now(); let result = handle(ctx, &test_message(0)).await; assert!(result.is_ok(), "got {result:?}"); - assert_eq!(started.elapsed(), std::time::Duration::from_secs(30)); - assert!(chain.cancelled.lock().unwrap().is_empty()); - // Marked cancelled anyway: the offer job withdraws its offer once it lands. + assert_eq!(*chain.marked_at_cancel.lock().unwrap(), vec![false]); assert_eq!(*marked.lock().unwrap(), vec![leaving.id]); } #[tokio::test] - async fn an_unaccepted_agreement_whose_cancel_fails_is_marked_cancelled_anyway() { - // Left unaccepted, its offer job would still send the offer. Marked - // cancelled, the job withdraws any offer instead, and the chain listener - // cancels the agreement if the indexer accepts one. + async fn an_unaccepted_agreement_whose_cancel_fails_is_marked_and_its_cancel_queued() { + // Nothing else would send its cancel again: the row is already terminal, + // and its offer job may have finished before the mark. let (mut ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::Created); ctx.chain_client.fail_cancel = true; let marked = ctx.registry.marked_cancelled.clone(); + let queue = ctx.queue.clone(); let result = handle(ctx, &test_message(0)).await; assert!(result.is_ok(), "got {result:?}"); assert_eq!(*marked.lock().unwrap(), vec![leaving.id]); + assert_eq!(*queue.cancels_queued.lock().unwrap(), vec![leaving.id]); } #[tokio::test] @@ -2102,11 +2026,13 @@ mod lifecycle_event_tests { let (mut ctx, _leaving) = ctx_cancelling_one(IndexingAgreementStatus::AcceptedOnChain); ctx.chain_client.fail_cancel = true; let marked = ctx.registry.marked_cancelled.clone(); + let queue = ctx.queue.clone(); let result = handle(ctx, &test_message(0)).await; assert!(result.is_ok(), "got {result:?}"); assert!(marked.lock().unwrap().is_empty()); + assert!(queue.cancels_queued.lock().unwrap().is_empty()); } #[tokio::test] @@ -2139,11 +2065,25 @@ mod lifecycle_event_tests { } } -/// Whether an offer for this agreement could still be on its way on-chain: it -/// isn't accepted yet and has the terms hash an on-chain cancel needs. -fn offer_may_be_in_flight(agreement: &crate::registry::IndexingAgreement) -> bool { - agreement.status == crate::registry::IndexingAgreementStatus::Created - && agreement.terms_version_hash.is_some() +/// Mark an unpaired old agreement CanceledByRequester; false, logged, if that fails. +async fn mark_unpaired_cancelled( + registry: &R, + agreement: &crate::registry::IndexingAgreement, +) -> bool { + match registry + .mark_indexing_agreement_as_canceled_by_requester(&agreement.id) + .await + { + Ok(()) => true, + Err(err) => { + tracing::error!( + error = %err, + agreement_id = %agreement.id, + "Failed to mark unpaired old agreement as canceled in local DB" + ); + false + } + } } /// Olds reserved from cancellation: one per add-cancel pairing lost to a diff --git a/bin/dipper-service/src/worker/handlers/submit_offer.rs b/bin/dipper-service/src/worker/handlers/submit_offer.rs index b3b39d46..172ddcdd 100644 --- a/bin/dipper-service/src/worker/handlers/submit_offer.rs +++ b/bin/dipper-service/src/worker/handlers/submit_offer.rs @@ -28,18 +28,12 @@ use crate::{ config::IndexingAgreementConfig, indexer_rpc_client::into_sol_rca, registry::{AgreementRegistry, IndexingAgreement, IndexingAgreementStatus}, - worker::{ - context::ReassessLock, - result::{JobError, JobResult}, - }, + worker::result::{JobError, JobResult}, }; /// Backoff base for a tx the RPC accepted and then dropped from the mempool. pub const DROPPED_TX_RETRY_BASE: Duration = Duration::from_secs(5); -/// Retry shortly while a reassessment runs or waits; not counted as a failure. -const DEFER_WHILE_REASSESSING: JobError = JobError::Deferred(Duration::from_secs(1)); - /// Backoff base for a transient submission failure: RPC, gas or nonce. pub const TRANSIENT_RETRY_BASE: Duration = Duration::from_secs(30); @@ -47,8 +41,6 @@ pub struct Ctx { pub registry: R, pub chain_client: T, pub agreement_conf: Arc, - /// Taken shared while the offer is checked and sent (see `ReassessLock`). - pub reassess_lock: ReassessLock, } /// Submit an RCA offer on-chain. @@ -79,15 +71,9 @@ where R: AgreementRegistry, T: ChainClient, { - // Held from the status check until the offer lands: a reassessment's cancel - // then either follows this offer (later nonce, same wallet) and withdraws it, - // or finished first and the check below sees the agreement cancelled. - let reassess_guard = ctx.reassess_lock.offer().ok_or(DEFER_WHILE_REASSESSING)?; - let agreement = match next_step(&ctx.registry, agreement_id).await? { NextStep::Offer(agreement) => agreement, NextStep::Withdraw(agreement) => { - drop(reassess_guard); return withdraw_offer_if_stored(&ctx, &agreement).await; } NextStep::Skip => return Ok(()), @@ -188,8 +174,8 @@ where } } - // Landed, so any later cancel follows it; stop holding up reassessments. - drop(reassess_guard); + // Every cancel of an unaccepted agreement marks it before it is sent, so a cancel + // that went out ahead of this offer, and found nothing to withdraw, shows here. if let Some(agreement) = cancelled_meanwhile(&ctx.registry, agreement_id).await { return withdraw_offer_if_stored(&ctx, &agreement).await; } @@ -242,8 +228,7 @@ async fn next_step( } /// Withdraw the agreement's offer if one is on-chain: dipper cancelled it while -/// this job's offer was in flight (the chain listener cancels replaced agreements -/// without the reassess lock) or before this retry. A failure retries the job, +/// this job's offer was in flight or before this retry. A failure retries the job, /// which comes back here through its status check. async fn withdraw_offer_if_stored( ctx: &Ctx, @@ -357,12 +342,9 @@ mod tests { /// Yields the configured result once; a second call means the handler /// retried inside one run, and a call with no result configured means the - /// handler sent when it must not. Records whether a reassessment could have - /// taken `reassess_lock` while the offer was being sent. + /// handler sent when it must not. struct MockChainClient { offer_result: Mutex, ChainClientError>>>, - reassess_lock: ReassessLock, - reassessment_could_start_mid_send: Arc>>, /// When set, the agreement is cancelled locally while the offer is sent, /// as the chain listener does when a replacement is accepted. cancel_mid_send: Option, @@ -379,8 +361,6 @@ mod tests { &self, _rca: &dipper_rpc::indexer::indexer_client::sol::RecurringCollectionAgreement, ) -> Result, ChainClientError> { - *self.reassessment_could_start_mid_send.lock().unwrap() = - Some(self.reassess_lock.reassessment_could_start_now()); if let Some(agreement) = &self.cancel_mid_send && let Some(row) = agreement.lock().unwrap().as_mut() { @@ -498,14 +478,13 @@ mod tests { agreement: IndexingAgreement, offer_result: Result, ChainClientError>, ) -> Ctx { - ctx_with_lock(agreement, Some(offer_result), ReassessLock::default()) + ctx_with(agreement, Some(offer_result)) } /// `offer_result: None` makes any send panic. - fn ctx_with_lock( + fn ctx_with( agreement: IndexingAgreement, offer_result: Option, ChainClientError>>, - reassess_lock: ReassessLock, ) -> Ctx { Ctx { registry: MockRegistry { @@ -513,15 +492,12 @@ mod tests { }, chain_client: MockChainClient { offer_result: Mutex::new(offer_result), - reassess_lock: reassess_lock.clone(), - reassessment_could_start_mid_send: Arc::default(), cancel_mid_send: None, cancelled: Arc::default(), on_chain: Arc::default(), fail_cancel: false, }, agreement_conf: Arc::new(test_agreement_conf()), - reassess_lock, } } @@ -587,70 +563,6 @@ mod tests { assert!(cancelled.lock().unwrap().is_empty()); } - #[tokio::test] - async fn waits_without_sending_while_a_reassessment_holds_the_lock() { - //* Arrange - a reassessment holds the lock; no offer result is configured, - // so a send would panic - let agreement = make_test_agreement(); - let message = make_message(agreement.id); - let lock = ReassessLock::default(); - let _reassessment = lock.reassessment().await.expect("lock is free"); - let ctx = ctx_with_lock(agreement, None, lock); - - //* Act - let result = handle(ctx, &message).await; - - //* Assert - deferred, not failed, so it runs once the reassessment ends - assert!( - matches!(result, Err(JobError::Deferred(delay)) if delay == Duration::from_secs(1)), - "an offer must wait while a reassessment runs, got {result:?}" - ); - } - - #[tokio::test] - async fn no_reassessment_can_start_while_the_offer_is_sent() { - //* Arrange - let agreement = make_test_agreement(); - let message = make_message(agreement.id); - let ctx = ctx_with_offer_result(agreement, Ok(Some(B256::repeat_byte(0xab)))); - let lock = ctx.reassess_lock.clone(); - let could_start = ctx.chain_client.reassessment_could_start_mid_send.clone(); - - //* Act - let result = handle(ctx, &message).await; - - //* Assert - the lock was held during the send and is free once the job ends - assert!(result.is_ok(), "got {result:?}"); - assert_eq!( - *could_start.lock().unwrap(), - Some(false), - "a reassessment must not be able to start while the offer is sent" - ); - assert!( - lock.reassessment_could_start_now(), - "the job must release the lock when it ends" - ); - } - - #[tokio::test] - async fn sends_alongside_another_offer() { - //* Arrange - another offer job holds the lock shared - let agreement = make_test_agreement(); - let message = make_message(agreement.id); - let lock = ReassessLock::default(); - let _other_offer = lock.offer().expect("lock is free"); - let ctx = ctx_with_lock(agreement, Some(Ok(Some(B256::repeat_byte(0xab)))), lock); - - //* Act - let result = handle(ctx, &message).await; - - //* Assert - assert!( - result.is_ok(), - "offers must not wait for each other, got {result:?}" - ); - } - #[tokio::test] async fn skips_an_agreement_a_reassessment_already_cancelled() { //* Arrange - the reassessment ran first and cancelled the agreement; no offer @@ -658,7 +570,7 @@ mod tests { let mut agreement = make_test_agreement(); agreement.status = IndexingAgreementStatus::CanceledByRequester; let message = make_message(agreement.id); - let ctx = ctx_with_lock(agreement, None, ReassessLock::default()); + let ctx = ctx_with(agreement, None); let cancelled = ctx.chain_client.cancelled.clone(); //* Act @@ -678,7 +590,7 @@ mod tests { agreement.terms_version_hash = Some(vec![7u8; 32]); let agreement_id = agreement.id; let message = make_message(agreement_id); - let ctx = ctx_with_lock(agreement, None, ReassessLock::default()); + let ctx = ctx_with(agreement, None); ctx.chain_client.on_chain.store(true, Ordering::SeqCst); let cancelled = ctx.chain_client.cancelled.clone(); diff --git a/bin/dipper-service/src/worker/reassess_lock.rs b/bin/dipper-service/src/worker/reassess_lock.rs deleted file mode 100644 index 3fddb696..00000000 --- a/bin/dipper-service/src/worker/reassess_lock.rs +++ /dev/null @@ -1,156 +0,0 @@ -//! Keeps a reassessment's cancels and offers apart: an offer sent just after such -//! a cancel would land after it and stay open. Offers hold this shared from their -//! status check until they land. In-process only (dipper is single-replica). - -use std::{sync::Arc, time::Duration}; - -use tokio::sync::{Mutex, OwnedMutexGuard, OwnedRwLockReadGuard, OwnedRwLockWriteGuard, RwLock}; - -/// Longest a reassessment waits for offers already being sent to land before it -/// gives up on its cancels. An offer holds the lock through its receipt wait -/// (15 s) and its turn at the chain client's send, so a slow RPC can outlast this. -const OFFER_DRAIN_WAIT: Duration = Duration::from_secs(30); - -#[derive(Clone, Default)] -pub struct ReassessLock { - /// Only one reassessment runs at a time across every worker loop, so two - /// loops can't diff the same baseline and both create agreements. - reassessment: Arc>, - /// Offers hold it shared; a reassessment holds it exclusively while it - /// cancels agreements whose offers may be in flight. - offers: Arc>, -} - -/// The running reassessment. -pub struct Reassessment { - _running: OwnedMutexGuard<()>, - offers: Arc>, -} - -impl ReassessLock { - /// Start a reassessment, or `None` (defer) if another one is running. - pub fn start_reassessment(&self) -> Option { - Some(Reassessment { - _running: self.reassessment.clone().try_lock_owned().ok()?, - offers: self.offers.clone(), - }) - } - - /// A reassessment holding offers back, as one does while it cancels. - #[cfg(test)] - pub async fn reassessment(&self) -> Option<(Reassessment, OwnedRwLockWriteGuard<()>)> { - let reassessment = self.start_reassessment()?; - let offers_held_back = reassessment.hold_back_offers().await?; - Some((reassessment, offers_held_back)) - } - - /// Start an offer submission, or `None` (defer) while a reassessment is - /// cancelling or waiting to. Offers never wait for each other. - pub fn offer(&self) -> Option> { - self.offers.clone().try_read_owned().ok() - } - - /// Whether a reassessment could start and take the lock right now. - #[cfg(test)] - pub fn reassessment_could_start_now(&self) -> bool { - self.reassessment.try_lock().is_ok() && self.offers.try_write().is_ok() - } -} - -impl Reassessment { - /// Wait for offers in flight to land and keep new ones from starting until - /// the guard drops, or `None` after `OFFER_DRAIN_WAIT`. Waiting already holds - /// new offers back, so a stream of them can't keep it out. - pub async fn hold_back_offers(&self) -> Option> { - let held = tokio::time::timeout(OFFER_DRAIN_WAIT, self.offers.clone().write_owned()).await; - if held.is_err() { - tracing::warn!( - wait_secs = OFFER_DRAIN_WAIT.as_secs(), - "Offers in flight did not land in time; skipping cancels that could race them" - ); - } - held.ok() - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[tokio::test] - async fn offers_run_alongside_each_other() { - let lock = ReassessLock::default(); - let _first = lock.offer().expect("first offer starts"); - assert!(lock.offer().is_some(), "a second offer must not wait"); - } - - #[tokio::test] - async fn offers_run_while_a_reassessment_is_not_cancelling() { - let lock = ReassessLock::default(); - let _running = lock.start_reassessment().expect("lock is free"); - assert!(lock.offer().is_some()); - } - - #[tokio::test] - async fn no_offer_starts_while_a_reassessment_holds_them_back() { - let lock = ReassessLock::default(); - let _reassessment = lock.reassessment().await.expect("lock is free"); - assert!(lock.offer().is_none()); - } - - #[tokio::test(start_paused = true)] - async fn a_second_reassessment_defers_without_waiting() { - let lock = ReassessLock::default(); - let _running = lock.reassessment().await.expect("lock is free"); - - let started = tokio::time::Instant::now(); - assert!(lock.reassessment().await.is_none()); - assert_eq!( - started.elapsed(), - Duration::ZERO, - "it must not park its loop" - ); - } - - #[tokio::test] - async fn a_reassessment_waits_for_an_offer_in_flight_and_blocks_new_ones() { - let lock = ReassessLock::default(); - let in_flight = lock.offer().expect("offer starts"); - - let waiting = tokio::spawn({ - let lock = lock.clone(); - async move { lock.reassessment().await.is_some() } - }); - // Let the reassessment queue behind the offer in flight. - while lock.reassessment.try_lock().is_ok() { - tokio::task::yield_now().await; - } - tokio::task::yield_now().await; - - assert!( - lock.offer().is_none(), - "a new offer must not start ahead of the waiting reassessment" - ); - drop(in_flight); - assert!( - waiting.await.unwrap(), - "the reassessment runs once the offer lands" - ); - } - - #[tokio::test(start_paused = true)] - async fn holding_back_offers_gives_up_when_they_do_not_land_in_time() { - let lock = ReassessLock::default(); - let _stuck = lock.offer().expect("offer starts"); - - let started = tokio::time::Instant::now(); - assert!(lock.reassessment().await.is_none()); - assert_eq!(started.elapsed(), OFFER_DRAIN_WAIT); - - drop(_stuck); - assert!( - lock.reassessment_could_start_now(), - "giving up must release both locks" - ); - } -} From 884e33d846da5803df5dc4500d33163c858a96b5 Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Fri, 2 Oct 2026 20:57:23 +0300 Subject: [PATCH 07/26] feat: keep retrying cancels until the chain confirms them (#716) * feat(agreements): keep retrying cancels until the chain confirms them Offers and cancels share 1 wallet, so marking an agreement Cancelling before its cancel goes out lets a late offer withdraw itself. The chain listener retries the cancel until the chain shows the end, with an ERROR after 10 failed attempts; only then is the end announced. * refactor(worker): share the read-then-cancel step between cancel paths The offer job's withdraw and the on-chain cancel job each read the chain and cancelled only a live agreement, with their own copy of that logic; both now use the helper the retry uses. * fix(worker): retry an offer job that can't recheck its agreement After its offer lands, the offer job rereads the agreement to withdraw the offer if it was cancelled meanwhile. A failed reread finished the job, leaving such an offer open; it retries. * fix(listener): announce a missed accept however late it is read Accepts dipper missed were announced only if it cancelled within 1 hour of the offer deadline, so a lagging listener dropped them. Only agreements created before events existed are skipped. * test(config): build every test's agreement config from one helper 6 test modules each spelled out the whole agreement config, so every new setting meant editing 6 copies. They now start from a shared test helper and override what they need. * docs(worker): write job counts as numerals in the cancel job's comments * fix(listener): record no accept for a withdrawn offer being cancelled The subgraph reports a withdrawn offer as cancelled by the payer with an accept time of 0. Recording that as an accept announced an agreement that was never live. * fix(listener): let the listener confirm an accepted agreement that ended The cancel retry reads the chain ahead of the listener, so marking an ended agreement there lost who ended it and when; it now waits for the listener, which records both from the chain. * fix(listener): retry cancels fairly, briefly and never twice at once Each sweep took the 100 oldest cancelling agreements, so a backlog stalled the listener and hid newer ones, and could resend a cancel still being mined. It now takes 10, least recently checked first, and skips any marked in the last 2 minutes. * fix(listener): count only cancels that are mined without effect A misconfigured or paused manager failed every agreement's cancel, using up their attempts so none was retried once fixed. Only a cancel mined without ending the agreement now counts, and an agreement with no stored terms hash, which can never be cancelled, is given up at once. * fix(worker): finish an offer job whose recheck fails without resending Retrying it sent the offer again. The chain listener's cancel retry already withdraws the offer of an agreement left cancelling, so the job now logs the failed read and finishes. * fix(listener): retry cancels on a timer, not by keeping polls fast Agreements still cancelling kept the listener at its fast poll rate so their retry ran often, and one given up on kept it there for good. The retry now runs every 5 minutes at any rate. * fix(registry): count fees of agreements still being cancelled An agreement dipper is cancelling is paid until the cancel lands, so it now stays in the fee estimates the indexer selection service uses to compare indexers. * fix(registry): cancel a replaced agreement marked expired as well A lagging listener can mark an agreement expired that was in fact accepted. Such an agreement couldn't be marked cancelling, so its replacement's acceptance no longer cancelled it. * fix(listener): start orphan cancels like any other, with a retry limit The sweep for accepted agreements left behind when a request was cancelled resent their cancel on every run with no limit. It now marks each one cancelling, and the cancel retry finishes it. * refactor(registry): require every registry to implement the cancel retry The production registry trait gave the retry's 2 queries do-nothing defaults, so a registry that forgot them would silently never retry a cancel; only the test stub keeps defaults now. * refactor(registry): stop reading a cancel count nothing uses The cancelling agreements query returned each row's failed cancel count, which no caller reads. * fix(listener): never confirm a cancel the chain could not be read for A failed read was treated like a clean check, so an offer past its deadline was marked ended without knowing it wasn't live, as an accept the listener hadn't yet recorded would be. * fix(listener): record no accept of an old agreement being cancelled Agreements created before dipper announced lifecycle events are never announced, but one being cancelled had its accept recorded from the chain, which then announced it. * fix(listener): cancel a replaced expired agreement only if it is live Every replaced agreement marked expired was relabelled cancelled, losing its expired event and the record that the indexer let the offer lapse. Only one the chain shows live is cancelled. * fix(listener): limit retries of a cancel that is mined and reverts A mined cancel that reverted was retried every 5 minutes without limit, costing gas each time. It now counts towards the limit; one that reverts before sending is retried, logged as an ERROR. * fix(listener): mark an ended accepted agreement after an hour's wait The retry left an accepted agreement that already ended for the chain listener to confirm, so one whose end the listener never read stayed cancelling for good. After an hour it marks it. * fix(listener): stop a cancel retry sweep after 30 seconds Each retried cancel can wait 15 seconds to be mined, and the sweep holds up the chain listener, so 10 slow ones stalled it for minutes. A sweep now leaves what it hasn't reached to the next. * docs(worker): say only the chain listener queues the on-chain cancel job A reassessment no longer queues it when its own cancel fails; the cancel retry handles that. --- .../handlers/indexing_agreements.rs | 1 + bin/dipper-service/src/cancel_dispatch.rs | 159 +++- bin/dipper-service/src/config.rs | 31 + bin/dipper-service/src/network/service.rs | 1 + .../src/network/service/cancel_retry.rs | 551 +++++++++++++ .../src/network/service/chain_listener.rs | 749 +++++++++++------- .../src/network/service/liveness_checker.rs | 24 +- bin/dipper-service/src/registry.rs | 41 +- bin/dipper-service/src/registry/agreement.rs | 52 +- .../src/registry/agreement_stub.rs | 54 +- .../cancel_rejected_agreement_on_chain.rs | 115 ++- .../handlers/reassess_indexing_request.rs | 273 +++---- .../send_indexing_agreement_proposal.rs | 24 + .../src/worker/handlers/submit_offer.rs | 135 ++-- .../20261002000000_add_cancelling_status.sql | 15 + dipper-pgregistry/src/indexing_agreement.rs | 6 + dipper-pgregistry/src/lib.rs | 6 +- dipper-pgregistry/src/postgres.rs | 178 ++++- .../tests/it_registry_postgres.rs | 178 +++++ dipper-rpc/src/admin/indexing_agreements.rs | 5 + 20 files changed, 1941 insertions(+), 657 deletions(-) create mode 100644 bin/dipper-service/src/network/service/cancel_retry.rs create mode 100644 dipper-pgregistry/migrations/20261002000000_add_cancelling_status.sql diff --git a/bin/dipper-service/src/admin_rpc_server/handlers/indexing_agreements.rs b/bin/dipper-service/src/admin_rpc_server/handlers/indexing_agreements.rs index 689b7415..827537ff 100644 --- a/bin/dipper-service/src/admin_rpc_server/handlers/indexing_agreements.rs +++ b/bin/dipper-service/src/admin_rpc_server/handlers/indexing_agreements.rs @@ -135,5 +135,6 @@ fn into_indexing_agreement_status( IndexingAgreementRecordStatus::AbandonedByIndexer => { IndexingAgreementStatus::AbandonedByIndexer } + IndexingAgreementRecordStatus::Cancelling => IndexingAgreementStatus::Cancelling, } } diff --git a/bin/dipper-service/src/cancel_dispatch.rs b/bin/dipper-service/src/cancel_dispatch.rs index 7452dbd6..e4c9c80f 100644 --- a/bin/dipper-service/src/cancel_dispatch.rs +++ b/bin/dipper-service/src/cancel_dispatch.rs @@ -1,12 +1,15 @@ //! On-chain cancel dispatch. Every cancel goes through //! [`cancel_agreement_on_chain`] so the manager-routed path lives in one place. +use dipper_core::time::now_secs; use thegraph_core::alloy::primitives::B256; use crate::{ chain_client::{ChainClient, ChainClientError}, config::IndexingAgreementConfig, - registry::IndexingAgreement, + registry::{ + AgreementRegistry, IndexingAgreement, IndexingAgreementStatus, Result as RegistryResult, + }, }; /// Pass both ACTIVE and PENDING; local status lags the chain, so let the @@ -59,8 +62,135 @@ pub async fn cancel_agreement_on_chain( Ok(outcome) } +/// What [`start_cancel`] left an agreement as. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum CancelStarted { + /// It was accepted and its cancel landed: now `CanceledByRequester`. + Ended, + /// Still `Cancelling`; the chain listener finishes it once it can't go live. + Cancelling, +} + +/// Start ending an agreement that may be live on-chain. It is marked `Cancelling` before +/// its cancel goes out, so an offer for it still in flight withdraws itself on landing. +/// Fails, sending nothing, when the mark can't be written. +pub async fn start_cancel( + registry: &R, + chain_client: &T, + agreement: &IndexingAgreement, + config: &IndexingAgreementConfig, +) -> RegistryResult +where + R: AgreementRegistry + Sync, + T: ChainClient, +{ + registry + .mark_indexing_agreement_as_cancelling(&agreement.id) + .await?; + let tx_hash = match cancel_agreement_on_chain(chain_client, agreement, config).await { + Ok(tx_hash) => tx_hash, + Err(err) => { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "On-chain cancel failed; the chain listener retries it" + ); + return Ok(CancelStarted::Cancelling); + } + }; + tracing::info!( + agreement_id = %agreement.id, + tx_hash = ?tx_hash, + "Submitted on-chain cancellation" + ); + // An offer never accepted could still land and be accepted until its deadline. + if agreement.status != IndexingAgreementStatus::AcceptedOnChain { + return Ok(CancelStarted::Cancelling); + } + Ok(confirm_cancelled(registry, agreement, tx_hash, config).await) +} + +/// Mark an accepted agreement whose cancel landed `CanceledByRequester` and record the +/// cancel, so the `terminated` sweep announces it. +async fn confirm_cancelled( + registry: &R, + agreement: &IndexingAgreement, + tx_hash: Option, + config: &IndexingAgreementConfig, +) -> CancelStarted { + if let Err(err) = registry + .mark_indexing_agreement_as_canceled_by_requester(&agreement.id) + .await + { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to mark a cancelled agreement; the chain listener finishes it" + ); + return CancelStarted::Cancelling; + } + record_cancel(registry, agreement, tx_hash, config).await; + CancelStarted::Ended +} + +/// Record dipper's own cancel of an accepted agreement, so the `terminated` sweep +/// announces it. +pub async fn record_cancel( + registry: &R, + agreement: &IndexingAgreement, + tx_hash: Option, + config: &IndexingAgreementConfig, +) { + let manager = config.recurring_agreement_manager().to_string(); + let tx = tx_hash.map(|hash| hash.to_string()); + if let Err(err) = registry + .record_cancel_audit(&agreement.id, now_secs(), &manager, tx.as_deref()) + .await + { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "failed to record cancel audit; terminated event may emit with fallback fields" + ); + } +} + +/// What [`cancel_if_live`] found and did. +#[derive(Debug)] +pub enum LiveCancel { + /// The chain showed nothing live, so no cancel was sent. + NotLive, + /// A cancel went out and the chain confirmed the agreement ended. + Ended(Option), + /// The chain could not be read, so nothing was sent. + ReadFailed(ChainClientError), + /// The cancel failed or did not end the agreement. + CancelFailed(ChainClientError), +} + +/// Cancel an agreement on-chain only if the chain shows it live: a pending offer, or +/// accepted and not yet ended. Reading first saves a wasted transaction, since a cancel +/// of an agreement that already ended still mines. +pub async fn cancel_if_live( + chain_client: &T, + agreement: &IndexingAgreement, + config: &IndexingAgreementConfig, +) -> LiveCancel { + match chain_client + .agreement_still_active(agreement.id.as_bytes()) + .await + { + Err(err) => LiveCancel::ReadFailed(err), + Ok(false) => LiveCancel::NotLive, + Ok(true) => match cancel_agreement_on_chain(chain_client, agreement, config).await { + Ok(tx_hash) => LiveCancel::Ended(tx_hash), + Err(err) => LiveCancel::CancelFailed(err), + }, + } +} + #[cfg(test)] -mod tests { +pub(crate) mod tests { use std::sync::Mutex; use async_trait::async_trait; @@ -154,31 +284,16 @@ mod tests { fn manager_conf(collector: Address) -> IndexingAgreementConfig { IndexingAgreementConfig { - data_service: Address::ZERO, recurring_collector: collector, recurring_agreement_manager: Address::repeat_byte(0x33), - max_agreement_grt_per_30_days: 0.0, - max_seconds_per_collection: 0, - min_seconds_per_collection: 0, - duration_seconds: 0, - deadline_seconds: 0, - max_grt_per_30_days: std::collections::BTreeMap::new(), - max_grt_per_billion_entities_per_30_days: 0.0, - declined_indexer_lookback_days: 0, - price_rejection_lookback_days: 0, - transient_rejection_lookback_minutes: 0, - uncertain_rejection_lookback_days: 0, - unresponsive_indexer_lookback_days: 0, - mass_unresponsive_trip_fraction: 0.5, - mass_unresponsive_reset_fraction: 0.25, - dips_accepting_snapshot_max_age_hours: 48, - dips_accepting_cache_ttl_seconds: 300, - max_in_flight_offers_per_indexer: None, - max_in_flight_offers_total: None, + ..IndexingAgreementConfig::for_tests() } } - fn agreement(status: IndexingAgreementStatus, hash: Option>) -> IndexingAgreement { + pub(crate) fn agreement( + status: IndexingAgreementStatus, + hash: Option>, + ) -> IndexingAgreement { let deployment_id: DeploymentId = "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" .parse() .unwrap(); diff --git a/bin/dipper-service/src/config.rs b/bin/dipper-service/src/config.rs index b480a33e..740974f3 100644 --- a/bin/dipper-service/src/config.rs +++ b/bin/dipper-service/src/config.rs @@ -1202,6 +1202,37 @@ pub struct IndexingAgreementConfig { pub max_in_flight_offers_total: Option, } +#[cfg(test)] +impl IndexingAgreementConfig { + /// Zero addresses and limits, with permissive breaker and cache settings, for tests + /// to adjust the fields they care about. + pub fn for_tests() -> Self { + Self { + data_service: Address::ZERO, + recurring_collector: Address::ZERO, + recurring_agreement_manager: Address::ZERO, + max_agreement_grt_per_30_days: 0.0, + max_seconds_per_collection: 0, + min_seconds_per_collection: 0, + duration_seconds: 0, + deadline_seconds: 0, + max_grt_per_30_days: BTreeMap::new(), + max_grt_per_billion_entities_per_30_days: 0.0, + declined_indexer_lookback_days: 0, + price_rejection_lookback_days: 0, + transient_rejection_lookback_minutes: 0, + uncertain_rejection_lookback_days: 0, + unresponsive_indexer_lookback_days: 0, + mass_unresponsive_trip_fraction: 0.5, + mass_unresponsive_reset_fraction: 0.25, + dips_accepting_snapshot_max_age_hours: 48, + dips_accepting_cache_ttl_seconds: 300, + max_in_flight_offers_per_indexer: None, + max_in_flight_offers_total: None, + } + } +} + /// Per-chain pricing for indexing agreements (runtime). #[derive(Debug)] pub struct IndexingAgreementChainPrices { diff --git a/bin/dipper-service/src/network/service.rs b/bin/dipper-service/src/network/service.rs index 86f9bb3e..3fc9d0f0 100644 --- a/bin/dipper-service/src/network/service.rs +++ b/bin/dipper-service/src/network/service.rs @@ -1,3 +1,4 @@ +pub mod cancel_retry; pub mod chain_events; pub mod chain_listener; pub mod domain_refresh; diff --git a/bin/dipper-service/src/network/service/cancel_retry.rs b/bin/dipper-service/src/network/service/cancel_retry.rs new file mode 100644 index 00000000..ca465fa9 --- /dev/null +++ b/bin/dipper-service/src/network/service/cancel_retry.rs @@ -0,0 +1,551 @@ +//! Finishes the cancels dipper starts. An agreement dipper wants ended is marked +//! `Cancelling` before its on-chain cancel goes out; this sweep re-sends the cancel while +//! the chain shows it live, and marks it `CanceledByRequester` once it can no longer be. + +use thegraph_core::alloy::primitives::B256; + +use crate::{ + cancel_dispatch::{LiveCancel, cancel_if_live, record_cancel}, + chain_client::{ChainClient, ChainClientError}, + config::IndexingAgreementConfig, + registry::{AgreementRegistry, CancellingAgreement, IndexingAgreement}, +}; + +/// Retried cancels mined without ending an agreement before dipper stops retrying it and +/// leaves it to an operator. Other failures don't count (see `failed_attempts`). +pub const MAX_CANCEL_ATTEMPTS: u32 = 10; + +/// Agreements checked per sweep, those checked longest ago first. +const BATCH_SIZE: i64 = 10; + +/// Time a sweep may take before leaving the rest to the next one: it holds up the chain +/// listener while it runs, and each cancel can wait up to 15 s to be mined. +const SWEEP_BUDGET: std::time::Duration = std::time::Duration::from_secs(30); + +/// Minutes an agreement stays out of the retry after it is marked, so the cancel sent +/// when it was marked can be mined first instead of being sent again. +const SETTLE_MINUTES: i32 = 2; + +/// How long the chain listener gets to record who ended an accepted agreement, and when, +/// before the retry marks it ended without those details. +const LISTENER_GRACE: time::Duration = time::Duration::HOUR; + +/// Retry the cancel of agreements still `Cancelling`. `chain_now`, in chain seconds, +/// decides when an offer that was never accepted no longer can be. +pub async fn retry_cancelling_agreements( + registry: &R, + chain_client: &T, + config: &IndexingAgreementConfig, + chain_now: u64, +) where + R: AgreementRegistry + Sync, + T: ChainClient, +{ + let cancelling = match registry + .get_cancelling_agreements(BATCH_SIZE, MAX_CANCEL_ATTEMPTS, SETTLE_MINUTES) + .await + { + Ok(cancelling) => cancelling, + Err(err) => { + tracing::warn!(error = %err, "Failed to list agreements still being cancelled"); + return; + } + }; + let started = std::time::Instant::now(); + for (done, row) in cancelling.iter().enumerate() { + if started.elapsed() >= SWEEP_BUDGET { + tracing::info!( + left = cancelling.len() - done, + "Cancel retry ran out of time; the rest wait for the next sweep" + ); + break; + } + retry_cancel(registry, chain_client, config, row, chain_now).await; + } +} + +async fn retry_cancel( + registry: &R, + chain_client: &T, + config: &IndexingAgreementConfig, + row: &CancellingAgreement, + chain_now: u64, +) where + R: AgreementRegistry + Sync, + T: ChainClient, +{ + let agreement_id = row.agreement.id; + let (tx_hash, failure) = match cancel_if_live(chain_client, &row.agreement, config).await { + LiveCancel::ReadFailed(err) => { + tracing::warn!( + %agreement_id, + error = %err, + "Failed to read a cancelling agreement on-chain, will retry" + ); + // Unread, it may still be live, so it can't be confirmed ended. + return note_check(registry, row, None).await; + } + LiveCancel::NotLive => (None, None), + LiveCancel::Ended(tx_hash) => { + tracing::info!( + %agreement_id, + tx_hash = ?tx_hash, + "Cancelled an agreement still live on-chain" + ); + (tx_hash, None) + } + LiveCancel::CancelFailed(err) => (None, Some(err)), + }; + if failure.is_none() && confirm_if_over(registry, config, row, tx_hash, chain_now).await { + return; + } + note_check(registry, row, failure.as_ref()).await; +} + +/// Mark the agreement `CanceledByRequester` once it can't go live again: this sweep's cancel +/// ended it, or nobody accepted its offer before the deadline to. One accepted that ended +/// otherwise is left to the chain listener, which reads who ended it and when, for a while. +async fn confirm_if_over( + registry: &R, + config: &IndexingAgreementConfig, + row: &CancellingAgreement, + tx_hash: Option, + chain_now: u64, +) -> bool { + let agreement = &row.agreement; + let can_confirm = if row.accepted_on_chain { + tx_hash.is_some() || agreement.updated_at < time::OffsetDateTime::now_utc() - LISTENER_GRACE + } else { + chain_now > agreement.terms.deadline + }; + if !can_confirm { + return false; + } + if let Err(err) = registry + .mark_indexing_agreement_as_canceled_by_requester(&agreement.id) + .await + { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to mark an ended agreement cancelled, will retry" + ); + return false; + } + tracing::info!( + agreement_id = %agreement.id, + indexing_request_id = %agreement.indexing_request_id, + old_status = "CANCELLING", + new_status = "CANCELED_BY_REQUESTER", + reason = "cancel_confirmed_on_chain", + "agreement state transition" + ); + // Without its own transaction, a late read by the chain listener fills in the cancel. + if row.accepted_on_chain && tx_hash.is_some() { + record_cancel(registry, agreement, tx_hash, config).await; + } + true +} + +/// Record that the agreement was checked and is still cancelling, counting a cancel the +/// chain answered without ending it; past the limit, dipper gives up with an ERROR. +async fn note_check( + registry: &R, + row: &CancellingAgreement, + failure: Option<&ChainClientError>, +) { + let agreement = &row.agreement; + let failed_attempts = failure.map_or(0, failed_attempts); + if let Some(err) = failure + && failed_attempts == 0 + { + log_uncounted_failure(agreement, err); + } + match registry + .record_cancel_check(&agreement.id, failed_attempts) + .await + { + Ok(attempts) => { + if let Some(err) = failure.filter(|_| failed_attempts > 0) { + log_failed_cancel(agreement, attempts, err); + } + } + Err(err) => tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to record a check of a cancelling agreement" + ), + } +} + +/// A revert before sending is logged as an ERROR: a paused or misconfigured manager makes +/// every cancel revert, so it is retried rather than counted against the agreement. +fn log_uncounted_failure(agreement: &IndexingAgreement, err: &ChainClientError) { + if matches!(err, ChainClientError::ContractRevert { .. }) { + tracing::error!( + agreement_id = %agreement.id, + error = %err, + "Cancel of an agreement reverted before it was sent, will retry" + ); + } else { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Cancel of an agreement failed or could not be confirmed, will retry" + ); + } +} + +fn log_failed_cancel(agreement: &IndexingAgreement, attempts: u32, err: &ChainClientError) { + if attempts < MAX_CANCEL_ATTEMPTS { + tracing::warn!( + agreement_id = %agreement.id, + attempts, + error = %err, + "Cancel did not end the agreement, will retry" + ); + return; + } + tracing::error!( + event = "agreement_cancel_abandoned", + agreement_id = %agreement.id, + indexer_id = %agreement.indexer.id, + indexing_request_id = %agreement.indexing_request_id, + attempts, + error = %err, + "Gave up cancelling an agreement on-chain; it may still be live" + ); +} + +/// How many of an agreement's cancel attempts a failure uses up. A cancel mined without +/// ending it, or reverted, counts, and one that can never be sent uses them all. An +/// unreachable chain or a revert before sending is retried freely. +fn failed_attempts(err: &ChainClientError) -> u32 { + match err { + ChainClientError::CancelNotConfirmed { .. } | ChainClientError::TxReverted { .. } => 1, + ChainClientError::MissingTermsVersionHash { .. } => MAX_CANCEL_ATTEMPTS, + _ => 0, + } +} + +#[cfg(test)] +mod tests { + use std::sync::{ + Mutex, + atomic::{AtomicBool, AtomicU32, Ordering}, + }; + + use async_trait::async_trait; + use dipper_core::ids::IndexingAgreementId; + use dipper_rpc::indexer::indexer_client::sol::RecurringCollectionAgreement; + use thegraph_core::alloy::primitives::Address; + + use super::*; + use crate::{ + cancel_dispatch::tests::agreement, + registry::{IndexingAgreementStatus, StubAgreementRegistry}, + }; + + const DEADLINE: u64 = 1_000; + + #[derive(Default)] + struct MockRegistry { + cancelling: Vec, + marked_cancelled: Mutex>, + audits: Mutex>>, + attempts: AtomicU32, + checks: AtomicU32, + } + + #[async_trait] + impl StubAgreementRegistry for MockRegistry { + async fn get_cancelling_agreements( + &self, + _batch_size: i64, + _max_attempts: u32, + _min_age_minutes: i32, + ) -> crate::registry::Result> { + Ok(self.cancelling.clone()) + } + async fn mark_indexing_agreement_as_canceled_by_requester( + &self, + id: &IndexingAgreementId, + ) -> crate::registry::Result<()> { + self.marked_cancelled.lock().unwrap().push(*id); + Ok(()) + } + async fn record_cancel_audit( + &self, + _id: &IndexingAgreementId, + _canceled_at: u64, + _canceled_by: &str, + canceled_tx: Option<&str>, + ) -> crate::registry::Result<()> { + self.audits + .lock() + .unwrap() + .push(canceled_tx.map(str::to_owned)); + Ok(()) + } + async fn record_cancel_check( + &self, + _id: &IndexingAgreementId, + failed_attempts: u32, + ) -> crate::registry::Result { + self.checks.fetch_add(1, Ordering::SeqCst); + Ok(self.attempts.fetch_add(failed_attempts, Ordering::SeqCst) + failed_attempts) + } + } + + /// An agreement live on-chain until a cancel ends it, unless set to ignore cancels. + #[derive(Default)] + struct MockChain { + live: AtomicBool, + read_fails: bool, + send_fails: bool, + mined_cancel_reverts: bool, + cancel_has_no_effect: bool, + cancels_sent: AtomicU32, + } + + #[async_trait] + impl ChainClient for MockChain { + async fn offer_via_manager( + &self, + _rca: &RecurringCollectionAgreement, + ) -> Result, ChainClientError> { + unimplemented!() + } + async fn cancel_via_manager( + &self, + _collector: Address, + _agreement_id: &[u8; 16], + _version_hash: B256, + _options: u16, + ) -> Result, ChainClientError> { + if self.send_fails { + return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); + } + if self.mined_cancel_reverts { + return Err(ChainClientError::TxReverted { + tx_hash: B256::repeat_byte(0xee), + }); + } + self.cancels_sent.fetch_add(1, Ordering::SeqCst); + if !self.cancel_has_no_effect { + self.live.store(false, Ordering::SeqCst); + } + Ok(Some(B256::repeat_byte(0xcd))) + } + async fn reconcile_provider( + &self, + _collector: Address, + _provider: Address, + ) -> Result, ChainClientError> { + unimplemented!() + } + async fn reconcile_agreement( + &self, + _collector: Address, + _agreement_id: &[u8; 16], + ) -> Result, ChainClientError> { + unimplemented!() + } + async fn agreement_still_active( + &self, + _agreement_id: &[u8; 16], + ) -> Result { + if self.read_fails { + return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); + } + Ok(self.live.load(Ordering::SeqCst)) + } + async fn latest_block_timestamp(&self) -> Result { + unimplemented!() + } + } + + fn registry_with_one(accepted_on_chain: bool) -> MockRegistry { + let mut cancelling = agreement(IndexingAgreementStatus::Cancelling, Some(vec![7u8; 32])); + cancelling.terms.deadline = DEADLINE; + MockRegistry { + cancelling: vec![CancellingAgreement { + agreement: cancelling, + accepted_on_chain, + }], + ..MockRegistry::default() + } + } + + fn live_chain() -> MockChain { + MockChain { + live: AtomicBool::new(true), + ..MockChain::default() + } + } + + async fn retry(registry: &MockRegistry, chain: &MockChain, chain_now: u64) { + let config = IndexingAgreementConfig::for_tests(); + retry_cancelling_agreements(registry, chain, &config, chain_now).await; + } + + #[tokio::test] + async fn cancels_a_live_accepted_agreement_and_records_the_cancel() { + let registry = registry_with_one(true); + let chain = live_chain(); + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 1); + assert_eq!(registry.marked_cancelled.lock().unwrap().len(), 1); + let tx = B256::repeat_byte(0xcd).to_string(); + assert_eq!(*registry.audits.lock().unwrap(), vec![Some(tx)]); + } + + #[tokio::test] + async fn leaves_an_accepted_agreement_that_already_ended_to_the_listener() { + // The indexer may have ended it, or an earlier cancel whose result went unread; + // the chain listener reads which, and records when and in which transaction. + let registry = registry_with_one(true); + let chain = MockChain::default(); + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 0); + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + assert!(registry.audits.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn marks_an_ended_accepted_agreement_itself_once_the_listener_has_had_long_enough() { + // In case the listener never reads its end, it would otherwise stay cancelling. + let mut registry = registry_with_one(true); + registry.cancelling[0].agreement.updated_at = + time::OffsetDateTime::now_utc() - LISTENER_GRACE - time::Duration::MINUTE; + let chain = MockChain::default(); + + retry(®istry, &chain, 0).await; + + assert_eq!(registry.marked_cancelled.lock().unwrap().len(), 1); + assert!(registry.audits.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn keeps_an_unaccepted_agreement_cancelling_until_its_deadline() { + // An offer still in flight could land and be accepted until then. + let registry = registry_with_one(false); + let chain = MockChain::default(); + + retry(®istry, &chain, DEADLINE).await; + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + + retry(®istry, &chain, DEADLINE + 1).await; + assert_eq!(registry.marked_cancelled.lock().unwrap().len(), 1); + assert!(registry.audits.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn withdraws_a_live_offer_before_its_deadline_and_keeps_watching() { + let registry = registry_with_one(false); + let chain = live_chain(); + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 1); + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn notes_each_check_that_leaves_an_agreement_cancelling() { + // So the next sweep starts with the agreements checked longest ago. + let registry = registry_with_one(false); + let chain = MockChain::default(); + + retry(®istry, &chain, DEADLINE).await; + + assert_eq!(registry.checks.load(Ordering::SeqCst), 1); + assert_eq!(registry.attempts.load(Ordering::SeqCst), 0); + } + + #[tokio::test] + async fn never_confirms_an_agreement_it_could_not_read() { + // Past its deadline but unread, it may be live: an accept the listener hasn't + // recorded, or one from before accepts were recorded. + let registry = registry_with_one(false); + let chain = MockChain { + read_fails: true, + ..live_chain() + }; + + retry(®istry, &chain, DEADLINE + 1).await; + + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + assert_eq!(registry.checks.load(Ordering::SeqCst), 1); + } + + #[tokio::test] + async fn an_unreachable_chain_neither_sends_nor_counts_an_attempt() { + for chain in [ + MockChain { + read_fails: true, + ..live_chain() + }, + MockChain { + send_fails: true, + ..live_chain() + }, + ] { + let registry = registry_with_one(true); + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 0); + assert_eq!(registry.attempts.load(Ordering::SeqCst), 0); + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + } + } + + #[tokio::test] + async fn gives_up_at_once_on_an_agreement_it_can_never_cancel() { + // Without a stored terms hash no cancel can be sent, so retrying only delays the alert. + let mut registry = registry_with_one(true); + registry.cancelling[0].agreement.terms_version_hash = None; + let chain = live_chain(); + + retry(®istry, &chain, 0).await; + + assert_eq!( + registry.attempts.load(Ordering::SeqCst), + MAX_CANCEL_ATTEMPTS + ); + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 0); + } + + #[tokio::test] + async fn counts_a_cancel_that_is_mined_and_reverts() { + // Each one costs gas, so it can't be retried without limit. + let registry = registry_with_one(true); + let chain = MockChain { + mined_cancel_reverts: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(registry.attempts.load(Ordering::SeqCst), 1); + } + + #[tokio::test] + async fn counts_a_cancel_that_mines_without_ending_the_agreement() { + let registry = registry_with_one(true); + let chain = MockChain { + cancel_has_no_effect: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(registry.attempts.load(Ordering::SeqCst), 1); + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + } +} diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index f5ce990d..1b16dc77 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -83,6 +83,9 @@ const SWEEP_BATCH_SIZE: i64 = 1000; /// crash-recovery; the steady-state fan-out fires from finalize on a /// fresh accept, so per-poll execution is wasted DB work. const SWEEP_POLLS: u64 = 60; +/// How often agreements still being cancelled get their cancel retried, whichever rate the +/// listener polls at: just under the slow poll interval, so every slow poll retries. +const CANCEL_RETRY_INTERVAL: Duration = Duration::from_secs(290); /// Handle for controlling the chain listener service lifecycle #[derive(Clone)] @@ -228,6 +231,7 @@ where // Starts at SWEEP_POLLS so the first poll runs the sweep, // recovering any pre-startup orphans. let mut polls_since_sweep: u64 = SWEEP_POLLS; + let mut last_cancel_retry: Option = None; // Pause the event sweeps after a Kafka send failure, backing off from one // poll interval up to the idle interval, so a hung broker cannot stall the // poll loop on every iteration. @@ -293,6 +297,21 @@ where tokio::time::sleep(backoff).await; } + // Ahead of the drain, which ends the poll early while the subgraph is down: + // finishing a cancel needs only the chain. + if last_cancel_retry.is_none_or(|at| at.elapsed() >= CANCEL_RETRY_INTERVAL) { + last_cancel_retry = Some(Instant::now()); + let chain_now = + last_persisted_timestamp.unwrap_or_else(dipper_core::time::now_secs); + super::cancel_retry::retry_cancelling_agreements( + ®istry, + &chain_client, + &agreement_conf, + chain_now, + ) + .await; + } + let outcome = match drain_once( &mut cursor, &mut last_persisted_timestamp, @@ -877,6 +896,7 @@ where } queue_cancel_if_cancelled_but_accepted(snapshot, &agreement, worker_queue).await?; + record_accept_of_cancelling(snapshot, &agreement, registry).await; // Both transitions are applied atomically downstream so the // Accept-then-Cancel-in-one-snapshot path can't leak an intermediate @@ -936,9 +956,10 @@ where } } -/// How long after an offer's deadline dipper's cancel can still have raced an -/// accept the listener hadn't caught up with. -const LISTENER_LAG_SLACK_SECS: i64 = 3_600; +/// When v0.1.10, the first release that announces lifecycle events, came out (2026-08-11 +/// UTC). Agreements dipper created before then are never announced, even when a replay of +/// the chain reads them again. +const LIFECYCLE_EVENTS_START: i64 = 1_786_406_400; /// Record the accept and cancel of an agreement dipper had already marked /// cancelled that went live on-chain first, so its accepted and terminated @@ -949,7 +970,7 @@ async fn record_accept_and_cancel_from_chain( agreement: &IndexingAgreement, registry: &R, ) { - if snapshot.accepted_at == 0 || !cancelled_while_offer_open(agreement) { + if snapshot.accepted_at == 0 || !created_after_events_started(agreement) { return; } let canceled_by = snapshot.canceled_by.to_string(); @@ -979,12 +1000,10 @@ async fn record_accept_and_cancel_from_chain( } } -/// Whether dipper cancelled the agreement while its offer could still be accepted, -/// so an accept on-chain may have slipped past it. One accepted before lifecycle -/// events existed was cancelled long after its deadline and is never announced. -fn cancelled_while_offer_open(agreement: &IndexingAgreement) -> bool { - let deadline = i64::try_from(agreement.terms.deadline).unwrap_or(i64::MAX); - agreement.updated_at.unix_timestamp() <= deadline.saturating_add(LISTENER_LAG_SLACK_SECS) +/// Whether dipper created the agreement once it announced lifecycle events. Its own clock +/// at creation, unlike any read of the chain, doesn't depend on how far the listener lags. +fn created_after_events_started(agreement: &IndexingAgreement) -> bool { + agreement.created_at.unix_timestamp() >= LIFECYCLE_EVENTS_START } /// Safety net for an agreement dipper cancelled whose offer the indexer accepted @@ -1012,6 +1031,33 @@ async fn queue_cancel_if_cancelled_but_accepted( Ok(()) } +/// An agreement dipper is cancelling stays `Cancelling` when the chain shows it accepted, +/// so its accept is recorded here; its end is then announced, along with the accept, +/// once the cancel lands. A withdrawn offer reads as cancelled with no accept time. +async fn record_accept_of_cancelling( + snapshot: &AgreementStateSnapshot, + agreement: &IndexingAgreement, + registry: &R, +) { + if agreement.status != IndexingAgreementStatus::Cancelling + || !snapshot.state.reached_accepted() + || snapshot.accepted_at == 0 + || !created_after_events_started(agreement) + { + return; + } + if let Err(err) = registry + .record_accepted_audit(&agreement.id, snapshot.accepted_at, &snapshot.accepted_tx) + .await + { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "failed to record the on-chain accept of an agreement being cancelled" + ); + } +} + /// Log the transition that landed and, on fresh accepts, fan out the /// linked pending cancellations. #[expect( @@ -1115,16 +1161,9 @@ where /// Execute pending cancellations linked to a newly-accepted agreement. /// /// Called from the Created -> AcceptedOnChain and Expired -> AcceptedOnChain -/// transitions. For each pending cancellation, fires -/// `cancelIndexingAgreementByPayer` against the RecurringCollector contract and -/// flips the dipper DB row to `CanceledByRequester`, an unaccepted one first. Each pending row is -/// deleted individually after both steps succeed; transient failures retain -/// the record so the next reconcile pass can retry. -#[expect( - clippy::cognitive_complexity, - clippy::too_many_lines, - reason = "predates this lint; fix when next touched" -)] +/// transitions. Each replaced agreement is marked `Cancelling` and its on-chain cancel +/// sent once; the cancel retry in this listener finishes any that don't end at once. +/// A pending row is deleted once its agreement is marked; a failed mark keeps it for retry. async fn execute_pending_cancellations( agreement_id: &IndexingAgreementId, registry: &R, @@ -1139,178 +1178,134 @@ where .get_pending_cancellations_by_new_agreement(*agreement_id) .await?; - if pending.is_empty() { - return Ok(()); - } - let mut transient_failures: u32 = 0; - for cancellation in &pending { - let old_agreement = match registry - .get_indexing_agreement_by_id(&cancellation.old_agreement_id) - .await? - { - None => { - tracing::warn!( - old_agreement_id = %cancellation.old_agreement_id, - "Pending cancellation references non-existent agreement, cleaning up" - ); - registry - .delete_pending_cancellation(*agreement_id, cancellation.old_agreement_id) - .await?; - continue; - } - Some(a) => a, - }; - - // An unaccepted one is marked first. An offer for it still in flight then - // lands before this cancel, which withdraws it, or sees the mark and withdraws itself. - let unaccepted = old_agreement.status == IndexingAgreementStatus::Created; - if unaccepted { - match registry - .mark_indexing_agreement_as_canceled_by_requester(&cancellation.old_agreement_id) - .await - { - Ok(()) | Err(crate::registry::Error::NoRecordsUpdated) => {} - Err(err) => { - tracing::error!( - old_agreement_id = %cancellation.old_agreement_id, - error = %err, - "Failed to mark replaced agreement cancelled before its on-chain cancel, \ - retaining pending row" - ); - transient_failures += 1; - continue; - } - } - } - - let mut on_chain_cancel_tx: Option = None; - match crate::cancel_dispatch::cancel_agreement_on_chain( + let old_agreement_id = cancellation.old_agreement_id; + if start_replaced_cancel( + agreement_id, + &old_agreement_id, + registry, chain_client, - &old_agreement, config, ) - .await + .await? { - Ok(Some(tx_hash)) => { - tracing::info!( - new_agreement_id = %agreement_id, - old_agreement_id = %cancellation.old_agreement_id, - %tx_hash, - "Submitted on-chain cancellation for replaced agreement" - ); - on_chain_cancel_tx = Some(tx_hash.to_string()); - } - Ok(None) => { - tracing::info!( - new_agreement_id = %agreement_id, - old_agreement_id = %cancellation.old_agreement_id, - "Agreement already canceled on-chain; proceeding with local cleanup" - ); - } - Err(err) => { - tracing::warn!( - old_agreement_id = %cancellation.old_agreement_id, - error = %err, - "On-chain cancel failed, retaining pending cancellation for retry" - ); - transient_failures += 1; - continue; - } + registry + .delete_pending_cancellation(*agreement_id, old_agreement_id) + .await?; + } else { + transient_failures += 1; } + } - if !unaccepted { - match registry - .mark_indexing_agreement_as_canceled_by_requester(&cancellation.old_agreement_id) - .await - { - Ok(()) => {} - Err(crate::registry::Error::NoRecordsUpdated) => { - tracing::debug!( - old_agreement_id = %cancellation.old_agreement_id, - "Old agreement already in terminal state, skipping local cancel flip" - ); - registry - .delete_pending_cancellation(*agreement_id, cancellation.old_agreement_id) - .await?; - continue; - } - Err(err) => { - tracing::error!( - old_agreement_id = %cancellation.old_agreement_id, - error = %err, - "On-chain cancel succeeded but DB update failed, retaining pending row" - ); - transient_failures += 1; - continue; - } - } - } + if transient_failures > 0 { + anyhow::bail!( + "{transient_failures} pending cancellation(s) failed for agreement {}; \ + records retained for retry", + agreement_id, + ); + } - registry - .delete_pending_cancellation(*agreement_id, cancellation.old_agreement_id) - .await?; + Ok(()) +} - tracing::info!( - new_agreement_id = %agreement_id, - old_agreement_id = %cancellation.old_agreement_id, - "Canceled old agreement on-chain and in dipper DB after replacement confirmed" +/// Start cancelling an agreement its accepted replacement supersedes. Returns whether its +/// pending cancellation is done with: marked, already ended, or gone. +async fn start_replaced_cancel( + new_agreement_id: &IndexingAgreementId, + old_agreement_id: &IndexingAgreementId, + registry: &R, + chain_client: &T, + config: &crate::config::IndexingAgreementConfig, +) -> anyhow::Result +where + R: AgreementRegistry + Sync, + T: ChainClient, +{ + let Some(old_agreement) = registry + .get_indexing_agreement_by_id(old_agreement_id) + .await? + else { + tracing::warn!( + %old_agreement_id, + "Pending cancellation references non-existent agreement, cleaning up" ); - - // Record the cancel audit so `sweep_pending_terminated_events` can emit - // the `terminated` durably. Only for an accepted agreement: one that never - // was gets the chain's own cancel data if the indexer accepts it after all. - if old_agreement.status != IndexingAgreementStatus::AcceptedOnChain { - continue; + return Ok(true); + }; + if old_agreement.status == IndexingAgreementStatus::Expired { + match expired_but_live(chain_client, &old_agreement).await { + Some(true) => {} + Some(false) => return Ok(true), + None => return Ok(false), } - let manager = config.recurring_agreement_manager().to_string(); - if let Err(err) = registry - .record_cancel_audit( - &cancellation.old_agreement_id, - dipper_core::time::now_secs(), - &manager, - on_chain_cancel_tx.as_deref(), - ) - .await - { + } + let started = + crate::cancel_dispatch::start_cancel(registry, chain_client, &old_agreement, config).await; + Ok(note_replaced_cancel( + new_agreement_id, + &old_agreement, + started, + )) +} + +/// Whether an agreement marked `Expired` is live on-chain after all, accepted unseen by a +/// lagging listener; `None`, logged, when the chain can't be read. One that really expired +/// stays `Expired`. +async fn expired_but_live( + chain_client: &T, + agreement: &IndexingAgreement, +) -> Option { + match chain_client + .agreement_still_active(agreement.id.as_bytes()) + .await + { + Ok(live) => Some(live), + Err(err) => { tracing::warn!( - old_agreement_id = %cancellation.old_agreement_id, + old_agreement_id = %agreement.id, error = %err, - "failed to record cancel audit; terminated event may emit with fallback fields" + "Failed to read whether a replaced expired agreement is live, retaining pending row" ); + None } } +} - if transient_failures > 0 { - anyhow::bail!( - "{transient_failures} pending cancellation(s) failed for agreement {}; \ - records retained for retry", - agreement_id, - ); +/// Log how cancelling a replaced agreement started; false when it couldn't be marked. +fn note_replaced_cancel( + new_agreement_id: &IndexingAgreementId, + old_agreement: &IndexingAgreement, + started: crate::registry::Result, +) -> bool { + let old_agreement_id = old_agreement.id; + match started { + Ok(started) => tracing::info!( + %new_agreement_id, + %old_agreement_id, + old_status = %old_agreement.status, + ?started, + reason = "replacement_accepted", + "Cancelling replaced agreement" + ), + Err(crate::registry::Error::NoRecordsUpdated) => tracing::debug!( + %old_agreement_id, + "Replaced agreement already ended or being cancelled" + ), + Err(err) => { + tracing::error!( + %old_agreement_id, + error = %err, + "Failed to mark replaced agreement cancelling, retaining pending row" + ); + return false; + } } - - Ok(()) + true } -/// Retry on-chain cancels for agreements orphaned by a failed shrink-to-zero. -/// -/// When a `set_indexing_target_candidates(num_candidates = 0)` call flips the -/// request row to `Canceled`, reassessment fires `cancelIndexingAgreementByPayer` -/// for every agreement under it. A transient chain-client error during that -/// fan-out leaves the request row `Canceled` and at least one agreement still -/// `AcceptedOnChain` — the local intent and on-chain state disagree, and the -/// admin RPC has nothing left to trigger. -/// -/// This sweep runs periodically on the chain_listener tick and re-fires the -/// on-chain cancel for each such orphan. The chain-side cancel is idempotent -/// (the `Ok(None)` revert path handles already-canceled agreements), and the -/// DB transition is gated on chain success, so this is safe to run on every -/// sweep without coordination with reassessment. -#[expect( - clippy::cognitive_complexity, - reason = "predates this lint; fix when next touched" -)] +/// Start cancelling agreements orphaned by a failed shrink-to-zero: still `AcceptedOnChain` +/// although their request is `Canceled`, because reassessment couldn't mark them. Each starts +/// like any other cancel, so the cancel retry finishes it with the same limit. async fn sweep_orphan_canceled_agreements( registry: &R, chain_client: &T, @@ -1333,76 +1328,29 @@ async fn sweep_orphan_canceled_agreements( } }; - if orphans.is_empty() { - return; - } - - tracing::debug!( - count = orphans.len(), - "Sweeping orphan agreements whose parent request is Canceled" - ); - for agreement in orphans { - let mut on_chain_cancel_tx: Option = None; - match crate::cancel_dispatch::cancel_agreement_on_chain(chain_client, &agreement, config) - .await - { - Ok(Some(tx_hash)) => { - tracing::info!( - agreement_id = %agreement.id, - %tx_hash, - "Submitted on-chain cancel for orphan agreement" - ); - on_chain_cancel_tx = Some(tx_hash.to_string()); - } - Ok(None) => { - tracing::info!( - agreement_id = %agreement.id, - "Orphan agreement already canceled on-chain; cleaning up local state" - ); - } - Err(err) => { - tracing::warn!( - error = %err, - agreement_id = %agreement.id, - "Failed to cancel orphan agreement on-chain; will retry next sweep" - ); - continue; - } - } - - if let Err(err) = registry - .mark_indexing_agreement_as_canceled_by_requester(&agreement.id) - .await - { - tracing::error!( - error = %err, - agreement_id = %agreement.id, - "Failed to mark orphan agreement as canceled in local DB" - ); - continue; - } + let started = + crate::cancel_dispatch::start_cancel(registry, chain_client, &agreement, config).await; + log_orphan_cancel(&agreement, started); + } +} - // Orphan (previously accepted) agreement canceled on-chain by dipper. - // Record the cancel audit; `sweep_pending_terminated_events` emits the - // `terminated` durably (the row is `AcceptedOnChain` -> terminal, so it is - // sweep-eligible). - let manager = config.recurring_agreement_manager().to_string(); - if let Err(err) = registry - .record_cancel_audit( - &agreement.id, - dipper_core::time::now_secs(), - &manager, - on_chain_cancel_tx.as_deref(), - ) - .await - { - tracing::warn!( - agreement_id = %agreement.id, - error = %err, - "failed to record cancel audit; terminated event may emit with fallback fields" - ); - } +fn log_orphan_cancel( + agreement: &IndexingAgreement, + started: crate::registry::Result, +) { + match started { + Ok(started) => tracing::info!( + agreement_id = %agreement.id, + ?started, + reason = "request_canceled", + "Cancelling orphan agreement" + ), + Err(err) => tracing::warn!( + error = %err, + agreement_id = %agreement.id, + "Failed to mark orphan agreement cancelling; will retry next sweep" + ), } } @@ -1903,29 +1851,7 @@ mod tests { } fn test_agreement_conf() -> std::sync::Arc { - std::sync::Arc::new(crate::config::IndexingAgreementConfig { - data_service: thegraph_core::alloy::primitives::Address::ZERO, - recurring_collector: thegraph_core::alloy::primitives::Address::ZERO, - recurring_agreement_manager: thegraph_core::alloy::primitives::Address::ZERO, - max_agreement_grt_per_30_days: 0.0, - max_seconds_per_collection: 0, - min_seconds_per_collection: 0, - duration_seconds: 0, - deadline_seconds: 0, - max_grt_per_30_days: std::collections::BTreeMap::new(), - max_grt_per_billion_entities_per_30_days: 0.0, - declined_indexer_lookback_days: 0, - price_rejection_lookback_days: 0, - transient_rejection_lookback_minutes: 0, - uncertain_rejection_lookback_days: 0, - unresponsive_indexer_lookback_days: 0, - mass_unresponsive_trip_fraction: 0.5, - mass_unresponsive_reset_fraction: 0.25, - dips_accepting_snapshot_max_age_hours: 48, - dips_accepting_cache_ttl_seconds: 300, - max_in_flight_offers_per_indexer: None, - max_in_flight_offers_total: None, - }) + std::sync::Arc::new(crate::config::IndexingAgreementConfig::for_tests()) } // Deterministic audit values used by `make_snapshot` so emit tests can assert @@ -1975,6 +1901,7 @@ mod tests { agreements: std::collections::HashMap, marked_accepted_on_chain: Vec, marked_canceled_by_requester: Vec, + marked_cancelling: Vec, marked_canceled_by_indexer: Vec, /// Ids passed to `record_cancel_audit` -- the signal a cancel path drives /// the terminated event (the sweep emits from this audit). @@ -2057,6 +1984,12 @@ mod tests { } } + fn set_agreement_created_at(&self, agreement_id: IndexingAgreementId, unix: i64) { + if let Some(a) = self.state.lock().unwrap().agreements.get_mut(&agreement_id) { + a.created_at = OffsetDateTime::from_unix_timestamp(unix).unwrap(); + } + } + fn set_agreement_request_id( &self, agreement_id: IndexingAgreementId, @@ -2075,6 +2008,10 @@ mod tests { .contains(id) } + fn was_marked_cancelling(&self, id: &IndexingAgreementId) -> bool { + self.state.lock().unwrap().marked_cancelling.contains(id) + } + fn was_marked_canceled_by_requester(&self, id: &IndexingAgreementId) -> bool { self.state .lock() @@ -2250,6 +2187,39 @@ mod tests { Ok(()) } + async fn mark_indexing_agreement_as_cancelling( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()> { + let mut state = self.state.lock().unwrap(); + if state.fail_cancel_for.contains(id) { + return Err(crate::registry::Error::BackendError( + dipper_pgregistry::Error::DbError(sqlx::Error::Protocol( + "simulated transient failure".into(), + )), + )); + } + state.marked_cancelling.push(*id); + Ok(()) + } + + async fn get_cancelling_agreements( + &self, + _batch_size: i64, + _max_attempts: u32, + _min_age_minutes: i32, + ) -> RegistryResult> { + Ok(Vec::new()) + } + + async fn record_cancel_check( + &self, + _id: &IndexingAgreementId, + failed_attempts: u32, + ) -> RegistryResult { + Ok(failed_attempts) + } + async fn record_cancel_audit( &self, agreement_id: &IndexingAgreementId, @@ -2322,12 +2292,16 @@ mod tests { Some( IndexingAgreementStatus::Created | IndexingAgreementStatus::AcceptedOnChain - | IndexingAgreementStatus::Rejected, + | IndexingAgreementStatus::Rejected + | IndexingAgreementStatus::Cancelling, ), ), Some(crate::registry::CancelKind::ByIndexer) => matches!( effective_status_for_cancel, - Some(IndexingAgreementStatus::AcceptedOnChain), + Some( + IndexingAgreementStatus::AcceptedOnChain + | IndexingAgreementStatus::Cancelling, + ), ), None => false, }; @@ -2594,6 +2568,9 @@ mod tests { /// When set, each cancel records whether its agreement was already marked. registry: Option, marked_at_cancel: Arc>>, + fail_cancels: bool, + /// When set, every agreement reads as live until a cancel is sent for it. + live_until_cancelled: bool, } impl MockChainClient { @@ -2638,12 +2615,17 @@ mod tests { Option, crate::chain_client::ChainClientError, > { + if self.fail_cancels { + return Err(crate::chain_client::ChainClientError::RpcError( + anyhow::anyhow!("rpc down"), + )); + } // Record manager-routed cancels through the same recorder so the // existing assertions hold. self.cancels.lock().unwrap().push(*agreement_id); if let Some(registry) = &self.registry { let id = IndexingAgreementId::from_bytes(*agreement_id); - let was_marked = registry.was_marked_canceled_by_requester(&id); + let was_marked = registry.was_marked_cancelling(&id); self.marked_at_cancel.lock().unwrap().push(was_marked); } // A manager-routed cancel has no "already canceled" result: the @@ -2675,11 +2657,11 @@ mod tests { async fn agreement_still_active( &self, - _agreement_id: &[u8; 16], + agreement_id: &[u8; 16], ) -> Result { // Cancel dispatch always reads back after a mined cancel; reporting // not-active here means "cancel confirmed", which these tests expect. - Ok(false) + Ok(self.live_until_cancelled && !self.cancels.lock().unwrap().contains(agreement_id)) } } @@ -2756,6 +2738,120 @@ mod tests { assert!(!worker_queue.was_cancellation_queued(&agreement_id)); } + #[tokio::test] + async fn reconcile_records_the_accept_of_an_agreement_being_cancelled() { + // It stays cancelling, and the recorded accept lets its end be announced. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let worker_queue = MockWorkerQueue::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::Cancelling); + + let snapshot = make_snapshot(agreement_id, AgreementState::Accepted, Address::ZERO); + reconcile_agreement( + &snapshot, + ®istry, + &worker_queue, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await + .expect("reconcile ok"); + + assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); + assert_eq!(registry.audit_writes(), vec![("accept", agreement_id)]); + assert!(!worker_queue.was_cancellation_queued(&agreement_id)); + } + + #[tokio::test] + async fn reconcile_records_no_accept_of_an_agreement_from_before_events_existed() { + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let worker_queue = MockWorkerQueue::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::Cancelling); + registry.set_agreement_created_at(agreement_id, LIFECYCLE_EVENTS_START - 1); + + let snapshot = make_snapshot(agreement_id, AgreementState::Accepted, Address::ZERO); + reconcile_agreement( + &snapshot, + ®istry, + &worker_queue, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await + .expect("reconcile ok"); + + assert!(registry.audit_writes().is_empty()); + } + + #[tokio::test] + async fn reconcile_records_no_accept_for_a_withdrawn_offer() { + // The subgraph reports a withdrawn offer as cancelled by the payer with no accept + // time; recording it would announce an agreement that was never live. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let worker_queue = MockWorkerQueue::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::Cancelling); + let mut snapshot = + make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); + snapshot.accepted_at = 0; + + reconcile_agreement( + &snapshot, + ®istry, + &worker_queue, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await + .expect("reconcile ok"); + + assert!(registry.was_marked_canceled_by_requester(&agreement_id)); + assert!( + registry + .audit_writes() + .iter() + .all(|(kind, _)| *kind != "accept") + ); + } + + #[tokio::test] + async fn reconcile_marks_an_agreement_being_cancelled_once_the_chain_shows_it_ended() { + for (state, ended_by_dipper) in [ + (AgreementState::CanceledByPayer, true), + (AgreementState::CanceledByServiceProvider, false), + ] { + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let worker_queue = MockWorkerQueue::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::Cancelling); + + let snapshot = make_snapshot(agreement_id, state, Address::ZERO); + reconcile_agreement( + &snapshot, + ®istry, + &worker_queue, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await + .expect("reconcile ok"); + + assert_eq!( + registry.was_marked_canceled_by_requester(&agreement_id), + ended_by_dipper + ); + assert_eq!( + registry.was_marked_canceled_by_indexer(&agreement_id), + !ended_by_dipper + ); + } + } + #[tokio::test] async fn reconcile_accept_marks_accepted_without_emitting() { use crate::test_support::CapturingEventsProducer; @@ -3095,8 +3191,9 @@ mod tests { let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); - // Dipper cancelled it while its offer could still be accepted. - registry.set_agreement_deadline_from_now(agreement_id, 600); + // The listener lagged: dipper only cancelled it locally long after the offer's + // deadline, which once made this accept look too old to announce. + registry.set_agreement_deadline_from_now(agreement_id, -2 * 86_400); let snapshot = make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); let result = reconcile_agreement( @@ -3124,7 +3221,6 @@ mod tests { let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); - registry.set_agreement_deadline_from_now(agreement_id, 600); registry.state.lock().unwrap().fail_cancel_audit = true; let snapshot = make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); @@ -3143,15 +3239,14 @@ mod tests { #[tokio::test] async fn test_reconcile_does_not_announce_an_agreement_accepted_before_events_existed() { - // Agreements accepted before lifecycle events existed have no recorded - // accept and are never announced. Dipper cancelled this one long after its - // offer's deadline, so it can't be an accept dipper missed. + // Agreements created before lifecycle events existed have no recorded accept + // and are never announced, even when a replay of the chain reads them again. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); - registry.set_agreement_deadline_from_now(agreement_id, -2 * 86_400); + registry.set_agreement_created_at(agreement_id, LIFECYCLE_EVENTS_START - 1); let snapshot = make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); let result = reconcile_agreement( @@ -3393,10 +3488,10 @@ mod tests { } #[tokio::test] - async fn test_pending_cancellations_records_no_audit_for_a_never_accepted_agreement() { - // A cancel record on an agreement that was never accepted would later win - // over the chain's own, if the indexer accepted it after all and it was - // then ended: the terminated event would report an end before the accept. + async fn test_pending_cancellations_leave_a_never_accepted_agreement_cancelling() { + // Its offer could still land and be accepted until the deadline, so the cancel + // retry finishes it then. No cancel is recorded: one would win over the chain's + // own if the indexer accepted it after all, reporting an end before the accept. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); let new_id = IndexingAgreementId::from_bytes(rand::random()); @@ -3415,12 +3510,13 @@ mod tests { .await; assert!(result.is_ok()); - assert!(registry.was_marked_canceled_by_requester(&old_id)); + assert!(registry.was_marked_cancelling(&old_id)); + assert!(!registry.was_marked_canceled_by_requester(&old_id)); assert!(!registry.was_cancel_audit_recorded(&old_id)); } /// Runs the pending cancellation of one old agreement in `status`, returning - /// whether it was already marked cancelled when its on-chain cancel went out. + /// whether it was already marked cancelling when its on-chain cancel went out. async fn marked_at_pending_cancel(status: IndexingAgreementStatus) -> Vec { let registry = MockRegistry::new(); let chain_client = MockChainClient { @@ -3442,7 +3538,6 @@ mod tests { .await; assert!(result.is_ok(), "got {result:?}"); - assert!(registry.was_marked_canceled_by_requester(&old_id)); assert!(registry.was_pending_cancellation_deleted(&new_id, &old_id)); chain_client.marked_at_cancel.lock().unwrap().clone() } @@ -3456,9 +3551,71 @@ mod tests { } #[tokio::test] - async fn test_pending_cancellations_mark_an_accepted_agreement_after_its_cancel() { + async fn test_pending_cancellations_cancel_an_expired_agreement_only_if_live() { + // A lagging listener can mark an agreement expired that was in fact accepted; one + // that really expired keeps its status, and no needless cancel is sent. + for live in [true, false] { + let registry = MockRegistry::new(); + let chain_client = MockChainClient { + live_until_cancelled: live, + ..MockChainClient::default() + }; + let new_id = IndexingAgreementId::from_bytes(rand::random()); + let old_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(new_id, IndexingAgreementStatus::AcceptedOnChain); + registry.add_agreement(old_id, IndexingAgreementStatus::Expired); + registry.add_pending_cancellation(new_id, old_id); + + let result = execute_pending_cancellations( + &new_id, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok(), "got {result:?}"); + assert_eq!(registry.was_marked_cancelling(&old_id), live); + assert_eq!(chain_client.was_on_chain_cancel_attempted(&old_id), live); + assert!(registry.was_pending_cancellation_deleted(&new_id, &old_id)); + } + } + + #[tokio::test] + async fn test_pending_cancellations_mark_an_accepted_agreement_before_its_cancel() { + // A failed cancel then leaves it cancelling, which the cancel retry picks up. let marked = marked_at_pending_cancel(IndexingAgreementStatus::AcceptedOnChain).await; - assert_eq!(marked, vec![false]); + assert_eq!(marked, vec![true]); + } + + #[tokio::test] + async fn test_pending_cancellations_leave_a_failed_cancel_to_the_cancel_retry() { + // Retried from here, a cancel that never works was resent on every sweep with no + // limit, for as long as the replacement stayed accepted. The cancel retry limits it. + let registry = MockRegistry::new(); + let chain_client = MockChainClient { + fail_cancels: true, + ..MockChainClient::default() + }; + let new_id = IndexingAgreementId::from_bytes(rand::random()); + let old_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(new_id, IndexingAgreementStatus::AcceptedOnChain); + registry.add_agreement(old_id, IndexingAgreementStatus::AcceptedOnChain); + registry.add_pending_cancellation(new_id, old_id); + + let result = execute_pending_cancellations( + &new_id, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await; + + assert!(result.is_ok(), "got {result:?}"); + assert!(registry.was_marked_cancelling(&old_id)); + assert!(!registry.was_marked_canceled_by_requester(&old_id)); + assert!(!registry.was_cancel_audit_recorded(&old_id)); + assert!(registry.was_pending_cancellation_deleted(&new_id, &old_id)); } #[tokio::test] @@ -3481,9 +3638,10 @@ mod tests { ) .await; - // The chain cancel failed, the row is NOT marked canceled, and no cancel - // audit is recorded (so the sweep emits nothing). + // The row couldn't be marked, so no cancel went out, nothing is recorded (the + // sweep emits nothing), and the pending row stays for a retry. assert!(result.is_err()); + assert!(!chain_client.was_on_chain_cancel_attempted(&old_fail)); assert!(!registry.was_marked_canceled_by_requester(&old_fail)); assert!(!registry.was_cancel_audit_recorded(&old_fail)); } @@ -4355,6 +4513,27 @@ mod tests { ); } + #[tokio::test] + async fn test_orphan_sweep_leaves_a_failed_cancel_to_the_cancel_retry() { + // Retried by this sweep, a cancel that never works was resent with no limit. + let registry = MockRegistry::new(); + let chain_client = MockChainClient { + fail_cancels: true, + ..MockChainClient::default() + }; + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + let request_id = IndexingRequestId::new(); + registry.add_agreement(agreement_id, IndexingAgreementStatus::AcceptedOnChain); + registry.set_agreement_request_id(agreement_id, request_id); + registry.mark_request_canceled(request_id); + + sweep_orphan_canceled_agreements(®istry, &chain_client, test_agreement_conf().as_ref()) + .await; + + assert!(registry.was_marked_cancelling(&agreement_id)); + assert!(!registry.was_marked_canceled_by_requester(&agreement_id)); + } + #[tokio::test] async fn test_orphan_sweep_records_cancel_audit() { // The orphan sweep no longer emits `terminated` directly; it records the diff --git a/bin/dipper-service/src/network/service/liveness_checker.rs b/bin/dipper-service/src/network/service/liveness_checker.rs index 09716aba..34e43ac2 100644 --- a/bin/dipper-service/src/network/service/liveness_checker.rs +++ b/bin/dipper-service/src/network/service/liveness_checker.rs @@ -1216,29 +1216,7 @@ mod tests { /// Default agreement config for the cancel-path tests. fn test_agreement_conf() -> crate::config::IndexingAgreementConfig { - crate::config::IndexingAgreementConfig { - data_service: thegraph_core::alloy::primitives::Address::ZERO, - recurring_collector: thegraph_core::alloy::primitives::Address::ZERO, - recurring_agreement_manager: thegraph_core::alloy::primitives::Address::ZERO, - max_agreement_grt_per_30_days: 0.0, - max_seconds_per_collection: 0, - min_seconds_per_collection: 0, - duration_seconds: 0, - deadline_seconds: 0, - max_grt_per_30_days: std::collections::BTreeMap::new(), - max_grt_per_billion_entities_per_30_days: 0.0, - declined_indexer_lookback_days: 0, - price_rejection_lookback_days: 0, - transient_rejection_lookback_minutes: 0, - uncertain_rejection_lookback_days: 0, - unresponsive_indexer_lookback_days: 0, - mass_unresponsive_trip_fraction: 0.5, - mass_unresponsive_reset_fraction: 0.25, - dips_accepting_snapshot_max_age_hours: 48, - dips_accepting_cache_ttl_seconds: 300, - max_in_flight_offers_per_indexer: None, - max_in_flight_offers_total: None, - } + crate::config::IndexingAgreementConfig::for_tests() } // ---- Pure function tests ---- diff --git a/bin/dipper-service/src/registry.rs b/bin/dipper-service/src/registry.rs index 82d9baf6..7d30552c 100644 --- a/bin/dipper-service/src/registry.rs +++ b/bin/dipper-service/src/registry.rs @@ -23,8 +23,8 @@ pub use self::agreement_stub::StubAgreementRegistry; use self::result::Result as RegistryResult; pub use self::{ agreement::{ - AgreementFeeRate, AgreementRegistry, CancelKind, IndexingAgreement, NewAgreementParams, - ReconciliationAudit, ReconciliationItem, ReconciliationOutcome, + AgreementFeeRate, AgreementRegistry, CancelKind, CancellingAgreement, IndexingAgreement, + NewAgreementParams, ReconciliationAudit, ReconciliationItem, ReconciliationOutcome, Status as IndexingAgreementStatus, Terms as IndexingAgreementTerms, TermsMetadata as IndexingAgreementTermsMetadata, }, @@ -391,6 +391,43 @@ impl AgreementRegistry for RegistryProvider { .map_err(Into::into) } + async fn mark_indexing_agreement_as_cancelling( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()> { + self.inner + .mark_indexing_agreement_as_cancelling(id) + .await + .map_err(Into::into) + } + + async fn get_cancelling_agreements( + &self, + batch_size: i64, + max_attempts: u32, + min_age_minutes: i32, + ) -> RegistryResult> { + Ok(self + .inner + .get_cancelling_agreements(batch_size, max_attempts, min_age_minutes) + .await? + .into_iter() + .map(CancellingAgreement::try_from) + .filter_map(filter_map_with_logging) + .collect()) + } + + async fn record_cancel_check( + &self, + id: &IndexingAgreementId, + failed_attempts: u32, + ) -> RegistryResult { + self.inner + .record_cancel_check(id, failed_attempts) + .await + .map_err(Into::into) + } + async fn apply_reconciliation( &self, id: &IndexingAgreementId, diff --git a/bin/dipper-service/src/registry/agreement.rs b/bin/dipper-service/src/registry/agreement.rs index 32227b2c..36d2c9ae 100644 --- a/bin/dipper-service/src/registry/agreement.rs +++ b/bin/dipper-service/src/registry/agreement.rs @@ -259,13 +259,37 @@ pub trait AgreementRegistry { /// Mark an indexing agreement as `CANCELED_BY_REQUESTER`. /// /// If there is no indexing agreement with the given ID, or if the agreement is not in the - /// `CREATED` or `ACCEPTED_ON_CHAIN` state, this method returns a + /// `CREATED`, `ACCEPTED_ON_CHAIN`, `REJECTED` or `CANCELLING` state, this method returns a /// [`NoRecordUpdated`](Error::NoRecordsUpdated) error. async fn mark_indexing_agreement_as_canceled_by_requester( &self, id: &IndexingAgreementId, ) -> RegistryResult<()>; + /// Mark a `CREATED`, `ACCEPTED_ON_CHAIN`, `REJECTED` or `EXPIRED` agreement `CANCELLING`, + /// before its cancel is sent; [`NoRecordUpdated`](Error::NoRecordsUpdated) otherwise. + async fn mark_indexing_agreement_as_cancelling( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()>; + + /// `CANCELLING` agreements marked over `min_age_minutes` ago whose cancel has failed + /// fewer than `max_attempts` times, those checked longest ago first. + async fn get_cancelling_agreements( + &self, + batch_size: i64, + max_attempts: u32, + min_age_minutes: i32, + ) -> RegistryResult>; + + /// Record a check of a `CANCELLING` agreement that left it cancelling, adding + /// `failed_attempts` to its failed cancels and returning the new count. + async fn record_cancel_check( + &self, + id: &IndexingAgreementId, + failed_attempts: u32, + ) -> RegistryResult; + /// Apply a reconciliation-driven state transition atomically. /// /// Used by `chain_listener::reconcile_agreement` so the @@ -522,6 +546,25 @@ pub struct AgreementFeeRate { pub tokens_per_entity_per_second: f64, } +/// An agreement dipper is still cancelling on-chain. +#[derive(Debug, Clone)] +pub struct CancellingAgreement { + pub agreement: IndexingAgreement, + /// Whether dipper saw it accepted on-chain, so its end is announced. + pub accepted_on_chain: bool, +} + +impl TryFrom for CancellingAgreement { + type Error = anyhow::Error; + + fn try_from(value: dipper_pgregistry::CancellingAgreement) -> Result { + Ok(Self { + agreement: value.agreement.try_into()?, + accepted_on_chain: value.accepted_on_chain, + }) + } +} + /// An Indexing Agreement represents the contract between the DIPs Gateway (Dipper) and the indexer /// to index the data. /// @@ -689,6 +732,11 @@ pub enum Status { /// /// This is a terminal state. AbandonedByIndexer, + + /// Dipper decided to end the agreement and is cancelling it on-chain, where it may + /// still be live. It becomes `CanceledByRequester`, announced as ended, only once + /// the chain confirms the end. + Cancelling, } impl std::fmt::Display for Status { @@ -702,6 +750,7 @@ impl std::fmt::Display for Status { Status::AcceptedOnChain => "ACCEPTED_ON_CHAIN", Status::Rejected => "REJECTED", Status::AbandonedByIndexer => "ABANDONED_BY_INDEXER", + Status::Cancelling => "CANCELLING", }; f.write_str(status) } @@ -733,6 +782,7 @@ impl TryFrom for IndexingAgreement { dipper_pgregistry::IndexingAgreementStatus::AbandonedByIndexer => { Status::AbandonedByIndexer } + dipper_pgregistry::IndexingAgreementStatus::Cancelling => Status::Cancelling, _ => { return Err(anyhow::anyhow!("Invalid status: {:?}", value.status)); } diff --git a/bin/dipper-service/src/registry/agreement_stub.rs b/bin/dipper-service/src/registry/agreement_stub.rs index ec9bfd3b..3e7477be 100644 --- a/bin/dipper-service/src/registry/agreement_stub.rs +++ b/bin/dipper-service/src/registry/agreement_stub.rs @@ -9,9 +9,9 @@ use thegraph_core::{DeploymentId, IndexerId, alloy::primitives::ChainId}; use super::{ agreement::{ - AgreementFeeRate, AgreementRegistry, CancelKind, IndexingAgreement, NewAgreementParams, - PendingAcceptedEvent, PendingExpiredEvent, PendingTerminatedEvent, ReconciliationItem, - ReconciliationOutcome, + AgreementFeeRate, AgreementRegistry, CancelKind, CancellingAgreement, IndexingAgreement, + NewAgreementParams, PendingAcceptedEvent, PendingExpiredEvent, PendingTerminatedEvent, + ReconciliationItem, ReconciliationOutcome, }, result::Result, }; @@ -133,6 +133,27 @@ pub trait StubAgreementRegistry: Send + Sync { unimplemented!("mark_indexing_agreement_as_canceled_by_requester") } + async fn mark_indexing_agreement_as_cancelling(&self, _id: &IndexingAgreementId) -> Result<()> { + unimplemented!("mark_indexing_agreement_as_cancelling") + } + + async fn get_cancelling_agreements( + &self, + _batch_size: i64, + _max_attempts: u32, + _min_age_minutes: i32, + ) -> Result> { + Ok(Vec::new()) + } + + async fn record_cancel_check( + &self, + _id: &IndexingAgreementId, + failed_attempts: u32, + ) -> Result { + Ok(failed_attempts) + } + async fn apply_reconciliation( &self, _id: &IndexingAgreementId, @@ -415,6 +436,33 @@ impl AgreementRegistry for T { StubAgreementRegistry::mark_indexing_agreement_as_canceled_by_requester(self, id).await } + async fn mark_indexing_agreement_as_cancelling(&self, id: &IndexingAgreementId) -> Result<()> { + StubAgreementRegistry::mark_indexing_agreement_as_cancelling(self, id).await + } + + async fn get_cancelling_agreements( + &self, + batch_size: i64, + max_attempts: u32, + min_age_minutes: i32, + ) -> Result> { + StubAgreementRegistry::get_cancelling_agreements( + self, + batch_size, + max_attempts, + min_age_minutes, + ) + .await + } + + async fn record_cancel_check( + &self, + id: &IndexingAgreementId, + failed_attempts: u32, + ) -> Result { + StubAgreementRegistry::record_cancel_check(self, id, failed_attempts).await + } + async fn apply_reconciliation( &self, id: &IndexingAgreementId, diff --git a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs index c135defb..46ae9be2 100644 --- a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs +++ b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs @@ -1,6 +1,6 @@ -//! Cancel on-chain, via the RecurringAgreementManager, an agreement dipper doesn't -//! want that is still live: one the indexer rejected off-chain, or one dipper had already -//! cancelled. The chain listener queues it, and a reassessment does when its own cancel fails. +//! Cancel on-chain, via the RecurringAgreementManager, an agreement dipper doesn't want +//! that was accepted anyway: one the indexer rejected off-chain, or one dipper had already +//! cancelled. The chain listener queues it. use std::{ collections::HashSet, @@ -11,7 +11,7 @@ use std::{ use dipper_core::ids::IndexingAgreementId; use crate::{ - cancel_dispatch::cancel_agreement_on_chain, + cancel_dispatch::{LiveCancel, cancel_agreement_on_chain, cancel_if_live}, chain_client::{ChainClient, ChainClientError}, config::IndexingAgreementConfig, registry::{AgreementRegistry, IndexingAgreement, IndexingAgreementStatus}, @@ -148,7 +148,7 @@ where } /// Agreements a job in this process is cancelling right now. The listener can -/// queue one twice before the chain shows it ended, and two jobs at once would +/// queue one twice before the chain shows it ended, and 2 jobs at once would /// both cancel it and both alert. static CANCELLING: LazyLock>> = LazyLock::new(Mutex::default); @@ -173,9 +173,9 @@ impl Drop for Cancelling { } } -/// Cancel on-chain an agreement dipper had already cancelled that is still live: accepted -/// anyway, or an offer a failed cancel left open. The row is already terminal. The chain is -/// read first, so a stale snapshot of one dipper has since ended raises no alert. +/// Cancel on-chain an agreement dipper had already cancelled that the indexer accepted +/// anyway; the row is already terminal. The chain is read first, so a stale snapshot of +/// one dipper has since ended raises no alert. async fn cancel_live_agreement_dipper_cancelled( ctx: &Ctx, agreement: &IndexingAgreement, @@ -190,25 +190,32 @@ where ); return Ok(()); }; - if !live_on_chain(&ctx.chain_client, agreement).await? { - tracing::info!( - agreement_id = %agreement.id, - "Cancelled agreement is no longer live on-chain; nothing to cancel" - ); - return Ok(()); - } - - match cancel_agreement_on_chain(&ctx.chain_client, agreement, &ctx.agreement_conf).await { - Ok(tx_hash) => { + match cancel_if_live(&ctx.chain_client, agreement, &ctx.agreement_conf).await { + LiveCancel::NotLive => { + tracing::info!( + agreement_id = %agreement.id, + "Cancelled agreement is no longer live on-chain; nothing to cancel" + ); + Ok(()) + } + LiveCancel::ReadFailed(err) => { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to read whether a cancelled agreement is live on-chain, will retry" + ); + Err(JobError::Retryable(err.into(), Duration::from_secs(30))) + } + LiveCancel::Ended(tx_hash) => { let tx = tx_hash.map_or_else(|| "none".to_owned(), |hash| hash.to_string()); log_caught_live_agreement(agreement, "cancelled", &tx); Ok(()) } - Err(err @ ChainClientError::MissingTermsVersionHash { .. }) => { + LiveCancel::CancelFailed(err @ ChainClientError::MissingTermsVersionHash { .. }) => { log_caught_live_agreement(agreement, "cancel_impossible", &err.to_string()); Err(JobError::Fatal(err.into())) } - Err(err) => { + LiveCancel::CancelFailed(err) => { log_caught_live_agreement(agreement, "cancel_failed", &err.to_string()); Err(JobError::Retryable(err.into(), Duration::from_secs(30))) } @@ -230,24 +237,6 @@ fn log_caught_live_agreement(agreement: &IndexingAgreement, outcome: &str, detai ); } -/// Read the chain for whether the agreement is still live; a failed read retries. -async fn live_on_chain( - chain_client: &T, - agreement: &IndexingAgreement, -) -> JobResult { - chain_client - .agreement_still_active(agreement.id.as_bytes()) - .await - .map_err(|err| { - tracing::warn!( - agreement_id = %agreement.id, - error = %err, - "Failed to read whether a cancelled agreement is live on-chain, will retry" - ); - JobError::Retryable(err.into(), Duration::from_secs(30)) - }) -} - /// Flip the row to CanceledByRequester once the chain shows it cancelled. A failure /// is logged, not fatal: the chain is already right and the listener retries the /// DB update. Returns whether the row is now terminal, gating the cancel audit. @@ -474,6 +463,30 @@ mod tests { Ok(()) } + async fn mark_indexing_agreement_as_cancelling( + &self, + _id: &IndexingAgreementId, + ) -> crate::registry::Result<()> { + Ok(()) + } + + async fn get_cancelling_agreements( + &self, + _batch_size: i64, + _max_attempts: u32, + _min_age_minutes: i32, + ) -> crate::registry::Result> { + Ok(Vec::new()) + } + + async fn record_cancel_check( + &self, + _id: &IndexingAgreementId, + failed_attempts: u32, + ) -> crate::registry::Result { + Ok(failed_attempts) + } + async fn apply_reconciliation( &self, _id: &IndexingAgreementId, @@ -629,29 +642,7 @@ mod tests { } fn test_agreement_conf() -> Arc { - Arc::new(IndexingAgreementConfig { - data_service: Address::ZERO, - recurring_collector: Address::ZERO, - recurring_agreement_manager: Address::ZERO, - max_agreement_grt_per_30_days: 0.0, - max_seconds_per_collection: 0, - min_seconds_per_collection: 0, - duration_seconds: 0, - deadline_seconds: 0, - max_grt_per_30_days: std::collections::BTreeMap::new(), - max_grt_per_billion_entities_per_30_days: 0.0, - declined_indexer_lookback_days: 0, - price_rejection_lookback_days: 0, - transient_rejection_lookback_minutes: 0, - uncertain_rejection_lookback_days: 0, - unresponsive_indexer_lookback_days: 0, - mass_unresponsive_trip_fraction: 0.5, - mass_unresponsive_reset_fraction: 0.25, - dips_accepting_snapshot_max_age_hours: 48, - dips_accepting_cache_ttl_seconds: 300, - max_in_flight_offers_per_indexer: None, - max_in_flight_offers_total: None, - }) + Arc::new(IndexingAgreementConfig::for_tests()) } fn test_deployment_id() -> DeploymentId { @@ -848,7 +839,7 @@ mod tests { #[tokio::test] async fn a_second_job_for_the_same_agreement_leaves_it_to_the_first() { // The listener can queue an agreement twice before the chain shows it - // ended; two jobs at once would both cancel it and both alert. + // ended; 2 jobs at once would both cancel it and both alert. let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); let agreement_id = agreement.id; let chain = MockChainClient::live(); diff --git a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs index 0fb5bfa7..340bf26b 100644 --- a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs +++ b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs @@ -634,122 +634,22 @@ where } // Cancel old agreements that have no replacement to pair with: these indexers - // leave the target group with nothing taking their place. An accepted one is - // cancelled on-chain before it is marked, so a failed cancel leaves it for a retry. + // leave the target group with nothing taking their place. let mut directly_cancelled = 0u32; let mut cancel_failures = 0u32; for old_agreement in old_iter { - let was_accepted = matches!( - old_agreement.status, - crate::registry::IndexingAgreementStatus::AcceptedOnChain - ); - // A Created agreement's offer may already be on-chain, so it is cancelled - // on-chain too; without a stored terms hash it can only be marked locally. - let needs_on_chain_cancel = was_accepted - || (old_agreement.status == crate::registry::IndexingAgreementStatus::Created - && old_agreement.terms_version_hash.is_some()); - // An unaccepted one is marked first. An offer in flight then lands before the - // cancel, which withdraws it, or after, when its job sees the mark and withdraws it. - if !was_accepted && !mark_unpaired_cancelled(&ctx.registry, old_agreement).await { - cancel_failures += 1; - continue; - } - - let mut on_chain_cancel_tx: Option = None; - if needs_on_chain_cancel { - match crate::cancel_dispatch::cancel_agreement_on_chain( - &ctx.chain_client, - old_agreement, - &ctx.agreement_conf, - ) - .await - { - Ok(Some(tx_hash)) => { - tracing::info!( - agreement_id = %old_agreement.id, - indexing_request_id = %indexing_request_id, - %tx_hash, - "Submitted on-chain cancellation for unpaired old agreement" - ); - on_chain_cancel_tx = Some(tx_hash.to_string()); - } - Ok(None) => { - tracing::info!( - agreement_id = %old_agreement.id, - indexing_request_id = %indexing_request_id, - "Unpaired old agreement already canceled on-chain; proceeding with local cleanup" - ); - } - Err(err) => { - tracing::warn!( - error = %err, - agreement_id = %old_agreement.id, - was_accepted, - "On-chain cancel failed; an accepted agreement is retried later, an \ - unaccepted one by a queued cancel job" - ); - cancel_failures += 1; - if was_accepted { - continue; - } - if let Err(err) = ctx - .queue - .cancel_rejected_agreement_on_chain( - old_agreement.id, - JobPriority::Background, - ) - .await - { - tracing::error!( - error = %err, - agreement_id = %old_agreement.id, - "Failed to queue a retry of the on-chain cancel; an offer already \ - on-chain stays open" - ); - } - } - } - } - - if was_accepted && !mark_unpaired_cancelled(&ctx.registry, old_agreement).await { + let Some(new_status) = cancel_unpaired(&ctx, old_agreement).await else { cancel_failures += 1; continue; - } - + }; tracing::info!( agreement_id = %old_agreement.id, indexing_request_id = %indexing_request_id, old_status = %old_agreement.status, - new_status = "CANCELED_BY_REQUESTER", + new_status, reason = "reassessment_not_in_target_group", "agreement state transition" ); - - // Record the cancel audit for the accepted-on-chain agreements dipper just - // cancelled, so the chain_listener's `terminated` sweep announces them - // durably. Never-accepted agreements (`!was_accepted`) were never live - // on-chain: they are not sweep-eligible (`accepted_at IS NULL`) and - // correctly emit nothing. - if was_accepted { - let manager = ctx.agreement_conf.recurring_agreement_manager().to_string(); - if let Err(err) = ctx - .registry - .record_cancel_audit( - &old_agreement.id, - now_secs(), - &manager, - on_chain_cancel_tx.as_deref(), - ) - .await - { - tracing::warn!( - agreement_id = %old_agreement.id, - error = %err, - "failed to record cancel audit; terminated event may emit with fallback fields" - ); - } - } - directly_cancelled += 1; } @@ -762,16 +662,15 @@ where } if cancel_failures > 0 { - // An unaccepted agreement's failed cancel was queued as its own job above. An - // accepted one stays AcceptedOnChain: the orphan-cancel sweep retries it once the - // request is Canceled, the periodic reassignment service (24 h) while it is Open. + // An agreement that couldn't be marked keeps its status: the orphan-cancel sweep + // retries it once the request is Canceled, the periodic reassignment service + // (24 h) while it is Open. tracing::warn!( indexing_request_id=%indexing_request_id, failures=cancel_failures, - "some agreement cancels failed during reassessment; retry will fire \ - via a queued cancel job (unaccepted agreements), the orphan-cancel \ - sweep (Canceled requests) or the periodic reassignment service \ - (Open requests over-target)" + "some agreements could not be marked for cancelling during reassessment; retry \ + will fire via the orphan-cancel sweep (Canceled requests) or the periodic \ + reassignment service (Open requests over-target)" ); } @@ -1062,8 +961,8 @@ mod lifecycle_event_tests { #[derive(Default, Clone)] struct MockChainClient { cancelled: Arc>>, - /// When set, each cancel records whether its agreement was already marked. - marked_cancelled: Option>>>, + /// When set, each cancel records whether its agreement was already marked cancelling. + marked_cancelling: Option>>>, marked_at_cancel: Arc>>, fail_cancel: bool, } @@ -1094,7 +993,7 @@ mod lifecycle_event_tests { return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); } self.cancelled.lock().unwrap().push(*agreement_id); - if let Some(marked) = &self.marked_cancelled { + if let Some(marked) = &self.marked_cancelling { let id = IndexingAgreementId::from_bytes(*agreement_id); let was_marked = marked.lock().unwrap().contains(&id); self.marked_at_cancel.lock().unwrap().push(was_marked); @@ -1144,6 +1043,7 @@ mod lifecycle_event_tests { shortfall_active: std::sync::Mutex, /// Ids marked CanceledByRequester locally. marked_cancelled: Arc>>, + marked_cancelling: Arc>>, } #[async_trait] @@ -1295,6 +1195,30 @@ mod lifecycle_event_tests { self.marked_cancelled.lock().unwrap().push(*id); Ok(()) } + async fn mark_indexing_agreement_as_cancelling( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()> { + self.marked_cancelling.lock().unwrap().push(*id); + Ok(()) + } + + async fn get_cancelling_agreements( + &self, + _batch_size: i64, + _max_attempts: u32, + _min_age_minutes: i32, + ) -> RegistryResult> { + Ok(Vec::new()) + } + + async fn record_cancel_check( + &self, + _id: &IndexingAgreementId, + failed_attempts: u32, + ) -> RegistryResult { + Ok(failed_attempts) + } async fn apply_reconciliation( &self, _id: &IndexingAgreementId, @@ -1449,19 +1373,11 @@ mod lifecycle_event_tests { min_seconds_per_collection: 60, duration_seconds: 86400, deadline_seconds: 3600, - max_grt_per_30_days: std::collections::BTreeMap::new(), - max_grt_per_billion_entities_per_30_days: 0.0, declined_indexer_lookback_days: 30, price_rejection_lookback_days: 1, transient_rejection_lookback_minutes: 30, uncertain_rejection_lookback_days: 1, - unresponsive_indexer_lookback_days: 0, - mass_unresponsive_trip_fraction: 0.5, - mass_unresponsive_reset_fraction: 0.25, - dips_accepting_snapshot_max_age_hours: 48, - dips_accepting_cache_ttl_seconds: 300, - max_in_flight_offers_per_indexer: None, - max_in_flight_offers_total: None, + ..IndexingAgreementConfig::for_tests() } } @@ -1700,6 +1616,7 @@ mod lifecycle_event_tests { // Already in shortfall. shortfall_active: std::sync::Mutex::new(true), marked_cancelled: Arc::default(), + marked_cancelling: Arc::default(), }; let ctx = build_ctx( registry, @@ -1737,6 +1654,7 @@ mod lifecycle_event_tests { chain_state_lookup_fails: false, shortfall_active: std::sync::Mutex::new(false), marked_cancelled: Arc::default(), + marked_cancelling: Arc::default(), }; let ctx = build_ctx( registry, @@ -1808,6 +1726,7 @@ mod lifecycle_event_tests { chain_state_lookup_fails: false, shortfall_active: std::sync::Mutex::new(false), marked_cancelled: Arc::default(), + marked_cancelling: Arc::default(), }; let ctx = build_ctx( registry, @@ -1862,6 +1781,7 @@ mod lifecycle_event_tests { chain_state_lookup_fails: false, shortfall_active: std::sync::Mutex::new(false), marked_cancelled: Arc::default(), + marked_cancelling: Arc::default(), }; let ctx = build_ctx( registry, @@ -1927,6 +1847,7 @@ mod lifecycle_event_tests { chain_state_lookup_fails: false, shortfall_active: std::sync::Mutex::new(false), marked_cancelled: Arc::default(), + marked_cancelling: Arc::default(), }; let ctx = build_ctx( registry, @@ -1950,8 +1871,8 @@ mod lifecycle_event_tests { assert!(matches!(captured[0], CapturedEvent::Proposed { .. })); } - /// Ctx whose only active agreement, in `status`, leaves the target group, with - /// the chain mock recording whether the row was already marked at each cancel. + /// Ctx whose only active agreement, in `status`, leaves the target group, with the + /// chain mock recording whether the row was already marked cancelling at each cancel. fn ctx_cancelling_one( status: IndexingAgreementStatus, ) -> ( @@ -1970,16 +1891,18 @@ mod lifecycle_event_tests { CapturingEventsProducer::new(), indexer_urls::Snapshot::new(), ); - ctx.chain_client.marked_cancelled = Some(ctx.registry.marked_cancelled.clone()); + ctx.chain_client.marked_cancelling = Some(ctx.registry.marked_cancelling.clone()); (ctx, leaving) } #[tokio::test] - async fn marks_an_unaccepted_agreement_cancelled_before_cancelling_it_on_chain() { + async fn marks_an_unaccepted_agreement_cancelling_before_cancelling_it_on_chain() { // Its offer may be in flight. An offer landing after the cancel can't be - // withdrawn by it, so its job must find the agreement already marked. + // withdrawn by it, so its job must find the agreement already marked. It + // stays cancelling: the offer could still be accepted until its deadline. let (ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::Created); let chain = ctx.chain_client.clone(); + let cancelled = ctx.registry.marked_cancelled.clone(); let result = handle(ctx, &test_message(0)).await; @@ -1989,50 +1912,43 @@ mod lifecycle_event_tests { vec![*leaving.id.as_bytes()] ); assert_eq!(*chain.marked_at_cancel.lock().unwrap(), vec![true]); + assert!(cancelled.lock().unwrap().is_empty()); } #[tokio::test] - async fn cancels_an_accepted_agreement_on_chain_before_marking_it() { + async fn marks_an_accepted_agreement_cancelled_once_its_cancel_lands() { let (ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::AcceptedOnChain); let chain = ctx.chain_client.clone(); - let marked = ctx.registry.marked_cancelled.clone(); + let cancelled = ctx.registry.marked_cancelled.clone(); let result = handle(ctx, &test_message(0)).await; assert!(result.is_ok(), "got {result:?}"); - assert_eq!(*chain.marked_at_cancel.lock().unwrap(), vec![false]); - assert_eq!(*marked.lock().unwrap(), vec![leaving.id]); - } - - #[tokio::test] - async fn an_unaccepted_agreement_whose_cancel_fails_is_marked_and_its_cancel_queued() { - // Nothing else would send its cancel again: the row is already terminal, - // and its offer job may have finished before the mark. - let (mut ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::Created); - ctx.chain_client.fail_cancel = true; - let marked = ctx.registry.marked_cancelled.clone(); - let queue = ctx.queue.clone(); - - let result = handle(ctx, &test_message(0)).await; - - assert!(result.is_ok(), "got {result:?}"); - assert_eq!(*marked.lock().unwrap(), vec![leaving.id]); - assert_eq!(*queue.cancels_queued.lock().unwrap(), vec![leaving.id]); + assert_eq!(*chain.marked_at_cancel.lock().unwrap(), vec![true]); + assert_eq!(*cancelled.lock().unwrap(), vec![leaving.id]); } #[tokio::test] - async fn an_accepted_agreement_whose_cancel_fails_stays_for_a_retry() { - // Marking it cancelled would leave it live with nothing to end it. - let (mut ctx, _leaving) = ctx_cancelling_one(IndexingAgreementStatus::AcceptedOnChain); - ctx.chain_client.fail_cancel = true; - let marked = ctx.registry.marked_cancelled.clone(); - let queue = ctx.queue.clone(); + async fn an_agreement_whose_cancel_fails_is_left_cancelling_for_the_listener() { + // The chain listener retries the cancel of every cancelling agreement until + // the chain shows it ended, so nothing is queued here. + for status in [ + IndexingAgreementStatus::Created, + IndexingAgreementStatus::AcceptedOnChain, + ] { + let (mut ctx, leaving) = ctx_cancelling_one(status); + ctx.chain_client.fail_cancel = true; + let cancelling = ctx.registry.marked_cancelling.clone(); + let cancelled = ctx.registry.marked_cancelled.clone(); + let queue = ctx.queue.clone(); - let result = handle(ctx, &test_message(0)).await; + let result = handle(ctx, &test_message(0)).await; - assert!(result.is_ok(), "got {result:?}"); - assert!(marked.lock().unwrap().is_empty()); - assert!(queue.cancels_queued.lock().unwrap().is_empty()); + assert!(result.is_ok(), "got {result:?}"); + assert_eq!(*cancelling.lock().unwrap(), vec![leaving.id]); + assert!(cancelled.lock().unwrap().is_empty()); + assert!(queue.cancels_queued.lock().unwrap().is_empty()); + } } #[tokio::test] @@ -2049,6 +1965,7 @@ mod lifecycle_event_tests { chain_state_lookup_fails: false, shortfall_active: std::sync::Mutex::new(false), marked_cancelled: Arc::default(), + marked_cancelling: Arc::default(), }; let ctx = build_ctx( registry, @@ -2065,6 +1982,46 @@ mod lifecycle_event_tests { } } +/// Take an old agreement out of the target group, returning its new status, or `None` +/// (logged) when it couldn't be marked. One that may be live on-chain is cancelled there +/// too, which the chain listener retries until it ends. +async fn cancel_unpaired( + ctx: &Ctx, + agreement: &crate::registry::IndexingAgreement, +) -> Option<&'static str> +where + R: AgreementRegistry + Sync, + T: ChainClient, +{ + let may_be_live = agreement.status == crate::registry::IndexingAgreementStatus::AcceptedOnChain + || (agreement.status == crate::registry::IndexingAgreementStatus::Created + && agreement.terms_version_hash.is_some()); + if !may_be_live { + return mark_unpaired_cancelled(&ctx.registry, agreement) + .await + .then_some("CANCELED_BY_REQUESTER"); + } + match crate::cancel_dispatch::start_cancel( + &ctx.registry, + &ctx.chain_client, + agreement, + &ctx.agreement_conf, + ) + .await + { + Ok(crate::cancel_dispatch::CancelStarted::Ended) => Some("CANCELED_BY_REQUESTER"), + Ok(crate::cancel_dispatch::CancelStarted::Cancelling) => Some("CANCELLING"), + Err(err) => { + tracing::error!( + error = %err, + agreement_id = %agreement.id, + "Failed to mark unpaired old agreement as cancelling in local DB" + ); + None + } + } +} + /// Mark an unpaired old agreement CanceledByRequester; false, logged, if that fails. async fn mark_unpaired_cancelled( registry: &R, diff --git a/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs b/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs index 1aa582ac..af9cbf83 100644 --- a/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs +++ b/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs @@ -522,6 +522,30 @@ mod tests { Ok(()) } + async fn mark_indexing_agreement_as_cancelling( + &self, + _id: &IndexingAgreementId, + ) -> crate::registry::Result<()> { + Ok(()) + } + + async fn get_cancelling_agreements( + &self, + _batch_size: i64, + _max_attempts: u32, + _min_age_minutes: i32, + ) -> crate::registry::Result> { + Ok(Vec::new()) + } + + async fn record_cancel_check( + &self, + _id: &IndexingAgreementId, + failed_attempts: u32, + ) -> crate::registry::Result { + Ok(failed_attempts) + } + async fn apply_reconciliation( &self, _id: &IndexingAgreementId, diff --git a/bin/dipper-service/src/worker/handlers/submit_offer.rs b/bin/dipper-service/src/worker/handlers/submit_offer.rs index 172ddcdd..e27f3d74 100644 --- a/bin/dipper-service/src/worker/handlers/submit_offer.rs +++ b/bin/dipper-service/src/worker/handlers/submit_offer.rs @@ -23,7 +23,7 @@ use thegraph_core::{DeploymentId, alloy::primitives::ChainId}; use url::Url; use crate::{ - cancel_dispatch::cancel_agreement_on_chain, + cancel_dispatch::{LiveCancel, cancel_if_live}, chain_client::{ChainClient, ChainClientError, decode_revert_reason}, config::IndexingAgreementConfig, indexer_rpc_client::into_sol_rca, @@ -213,9 +213,7 @@ async fn next_step( NextStep::Skip } Some(a) if a.status == IndexingAgreementStatus::Created => NextStep::Offer(a), - Some(a) if a.status == IndexingAgreementStatus::CanceledByRequester => { - NextStep::Withdraw(a) - } + Some(a) if dipper_cancelled(a.status) => NextStep::Withdraw(a), Some(a) => { tracing::warn!( agreement_id = %agreement_id, @@ -234,16 +232,9 @@ async fn withdraw_offer_if_stored( ctx: &Ctx, agreement: &IndexingAgreement, ) -> JobResult<()> { - let stored = ctx - .chain_client - .agreement_still_active(agreement.id.as_bytes()) - .await - .map_err(|err| retry_withdraw(agreement, err))?; - if !stored { - return Ok(()); - } - match cancel_agreement_on_chain(&ctx.chain_client, agreement, &ctx.agreement_conf).await { - Ok(tx_hash) => { + match cancel_if_live(&ctx.chain_client, agreement, &ctx.agreement_conf).await { + LiveCancel::NotLive => Ok(()), + LiveCancel::Ended(tx_hash) => { tracing::info!( agreement_id = %agreement.id, tx_hash = ?tx_hash, @@ -251,7 +242,7 @@ async fn withdraw_offer_if_stored( ); Ok(()) } - Err(err @ ChainClientError::MissingTermsVersionHash { .. }) => { + LiveCancel::CancelFailed(err @ ChainClientError::MissingTermsVersionHash { .. }) => { tracing::error!( agreement_id = %agreement.id, error = %err, @@ -259,7 +250,9 @@ async fn withdraw_offer_if_stored( ); Err(JobError::Fatal(err.into())) } - Err(err) => Err(retry_withdraw(agreement, err)), + LiveCancel::ReadFailed(err) | LiveCancel::CancelFailed(err) => { + Err(retry_withdraw(agreement, err)) + } } } @@ -272,22 +265,30 @@ fn retry_withdraw(agreement: &IndexingAgreement, err: ChainClientError) -> JobEr JobError::Retryable(err.into(), TRANSIENT_RETRY_BASE) } -/// The agreement, if it was cancelled after this job's status check. +/// Whether dipper has cancelled the agreement, or started to. +fn dipper_cancelled(status: IndexingAgreementStatus) -> bool { + matches!( + status, + IndexingAgreementStatus::Cancelling | IndexingAgreementStatus::CanceledByRequester + ) +} + +/// The agreement, if it was cancelled after this job's status check. After a failed read +/// the job still finishes: retrying would send the offer again, and the chain listener's +/// cancel retry withdraws the offer of an agreement left cancelling. async fn cancelled_meanwhile( registry: &R, agreement_id: &IndexingAgreementId, ) -> Option { match registry.get_indexing_agreement_by_id(agreement_id).await { - Ok(Some(agreement)) if agreement.status == IndexingAgreementStatus::CanceledByRequester => { - Some(agreement) - } + Ok(Some(agreement)) if dipper_cancelled(agreement.status) => Some(agreement), Ok(_) => None, Err(err) => { tracing::warn!( agreement_id = %agreement_id, error = %err, - "Failed to re-read agreement after its offer landed; a cancel made meanwhile \ - would leave the offer open until its deadline" + "Failed to re-read agreement after its offer landed; the cancel retry \ + withdraws the offer if it was cancelled" ); None } @@ -298,7 +299,7 @@ async fn cancelled_meanwhile( mod tests { use std::sync::{ Arc, Mutex, - atomic::{AtomicBool, Ordering}, + atomic::{AtomicBool, AtomicU32, Ordering}, }; use async_trait::async_trait; @@ -321,6 +322,9 @@ mod tests { struct MockRegistry { agreement: SharedAgreement, + /// When set, every read after the first fails. + later_reads_fail: bool, + reads: AtomicU32, } #[async_trait] @@ -329,6 +333,9 @@ mod tests { &self, _id: &IndexingAgreementId, ) -> crate::registry::Result> { + if self.reads.fetch_add(1, Ordering::SeqCst) > 0 && self.later_reads_fail { + return Err(crate::registry::Error::NoRecordsUpdated); + } Ok(self.agreement.lock().unwrap().clone()) } async fn update_offer_tx_hash( @@ -364,7 +371,7 @@ mod tests { if let Some(agreement) = &self.cancel_mid_send && let Some(row) = agreement.lock().unwrap().as_mut() { - row.status = IndexingAgreementStatus::CanceledByRequester; + row.status = IndexingAgreementStatus::Cancelling; } let result = self .offer_result @@ -489,6 +496,8 @@ mod tests { Ctx { registry: MockRegistry { agreement: Arc::new(Mutex::new(Some(agreement))), + later_reads_fail: false, + reads: AtomicU32::new(0), }, chain_client: MockChainClient { offer_result: Mutex::new(offer_result), @@ -502,29 +511,7 @@ mod tests { } fn test_agreement_conf() -> crate::config::IndexingAgreementConfig { - crate::config::IndexingAgreementConfig { - data_service: Address::ZERO, - recurring_collector: Address::ZERO, - recurring_agreement_manager: Address::ZERO, - max_agreement_grt_per_30_days: 0.0, - max_seconds_per_collection: 0, - min_seconds_per_collection: 0, - duration_seconds: 0, - deadline_seconds: 0, - max_grt_per_30_days: std::collections::BTreeMap::new(), - max_grt_per_billion_entities_per_30_days: 0.0, - declined_indexer_lookback_days: 0, - price_rejection_lookback_days: 0, - transient_rejection_lookback_minutes: 0, - uncertain_rejection_lookback_days: 0, - unresponsive_indexer_lookback_days: 0, - mass_unresponsive_trip_fraction: 0.5, - mass_unresponsive_reset_fraction: 0.25, - dips_accepting_snapshot_max_age_hours: 48, - dips_accepting_cache_ttl_seconds: 300, - max_in_flight_offers_per_indexer: None, - max_in_flight_offers_total: None, - } + crate::config::IndexingAgreementConfig::for_tests() } #[tokio::test] @@ -547,6 +534,21 @@ mod tests { assert_eq!(*cancelled.lock().unwrap(), vec![*agreement_id.as_bytes()]); } + #[tokio::test] + async fn finishes_without_resending_when_it_cannot_recheck_the_agreement() { + //* Arrange - a retry would send the offer again; the mock panics on a second send + let agreement = make_test_agreement(); + let message = make_message(agreement.id); + let mut ctx = ctx_with_offer_result(agreement, Ok(Some(B256::repeat_byte(0xab)))); + ctx.registry.later_reads_fail = true; + + //* Act + let result = handle(ctx, &message).await; + + //* Assert + assert!(result.is_ok(), "got {result:?}"); + } + #[tokio::test] async fn keeps_its_offer_when_the_agreement_is_still_wanted() { //* Arrange @@ -583,23 +585,28 @@ mod tests { #[tokio::test] async fn withdraws_a_stored_offer_of_an_agreement_already_cancelled() { - //* Arrange - an earlier attempt sent the offer, then the agreement was - // cancelled before this retry; no offer result, so a send would panic - let mut agreement = make_test_agreement(); - agreement.status = IndexingAgreementStatus::CanceledByRequester; - agreement.terms_version_hash = Some(vec![7u8; 32]); - let agreement_id = agreement.id; - let message = make_message(agreement_id); - let ctx = ctx_with(agreement, None); - ctx.chain_client.on_chain.store(true, Ordering::SeqCst); - let cancelled = ctx.chain_client.cancelled.clone(); - - //* Act - let result = handle(ctx, &message).await; - - //* Assert - assert!(result.is_ok(), "got {result:?}"); - assert_eq!(*cancelled.lock().unwrap(), vec![*agreement_id.as_bytes()]); + for status in [ + IndexingAgreementStatus::Cancelling, + IndexingAgreementStatus::CanceledByRequester, + ] { + //* Arrange - an earlier attempt sent the offer, then the agreement was + // cancelled before this retry; no offer result, so a send would panic + let mut agreement = make_test_agreement(); + agreement.status = status; + agreement.terms_version_hash = Some(vec![7u8; 32]); + let agreement_id = agreement.id; + let message = make_message(agreement_id); + let ctx = ctx_with(agreement, None); + ctx.chain_client.on_chain.store(true, Ordering::SeqCst); + let cancelled = ctx.chain_client.cancelled.clone(); + + //* Act + let result = handle(ctx, &message).await; + + //* Assert + assert!(result.is_ok(), "got {result:?}"); + assert_eq!(*cancelled.lock().unwrap(), vec![*agreement_id.as_bytes()]); + } } #[tokio::test] diff --git a/dipper-pgregistry/migrations/20261002000000_add_cancelling_status.sql b/dipper-pgregistry/migrations/20261002000000_add_cancelling_status.sql new file mode 100644 index 00000000..e9c86d29 --- /dev/null +++ b/dipper-pgregistry/migrations/20261002000000_add_cancelling_status.sql @@ -0,0 +1,15 @@ +-- Cancelling (status = 9): dipper has decided to end the agreement and keeps sending the +-- on-chain cancel until the chain confirms it ended; only then does the row become +-- CanceledByRequester, which announces the end. +-- +-- cancel_attempts counts cancels that reached the chain without ending the agreement, +-- so one that can never work stops being retried and is left for an operator. +-- cancel_checked_at lets each retry sweep start with the agreements checked longest ago. +ALTER TABLE dipper_reg_indexing_agreements + ADD COLUMN cancel_attempts INTEGER NOT NULL DEFAULT 0, + ADD COLUMN cancel_checked_at TIMESTAMPTZ; + +-- Partial, so it only covers agreements still being cancelled (normally none). +CREATE INDEX idx_indexing_agreements_cancelling + ON dipper_reg_indexing_agreements (cancel_checked_at NULLS FIRST, updated_at) + WHERE status = 9; diff --git a/dipper-pgregistry/src/indexing_agreement.rs b/dipper-pgregistry/src/indexing_agreement.rs index 53c20c32..4a3b24fe 100644 --- a/dipper-pgregistry/src/indexing_agreement.rs +++ b/dipper-pgregistry/src/indexing_agreement.rs @@ -206,6 +206,11 @@ pub enum Status { /// This is a terminal state. AbandonedByIndexer = 8, + /// Dipper decided to end the agreement and is cancelling it on-chain, where it may + /// still be live. It becomes `CanceledByRequester`, announced as ended, only once + /// the chain confirms the end. + Cancelling = 9, + /// A fallback for unknown status values. Unknown = i32::MAX, } @@ -221,6 +226,7 @@ impl std::fmt::Display for Status { Status::AcceptedOnChain => "ACCEPTED_ON_CHAIN", Status::Rejected => "REJECTED", Status::AbandonedByIndexer => "ABANDONED_BY_INDEXER", + Status::Cancelling => "CANCELLING", Status::Unknown => "UNKNOWN", }; f.write_str(status) diff --git a/dipper-pgregistry/src/lib.rs b/dipper-pgregistry/src/lib.rs index 20b40ea6..8e0ea869 100644 --- a/dipper-pgregistry/src/lib.rs +++ b/dipper-pgregistry/src/lib.rs @@ -20,9 +20,9 @@ pub use indexing_request::{ Status as IndexingRequestStatus, }; pub use postgres::{ - CancelKind, ChainListenerStateRow, NewAgreementParams, PendingAcceptedEvent, - PendingExpiredEvent, PendingTerminatedEvent, PgRegistry, ReconciliationAudit, - ReconciliationItem, ReconciliationOutcome, + CancelKind, CancellingAgreement, ChainListenerStateRow, NewAgreementParams, + PendingAcceptedEvent, PendingExpiredEvent, PendingTerminatedEvent, PgRegistry, + ReconciliationAudit, ReconciliationItem, ReconciliationOutcome, }; pub use result::{Error, Result}; diff --git a/dipper-pgregistry/src/postgres.rs b/dipper-pgregistry/src/postgres.rs index 14c85160..d40439b3 100644 --- a/dipper-pgregistry/src/postgres.rs +++ b/dipper-pgregistry/src/postgres.rs @@ -187,6 +187,39 @@ impl sqlx::FromRow<'_, sqlx::postgres::PgRow> for PendingAcceptedEvent { } } +/// An agreement dipper is still cancelling on-chain. +#[derive(Debug, Clone)] +pub struct CancellingAgreement { + pub agreement: IndexingAgreement, + /// Whether dipper saw it accepted on-chain, so its end is announced. + pub accepted_on_chain: bool, +} + +impl sqlx::FromRow<'_, sqlx::postgres::PgRow> for CancellingAgreement { + fn from_row(row: &sqlx::postgres::PgRow) -> Result { + use sqlx::Row as _; + let accepted_at: Option = row.try_get("accepted_at")?; + Ok(Self { + agreement: IndexingAgreement::from_row(row)?, + accepted_on_chain: accepted_at.is_some(), + }) + } +} + +/// Statuses an on-chain cancel by dipper ends. +const CANCEL_BY_REQUESTER_FROM: &[IndexingAgreementStatus] = &[ + IndexingAgreementStatus::Created, + IndexingAgreementStatus::AcceptedOnChain, + IndexingAgreementStatus::Rejected, + IndexingAgreementStatus::Cancelling, +]; + +/// Statuses an on-chain cancel by the indexer ends. +const CANCEL_BY_INDEXER_FROM: &[IndexingAgreementStatus] = &[ + IndexingAgreementStatus::AcceptedOnChain, + IndexingAgreementStatus::Cancelling, +]; + /// A row that needs a `request.expired` lifecycle event emitted. Sourced from /// the agreement row alone; `request_expired_at` is the terms deadline (the true /// expiry instant), so the sweep needs no chain-time snapshot. @@ -911,29 +944,117 @@ impl PgRegistry { &self, agreement_id: &IndexingAgreementId, ) -> Result<(), Error> { - let record: Option<(IndexingAgreementId,)> = sqlx::query_as( + self.set_status_from( + agreement_id, + IndexingAgreementStatus::CanceledByRequester, + CANCEL_BY_REQUESTER_FROM, + ) + .await + } + + /// Mark an agreement that may be live on-chain `Cancelling`, before dipper sends its + /// on-chain cancel. One marked `Expired` may have been accepted unseen by a lagging listener. + pub async fn mark_indexing_agreement_as_cancelling( + &self, + agreement_id: &IndexingAgreementId, + ) -> Result<(), Error> { + self.set_status_from( + agreement_id, + IndexingAgreementStatus::Cancelling, + &[ + IndexingAgreementStatus::Created, + IndexingAgreementStatus::AcceptedOnChain, + IndexingAgreementStatus::Rejected, + IndexingAgreementStatus::Expired, + ], + ) + .await + } + + /// `Cancelling` agreements marked over `min_age_minutes` ago whose cancel has failed + /// fewer than `max_attempts` times, those checked longest ago first. + pub async fn get_cancelling_agreements( + &self, + batch_size: i64, + max_attempts: u32, + min_age_minutes: i32, + ) -> Result, Error> { + sqlx::query_as( + r#" + SELECT + id, + nonce_uuid, + created_at, + updated_at, + status, + indexing_request_id, + deployment_id, + indexer_id, + indexer_url, + terms, + last_block_height, + last_progress_at, + rejection_reason, + terms_version_hash, + accepted_at + FROM dipper_reg_indexing_agreements + WHERE status = $1 + AND cancel_attempts < $2 + AND updated_at < timezone('UTC', now()) - make_interval(mins => $4) + ORDER BY cancel_checked_at ASC NULLS FIRST, updated_at ASC + LIMIT $3 + "#, + ) + .bind(IndexingAgreementStatus::Cancelling) + .bind(i32::try_from(max_attempts).unwrap_or(i32::MAX)) + .bind(batch_size) + .bind(min_age_minutes) + .fetch_all(&self.pool) + .await + .map_err(Into::into) + } + + /// Record a check of a `Cancelling` agreement that left it cancelling, adding + /// `failed_attempts` to its failed cancels and returning the new count. + pub async fn record_cancel_check( + &self, + agreement_id: &IndexingAgreementId, + failed_attempts: u32, + ) -> Result { + let record: Option<(i32,)> = sqlx::query_as( r#" UPDATE dipper_reg_indexing_agreements SET - status = $1, - updated_at = timezone('UTC', now()) - WHERE id = $2 AND status IN ($3, $4, $5) - RETURNING id + cancel_attempts = LEAST(cancel_attempts::BIGINT + $3, 2147483647)::INTEGER, + cancel_checked_at = timezone('UTC', now()) + WHERE id = $1 AND status = $2 + RETURNING cancel_attempts "#, ) - .bind(IndexingAgreementStatus::CanceledByRequester) .bind(agreement_id) - .bind(IndexingAgreementStatus::Created) - .bind(IndexingAgreementStatus::AcceptedOnChain) - .bind(IndexingAgreementStatus::Rejected) + .bind(IndexingAgreementStatus::Cancelling) + .bind(i64::from(failed_attempts)) .fetch_optional(&self.pool) .await?; + let (attempts,) = record.ok_or(Error::NoRecordsUpdated)?; + Ok(u32::try_from(attempts).unwrap_or_default()) + } - if record.is_none() { - return Err(Error::NoRecordsUpdated); + /// Move an agreement to `new_status` if it is in one of `allowed_from`. + async fn set_status_from( + &self, + agreement_id: &IndexingAgreementId, + new_status: IndexingAgreementStatus, + allowed_from: &[IndexingAgreementStatus], + ) -> Result<(), Error> { + let mut tx = self.pool.begin().await?; + let updated = update_status_from(&mut tx, agreement_id, new_status, allowed_from).await?; + tx.commit().await?; + if updated { + Ok(()) + } else { + Err(Error::NoRecordsUpdated) } - - Ok(()) } /// Atomically apply a reconciliation-driven state transition (accept @@ -986,15 +1107,11 @@ impl PgRegistry { let (new_status, allowed_from): (_, &[IndexingAgreementStatus]) = match kind { CancelKind::ByRequester => ( IndexingAgreementStatus::CanceledByRequester, - &[ - IndexingAgreementStatus::Created, - IndexingAgreementStatus::AcceptedOnChain, - IndexingAgreementStatus::Rejected, - ], + CANCEL_BY_REQUESTER_FROM, ), CancelKind::ByIndexer => ( IndexingAgreementStatus::CanceledByIndexer, - &[IndexingAgreementStatus::AcceptedOnChain], + CANCEL_BY_INDEXER_FROM, ), }; did_cancel = @@ -1084,15 +1201,11 @@ impl PgRegistry { let (new_status, allowed_from): (_, &[IndexingAgreementStatus]) = match cancel_kind { CancelKind::ByRequester => ( IndexingAgreementStatus::CanceledByRequester, - &[ - IndexingAgreementStatus::Created, - IndexingAgreementStatus::AcceptedOnChain, - IndexingAgreementStatus::Rejected, - ], + CANCEL_BY_REQUESTER_FROM, ), CancelKind::ByIndexer => ( IndexingAgreementStatus::CanceledByIndexer, - &[IndexingAgreementStatus::AcceptedOnChain], + CANCEL_BY_INDEXER_FROM, ), }; let did_cancel = @@ -1127,11 +1240,7 @@ impl PgRegistry { &mut tx, &cancel_by_requester, IndexingAgreementStatus::CanceledByRequester, - &[ - IndexingAgreementStatus::Created, - IndexingAgreementStatus::AcceptedOnChain, - IndexingAgreementStatus::Rejected, - ], + CANCEL_BY_REQUESTER_FROM, ) .await? { @@ -1142,7 +1251,7 @@ impl PgRegistry { &mut tx, &cancel_by_indexer, IndexingAgreementStatus::CanceledByIndexer, - &[IndexingAgreementStatus::AcceptedOnChain], + CANCEL_BY_INDEXER_FROM, ) .await? { @@ -1834,8 +1943,8 @@ impl PgRegistry { /// Returns (agreement_id, indexer_id, deployment_id, base_rate_wei, /// entity_rate_wei) per active agreement for optimistic fee estimation. /// - /// Queries all `Created` or `AcceptedOnChain` agreements and extracts - /// both rate fields from the terms metadata. + /// Queries all `Created`, `AcceptedOnChain` or `Cancelling` agreements, the last + /// still paid until their cancel lands, and extracts both rate fields from the terms. pub async fn get_agreement_fee_rates( &self, ) -> Result, Error> { @@ -1847,11 +1956,12 @@ impl PgRegistry { r#" SELECT id, indexer_id, terms FROM dipper_reg_indexing_agreements - WHERE status IN ($1, $2) + WHERE status IN ($1, $2, $3) "#, ) .bind(IndexingAgreementStatus::Created) .bind(IndexingAgreementStatus::AcceptedOnChain) + .bind(IndexingAgreementStatus::Cancelling) .fetch_all(&self.pool) .await?; diff --git a/dipper-pgregistry/tests/it_registry_postgres.rs b/dipper-pgregistry/tests/it_registry_postgres.rs index cd6f3e7a..18530ded 100644 --- a/dipper-pgregistry/tests/it_registry_postgres.rs +++ b/dipper-pgregistry/tests/it_registry_postgres.rs @@ -3344,3 +3344,181 @@ async fn count_created_agreements_by_indexer_counts_only_created() { ); assert_eq!(global, 3, "global counts only the 3 Created rows"); } + +fn fixture_agreement(prefix: u8) -> IndexingAgreementId { + let mut bytes = [0u8; 16]; + bytes[0] = prefix; + bytes[15] = 1; + IndexingAgreementId::from_bytes(bytes) +} + +#[tokio::test] +async fn cancelling_agreements_are_listed_until_their_cancel_fails_too_often() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let registry = PgRegistry::new(db); + let created = fixture_agreement(0xaa); + let accepted = + IndexingAgreementId::from_bytes([0xaa, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2]); + let ended = + IndexingAgreementId::from_bytes([0xaa, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 3]); + let expired = fixture_agreement(0xcc); + + for id in [created, accepted, expired] { + registry + .mark_indexing_agreement_as_cancelling(&id) + .await + .expect("an agreement that may be live can be marked cancelling"); + } + let result = registry.mark_indexing_agreement_as_cancelling(&ended).await; + assert!( + matches!(result, Err(Error::NoRecordsUpdated)), + "got {result:?}" + ); + registry + .mark_indexing_agreement_as_canceled_by_requester(&expired) + .await + .expect("leave 2 cancelling"); + registry + .record_accepted_audit(&accepted, 1_700_000_000, "0xacc") + .await + .expect("accept record"); + + let just_marked = registry + .get_cancelling_agreements(100, 2, 5) + .await + .expect("cancelling query"); + assert!( + just_marked.is_empty(), + "one just marked waits for the cancel sent with the mark to be mined" + ); + + let listed = registry + .get_cancelling_agreements(100, 2, 0) + .await + .expect("cancelling query"); + let mut seen: Vec<_> = listed + .iter() + .map(|row| { + ( + row.agreement.id, + row.accepted_on_chain, + row.agreement.status, + ) + }) + .collect(); + seen.sort_by_key(|(id, ..)| *id); + assert_eq!( + seen, + vec![ + (created, false, IndexingAgreementStatus::Cancelling), + (accepted, true, IndexingAgreementStatus::Cancelling), + ] + ); + + assert_eq!(registry.record_cancel_check(&accepted, 0).await.unwrap(), 0); + let listed = registry + .get_cancelling_agreements(100, 2, 0) + .await + .expect("cancelling query"); + let ids: Vec<_> = listed.iter().map(|row| row.agreement.id).collect(); + assert_eq!( + ids, + vec![created, accepted], + "one checked longest ago goes first" + ); + + assert_eq!(registry.record_cancel_check(&created, 1).await.unwrap(), 1); + assert_eq!(registry.record_cancel_check(&created, 1).await.unwrap(), 2); + let listed = registry + .get_cancelling_agreements(100, 2, 0) + .await + .expect("cancelling query"); + let ids: Vec<_> = listed.iter().map(|row| row.agreement.id).collect(); + assert_eq!( + ids, + vec![accepted], + "one that failed too often is left alone" + ); + + let not_cancelling = registry + .record_cancel_check(&fixture_agreement(0xbb), 1) + .await; + assert!(matches!(not_cancelling, Err(Error::NoRecordsUpdated))); +} + +#[tokio::test] +async fn a_cancelling_agreement_stays_live_and_unannounced_until_it_ends() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let registry = PgRegistry::new(db); + let cancelling = + IndexingAgreementId::from_bytes([0xaa, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2]); + let by_indexer = fixture_agreement(0xbb); + registry + .mark_indexing_agreement_as_canceled_by_requester(&fixture_agreement(0xaa)) + .await + .expect("cancel the other live agreement"); + for id in [cancelling, by_indexer] { + registry + .mark_indexing_agreement_as_cancelling(&id) + .await + .expect("mark cancelling"); + } + registry + .record_accepted_audit(&cancelling, 1_700_000_000, "0xacc") + .await + .expect("accept record"); + registry + .record_cancel_audit(&cancelling, 1_700_000_001, "0xmgr", Some("0xcxl")) + .await + .expect("cancel record"); + + assert!( + !registry.exists_active_agreements().await.unwrap(), + "agreements being cancelled don't keep the listener polling fast; their retry \ + runs on its own timer" + ); + let fee_rates = registry + .get_agreement_fee_rates() + .await + .expect("fee rates query"); + assert!( + fee_rates.iter().any(|(id, ..)| *id == cancelling), + "its fees still count until the cancel lands" + ); + let terminated = registry + .get_agreements_pending_terminated_emission(100) + .await + .expect("terminated query"); + assert!( + terminated.iter().all(|p| p.agreement_id != cancelling), + "not announced as ended while still cancelling" + ); + + let outcome = registry + .apply_reconciliation(&by_indexer, false, Some(CancelKind::ByIndexer)) + .await + .expect("indexer's cancel read from the chain"); + assert!(outcome.did_cancel); + registry + .mark_indexing_agreement_as_canceled_by_requester(&cancelling) + .await + .expect("dipper's cancel confirmed"); + + let terminated = registry + .get_agreements_pending_terminated_emission(100) + .await + .expect("terminated query"); + assert!(terminated.iter().any(|p| p.agreement_id == cancelling)); +} diff --git a/dipper-rpc/src/admin/indexing_agreements.rs b/dipper-rpc/src/admin/indexing_agreements.rs index a08c7e9a..7be13742 100644 --- a/dipper-rpc/src/admin/indexing_agreements.rs +++ b/dipper-rpc/src/admin/indexing_agreements.rs @@ -122,6 +122,9 @@ pub enum Status { /// This is a terminal state. AbandonedByIndexer, + /// Dipper is cancelling the agreement on-chain; it may still be live there. + Cancelling, + /// A fallback for unknown status values. Unknown, } @@ -140,6 +143,7 @@ impl serde::Serialize for Status { Status::AcceptedOnChain => "ACCEPTED_ON_CHAIN", Status::Rejected => "REJECTED", Status::AbandonedByIndexer => "ABANDONED_BY_INDEXER", + Status::Cancelling => "CANCELLING", Status::Unknown => "UNKNOWN", }; serializer.serialize_str(status) @@ -161,6 +165,7 @@ impl<'de> serde::Deserialize<'de> for Status { "ACCEPTED_ON_CHAIN" => Status::AcceptedOnChain, "REJECTED" => Status::Rejected, "ABANDONED_BY_INDEXER" => Status::AbandonedByIndexer, + "CANCELLING" => Status::Cancelling, _ => Status::Unknown, }; Ok(status) From d7349e67e4547039019279b098aa584bed5d9f43 Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Sat, 3 Oct 2026 01:12:31 +0300 Subject: [PATCH 08/26] fix: stop paying an indexer twice while its cancel is retried (#718) * fix(selection): keep indexers dipper is cancelling out of selection An agreement dipper is still cancelling may be live and paying its indexer, yet IISA wasn't told to skip that indexer, so it could be picked again and paid twice. It now goes on the deployment's declined list, which blocks a pick without counting it as part of the group. * fix(cancel): never record an indexer's cancel as dipper's The cancel retry marked an accepted agreement it found ended as cancelled by dipper after an hour, even when the indexer had ended it, so the wrong end was announced. It now reads who ended it from the contract, and leaves an end by the indexer for the chain listener to record as theirs. * fix(cancel): retry cancels of agreements that may be paying first The cancel retry took 10 agreements per run, oldest check first, so offers nobody had accepted could hold up a live agreement for hours. It now takes up to 50, putting first those accepted or past their offer deadline, and its 30 second time limit decides how many it gets through. * fix(cancel): close an indexer's cancel the listener never recorded An agreement the indexer ended was left for the chain listener to record, so if the listener never did, it stayed cancelling for good and its indexer was kept out of selection. After an hour, the cancel retry now marks it cancelled by the indexer itself, naming them as the one who ended it. * fix(cancel): stop open offers waiting forever behind paying agreements Agreements that may be paying an indexer were always retried first, so if enough of them kept failing, offers an indexer could still accept were never retried. Those agreements now get an hour's head start instead, so an offer left unchecked for over an hour still gets its turn. --- bin/dipper-service/src/cancel_dispatch.rs | 6 + bin/dipper-service/src/chain_client.rs | 14 ++ bin/dipper-service/src/chain_client/client.rs | 55 +++-- .../src/network/service/cancel_retry.rs | 197 ++++++++++++++++-- .../src/network/service/chain_listener.rs | 6 + .../src/network/service/escrow_reconciler.rs | 6 + .../src/network/service/liveness_checker.rs | 6 + bin/dipper-service/src/registry/agreement.rs | 4 +- .../cancel_rejected_agreement_on_chain.rs | 6 + .../handlers/reassess_indexing_request.rs | 12 ++ .../src/worker/handlers/selection_context.rs | 79 ++++++- .../src/worker/handlers/submit_offer.rs | 6 + dipper-pgregistry/src/postgres.rs | 13 +- .../tests/it_registry_postgres.rs | 34 ++- 14 files changed, 390 insertions(+), 54 deletions(-) diff --git a/bin/dipper-service/src/cancel_dispatch.rs b/bin/dipper-service/src/cancel_dispatch.rs index e4c9c80f..4f6734a2 100644 --- a/bin/dipper-service/src/cancel_dispatch.rs +++ b/bin/dipper-service/src/cancel_dispatch.rs @@ -280,6 +280,12 @@ pub(crate) mod tests { *self.active_reads.lock().unwrap() += 1; Ok(self.still_active_after_cancel) } + async fn agreement_ended_by_indexer( + &self, + _agreement_id: &[u8; 16], + ) -> Result { + Ok(false) + } } fn manager_conf(collector: Address) -> IndexingAgreementConfig { diff --git a/bin/dipper-service/src/chain_client.rs b/bin/dipper-service/src/chain_client.rs index 5c734b26..f9f1583f 100644 --- a/bin/dipper-service/src/chain_client.rs +++ b/bin/dipper-service/src/chain_client.rs @@ -128,6 +128,13 @@ pub trait ChainClient { agreement_id: &[u8; 16], ) -> Result; + /// Read whether the indexer ended the agreement on-chain, which `getAgreementDetails` + /// reports with its BY_PROVIDER flag. False for one still live or ended by dipper. + async fn agreement_ended_by_indexer( + &self, + agreement_id: &[u8; 16], + ) -> Result; + /// Read the latest block's unix timestamp from the chain. Lets agreement /// deadlines be stamped from live chain time when the chain-clock bypass is /// on, instead of a cached listener timestamp that can lag a fast chain. @@ -182,6 +189,13 @@ impl ChainClient for Arc { (**self).agreement_still_active(agreement_id).await } + async fn agreement_ended_by_indexer( + &self, + agreement_id: &[u8; 16], + ) -> Result { + (**self).agreement_ended_by_indexer(agreement_id).await + } + async fn latest_block_timestamp(&self) -> Result { (**self).latest_block_timestamp().await } diff --git a/bin/dipper-service/src/chain_client/client.rs b/bin/dipper-service/src/chain_client/client.rs index e759edb4..8aae2df7 100644 --- a/bin/dipper-service/src/chain_client/client.rs +++ b/bin/dipper-service/src/chain_client/client.rs @@ -58,12 +58,13 @@ const RECEIPT_POLL_INTERVAL: Duration = Duration::from_millis(500); const VERSION_CURRENT: u64 = 0; /// `AgreementDetails.state` flags from `IAgreementCollector.sol` (REGISTERED=1, -/// ACCEPTED=2, NOTICE_GIVEN=4). `getAgreementDetails` keeps ACCEPTED set on a -/// canceled agreement and ORs in NOTICE_GIVEN, so a cancel must clear it, not -/// just lack it. REGISTERED without ACCEPTED is an offer still waiting. +/// ACCEPTED=2, NOTICE_GIVEN=4, BY_PROVIDER=32). `getAgreementDetails` keeps ACCEPTED set on +/// a canceled agreement and ORs in NOTICE_GIVEN, so a cancel must clear it, not just lack +/// it. REGISTERED without ACCEPTED is an offer still waiting. BY_PROVIDER: the indexer cancelled. const STATE_REGISTERED: u16 = 1; const STATE_ACCEPTED: u16 = 2; const STATE_NOTICE_GIVEN: u16 = 4; +const STATE_BY_PROVIDER: u16 = 32; /// Live iff the terms are accepted and no cancellation notice exists, or an /// offer is still stored for the indexer to accept. A cancel sets NOTICE_GIVEN @@ -602,6 +603,19 @@ impl AlloyChainClient { } } + /// The agreement's `AgreementDetails.state` flags for its current terms. + async fn agreement_state(&self, agreement_id: &[u8; 16]) -> Result { + let call = IRecurringCollector::getAgreementDetailsCall { + agreementId: FixedBytes::<16>::from_slice(agreement_id), + index: thegraph_core::alloy::primitives::U256::from(VERSION_CURRENT), + }; + let collector = self.inner.recurring_collector_address; + Ok(self + .view(collector, call, "get_agreement_details") + .await? + .state) + } + /// Run a read-only contract call and decode its return value. async fn view( &self, @@ -789,35 +803,14 @@ impl ChainClient for AlloyChainClient { &self, agreement_id: &[u8; 16], ) -> Result { - let calldata = IRecurringCollector::getAgreementDetailsCall { - agreementId: FixedBytes::<16>::from_slice(agreement_id), - index: thegraph_core::alloy::primitives::U256::from(VERSION_CURRENT), - } - .abi_encode(); - - let collector = self.inner.recurring_collector_address; - let output = self - .inner - .rpc_pool - .execute("get_agreement_details", |provider| { - let calldata = calldata.clone(); - async move { - let tx = TransactionRequest::default() - .to(collector) - .input(calldata.into()); - provider.call(tx).await - } - }) - .await?; - - let details = IRecurringCollector::getAgreementDetailsCall::abi_decode_returns(&output) - .map_err(|err| { - ChainClientError::RpcError(anyhow::anyhow!( - "undecodable getAgreementDetails from {collector}: {err}" - )) - })?; + Ok(still_live(self.agreement_state(agreement_id).await?)) + } - Ok(still_live(details.state)) + async fn agreement_ended_by_indexer( + &self, + agreement_id: &[u8; 16], + ) -> Result { + Ok(self.agreement_state(agreement_id).await? & STATE_BY_PROVIDER != 0) } async fn reconcile_provider( diff --git a/bin/dipper-service/src/network/service/cancel_retry.rs b/bin/dipper-service/src/network/service/cancel_retry.rs index ca465fa9..dc4874b2 100644 --- a/bin/dipper-service/src/network/service/cancel_retry.rs +++ b/bin/dipper-service/src/network/service/cancel_retry.rs @@ -2,21 +2,23 @@ //! `Cancelling` before its on-chain cancel goes out; this sweep re-sends the cancel while //! the chain shows it live, and marks it `CanceledByRequester` once it can no longer be. +use dipper_core::time::now_secs; use thegraph_core::alloy::primitives::B256; use crate::{ cancel_dispatch::{LiveCancel, cancel_if_live, record_cancel}, chain_client::{ChainClient, ChainClientError}, config::IndexingAgreementConfig, - registry::{AgreementRegistry, CancellingAgreement, IndexingAgreement}, + registry::{AgreementRegistry, CancelKind, CancellingAgreement, IndexingAgreement}, }; /// Retried cancels mined without ending an agreement before dipper stops retrying it and /// leaves it to an operator. Other failures don't count (see `failed_attempts`). pub const MAX_CANCEL_ATTEMPTS: u32 = 10; -/// Agreements checked per sweep, those checked longest ago first. -const BATCH_SIZE: i64 = 10; +/// Agreements a sweep takes on, those that may be paying an indexer first; the time budget +/// below decides how many it gets through. +const BATCH_SIZE: i64 = 50; /// Time a sweep may take before leaving the rest to the next one: it holds up the chain /// listener while it runs, and each cancel can wait up to 15 s to be mined. @@ -26,8 +28,8 @@ const SWEEP_BUDGET: std::time::Duration = std::time::Duration::from_secs(30); /// when it was marked can be mined first instead of being sent again. const SETTLE_MINUTES: i32 = 2; -/// How long the chain listener gets to record who ended an accepted agreement, and when, -/// before the retry marks it ended without those details. +/// How long the chain listener gets to record when, and in which transaction, an accepted +/// agreement ended, before the retry marks it ended without those details. const LISTENER_GRACE: time::Duration = time::Duration::HOUR; /// Retry the cancel of agreements still `Cancelling`. `chain_now`, in chain seconds, @@ -96,31 +98,46 @@ async fn retry_cancel( } LiveCancel::CancelFailed(err) => (None, Some(err)), }; - if failure.is_none() && confirm_if_over(registry, config, row, tx_hash, chain_now).await { + if failure.is_none() + && confirm_if_over(registry, chain_client, config, row, tx_hash, chain_now).await + { return; } note_check(registry, row, failure.as_ref()).await; } /// Mark the agreement `CanceledByRequester` once it can't go live again: this sweep's cancel -/// ended it, or nobody accepted its offer before the deadline to. One accepted that ended -/// otherwise is left to the chain listener, which reads who ended it and when, for a while. -async fn confirm_if_over( +/// ended it, or nobody accepted its offer before the deadline to. One ended otherwise is left +/// to the chain listener for a while; one the indexer ended then becomes `CanceledByIndexer`. +async fn confirm_if_over( registry: &R, + chain_client: &T, config: &IndexingAgreementConfig, row: &CancellingAgreement, tx_hash: Option, chain_now: u64, -) -> bool { +) -> bool +where + R: AgreementRegistry + Sync, + T: ChainClient, +{ let agreement = &row.agreement; + let past_grace = agreement.updated_at < time::OffsetDateTime::now_utc() - LISTENER_GRACE; let can_confirm = if row.accepted_on_chain { - tx_hash.is_some() || agreement.updated_at < time::OffsetDateTime::now_utc() - LISTENER_GRACE + tx_hash.is_some() || past_grace } else { chain_now > agreement.terms.deadline }; if !can_confirm { return false; } + if tx_hash.is_none() { + match ended_by_indexer(chain_client, agreement).await { + None => return false, + Some(true) => return past_grace && record_end_by_indexer(registry, agreement).await, + Some(false) => {} + } + } if let Err(err) = registry .mark_indexing_agreement_as_canceled_by_requester(&agreement.id) .await @@ -147,6 +164,79 @@ async fn confirm_if_over( true } +/// Whether the chain shows the indexer ended the agreement, or `None` when it can't be read, +/// so an end is never wrongly put down to dipper. +async fn ended_by_indexer( + chain_client: &T, + agreement: &IndexingAgreement, +) -> Option { + match chain_client + .agreement_ended_by_indexer(agreement.id.as_bytes()) + .await + { + Ok(by_indexer) => { + if by_indexer { + tracing::info!( + agreement_id = %agreement.id, + "The indexer ended an agreement dipper was cancelling" + ); + } + Some(by_indexer) + } + Err(err) => { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to read who ended a cancelling agreement, will retry" + ); + None + } + } +} + +/// Mark an agreement the indexer ended `CanceledByIndexer` when the chain listener hasn't in +/// time, so it doesn't stay cancelling for good. The indexer is recorded as ending it first, +/// so its announcement names them; the time recorded is when dipper noticed. +async fn record_end_by_indexer( + registry: &R, + agreement: &IndexingAgreement, +) -> bool { + let indexer = agreement.indexer.id.to_string(); + let marked = match registry + .record_cancel_audit(&agreement.id, now_secs(), &indexer, None) + .await + { + Ok(()) => { + registry + .apply_reconciliation(&agreement.id, false, Some(CancelKind::ByIndexer)) + .await + } + Err(err) => Err(err), + }; + match marked { + Ok(outcome) => { + tracing::info!( + agreement_id = %agreement.id, + indexing_request_id = %agreement.indexing_request_id, + old_status = "CANCELLING", + new_status = "CANCELED_BY_INDEXER", + applied = outcome.did_cancel, + reason = "indexer_cancel_seen_on_chain", + "agreement state transition" + ); + true + } + Err(err) => { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to mark an agreement the indexer ended, will retry" + ); + false + } + } +} + /// Record that the agreement was checked and is still cancelling, counting a cancel the /// chain answered without ending it; past the limit, dipper gives up with an ERROR. async fn note_check( @@ -252,7 +342,9 @@ mod tests { struct MockRegistry { cancelling: Vec, marked_cancelled: Mutex>, + marked_by_indexer: Mutex>, audits: Mutex>>, + audited_by: Mutex>, attempts: AtomicU32, checks: AtomicU32, } @@ -278,15 +370,29 @@ mod tests { &self, _id: &IndexingAgreementId, _canceled_at: u64, - _canceled_by: &str, + canceled_by: &str, canceled_tx: Option<&str>, ) -> crate::registry::Result<()> { + self.audited_by.lock().unwrap().push(canceled_by.to_owned()); self.audits .lock() .unwrap() .push(canceled_tx.map(str::to_owned)); Ok(()) } + async fn apply_reconciliation( + &self, + id: &IndexingAgreementId, + _apply_accept: bool, + cancel: Option, + ) -> crate::registry::Result { + assert_eq!(cancel, Some(CancelKind::ByIndexer)); + self.marked_by_indexer.lock().unwrap().push(*id); + Ok(crate::registry::ReconciliationOutcome { + did_accept: false, + did_cancel: true, + }) + } async fn record_cancel_check( &self, _id: &IndexingAgreementId, @@ -305,6 +411,8 @@ mod tests { send_fails: bool, mined_cancel_reverts: bool, cancel_has_no_effect: bool, + ended_by_indexer: bool, + who_read_fails: bool, cancels_sent: AtomicU32, } @@ -360,6 +468,15 @@ mod tests { } Ok(self.live.load(Ordering::SeqCst)) } + async fn agreement_ended_by_indexer( + &self, + _agreement_id: &[u8; 16], + ) -> Result { + if self.who_read_fails { + return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); + } + Ok(self.ended_by_indexer) + } async fn latest_block_timestamp(&self) -> Result { unimplemented!() } @@ -430,6 +547,62 @@ mod tests { assert!(registry.audits.lock().unwrap().is_empty()); } + #[tokio::test] + async fn leaves_an_end_by_the_indexer_to_the_listener_for_a_while() { + // The listener records when and in which transaction. An accepted agreement can lack + // an accept time, if accepted before accepts were recorded, so both kinds are checked. + for accepted_on_chain in [true, false] { + let registry = registry_with_one(accepted_on_chain); + let chain = MockChain { + ended_by_indexer: true, + ..MockChain::default() + }; + + retry(®istry, &chain, DEADLINE + 1).await; + + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + assert!(registry.marked_by_indexer.lock().unwrap().is_empty()); + assert_eq!(registry.checks.load(Ordering::SeqCst), 1); + } + } + + #[tokio::test] + async fn marks_an_end_by_the_indexer_as_theirs_once_the_listener_has_had_long_enough() { + for accepted_on_chain in [true, false] { + let mut registry = registry_with_one(accepted_on_chain); + registry.cancelling[0].agreement.updated_at = + time::OffsetDateTime::now_utc() - LISTENER_GRACE - time::Duration::MINUTE; + let chain = MockChain { + ended_by_indexer: true, + ..MockChain::default() + }; + + retry(®istry, &chain, DEADLINE + 1).await; + + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + assert_eq!(registry.marked_by_indexer.lock().unwrap().len(), 1); + let indexer = registry.cancelling[0].agreement.indexer.id.to_string(); + assert_eq!(*registry.audited_by.lock().unwrap(), vec![indexer]); + assert_eq!(*registry.audits.lock().unwrap(), vec![None]); + } + } + + #[tokio::test] + async fn does_not_confirm_an_end_when_who_ended_it_cannot_be_read() { + let mut registry = registry_with_one(true); + registry.cancelling[0].agreement.updated_at = + time::OffsetDateTime::now_utc() - LISTENER_GRACE - time::Duration::MINUTE; + let chain = MockChain { + who_read_fails: true, + ..MockChain::default() + }; + + retry(®istry, &chain, 0).await; + + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + assert!(registry.marked_by_indexer.lock().unwrap().is_empty()); + } + #[tokio::test] async fn keeps_an_unaccepted_agreement_cancelling_until_its_deadline() { // An offer still in flight could land and be accepted until then. diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index 1b16dc77..145f8cb2 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -2663,6 +2663,12 @@ mod tests { // not-active here means "cancel confirmed", which these tests expect. Ok(self.live_until_cancelled && !self.cancels.lock().unwrap().contains(agreement_id)) } + async fn agreement_ended_by_indexer( + &self, + _agreement_id: &[u8; 16], + ) -> Result { + Ok(false) + } } #[async_trait::async_trait] diff --git a/bin/dipper-service/src/network/service/escrow_reconciler.rs b/bin/dipper-service/src/network/service/escrow_reconciler.rs index f24fbaf2..5fba0f1a 100644 --- a/bin/dipper-service/src/network/service/escrow_reconciler.rs +++ b/bin/dipper-service/src/network/service/escrow_reconciler.rs @@ -698,6 +698,12 @@ mod tests { ) -> Result { unimplemented!() } + async fn agreement_ended_by_indexer( + &self, + _agreement_id: &[u8; 16], + ) -> Result { + Ok(false) + } async fn reconcile_agreement( &self, _collector: Address, diff --git a/bin/dipper-service/src/network/service/liveness_checker.rs b/bin/dipper-service/src/network/service/liveness_checker.rs index 34e43ac2..9e520241 100644 --- a/bin/dipper-service/src/network/service/liveness_checker.rs +++ b/bin/dipper-service/src/network/service/liveness_checker.rs @@ -1209,6 +1209,12 @@ mod tests { // not-active means "cancel confirmed", which these tests expect. Ok(false) } + async fn agreement_ended_by_indexer( + &self, + _agreement_id: &[u8; 16], + ) -> Result { + Ok(false) + } } const DB_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(5); diff --git a/bin/dipper-service/src/registry/agreement.rs b/bin/dipper-service/src/registry/agreement.rs index 36d2c9ae..1309d1d3 100644 --- a/bin/dipper-service/src/registry/agreement.rs +++ b/bin/dipper-service/src/registry/agreement.rs @@ -274,7 +274,9 @@ pub trait AgreementRegistry { ) -> RegistryResult<()>; /// `CANCELLING` agreements marked over `min_age_minutes` ago whose cancel has failed - /// fewer than `max_attempts` times, those checked longest ago first. + /// fewer than `max_attempts` times, those checked longest ago first. One that may be paying + /// an indexer (accepted, or past the offer deadline, which only an accepted one outlives) + /// counts as checked an hour earlier, so it goes first without holding the rest back. async fn get_cancelling_agreements( &self, batch_size: i64, diff --git a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs index 46ae9be2..651245c0 100644 --- a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs +++ b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs @@ -639,6 +639,12 @@ mod tests { } Ok(self.live.load(Ordering::SeqCst)) } + async fn agreement_ended_by_indexer( + &self, + _agreement_id: &[u8; 16], + ) -> Result { + Ok(false) + } } fn test_agreement_conf() -> Arc { diff --git a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs index 340bf26b..51b46811 100644 --- a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs +++ b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs @@ -1023,6 +1023,12 @@ mod lifecycle_event_tests { // Cancel confirmed: agreement is no longer active on-chain. Ok(false) } + async fn agreement_ended_by_indexer( + &self, + _agreement_id: &[u8; 16], + ) -> Result { + Ok(false) + } } // ---- Mock: registry (all five traits) ----------------------------------- @@ -2454,6 +2460,12 @@ mod deadline_clock_tests { ) -> Result { Ok(false) } + async fn agreement_ended_by_indexer( + &self, + _agreement_id: &[u8; 16], + ) -> Result { + Ok(false) + } } /// `fail: true` makes the state lookup itself error, exercising the diff --git a/bin/dipper-service/src/worker/handlers/selection_context.rs b/bin/dipper-service/src/worker/handlers/selection_context.rs index 0f65da62..3485463d 100644 --- a/bin/dipper-service/src/worker/handlers/selection_context.rs +++ b/bin/dipper-service/src/worker/handlers/selection_context.rs @@ -7,7 +7,9 @@ use thegraph_core::{DeploymentId, IndexerId, alloy::primitives::ChainId}; use crate::{ network::service::entity_count_cache::EntityCountCache, - registry::{AgreementRegistry, IndexerDenylistRegistry, IndexingAgreementStatus}, + registry::{ + AgreementRegistry, IndexerDenylistRegistry, IndexingAgreement, IndexingAgreementStatus, + }, worker::result::{JobError, JobResult}, }; @@ -42,11 +44,12 @@ where R: AgreementRegistry + IndexerDenylistRegistry, { // Get indexers that already have active agreements for this deployment - let existing_indexers = registry + let agreements = registry .get_indexing_agreements_by_deployment_id(deployment_id) .await - .map_err(|err| JobError::Fatal(err.into()))? - .into_iter() + .map_err(|err| JobError::Fatal(err.into()))?; + let existing_indexers = agreements + .iter() .filter(|a| is_active_agreement(&a.status)) .map(|a| a.indexer.id) .collect::>(); @@ -58,7 +61,7 @@ where .map_err(|err| JobError::Fatal(err.into()))?; // Get indexers that declined within their respective lookback periods - let declined_indexers = registry + let mut declined_indexers = registry .get_declined_indexers_by_deployment( declined_indexer_lookback_days, price_rejection_lookback_days, @@ -67,6 +70,7 @@ where ) .await .map_err(|err| JobError::Fatal(err.into()))?; + exclude_cancelling_indexers(&mut declined_indexers, *deployment_id, &agreements); // Get denied indexers that should be excluded from selection let indexer_denylist = registry @@ -173,6 +177,30 @@ fn wei_per_second_to_grt_per_28d(wei_per_second: f64) -> f64 { wei_per_second * SECONDS_PER_28_DAYS / WEI_PER_GRT } +/// Add to the deployment's declined list the indexers whose agreement dipper is still +/// cancelling. That agreement may still be live and paid on-chain, so its indexer must not +/// be picked again, but it no longer counts towards the group IISA sizes. +fn exclude_cancelling_indexers( + declined: &mut HashMap>, + deployment_id: DeploymentId, + agreements: &[IndexingAgreement], +) { + let cancelling = agreements + .iter() + .filter(|a| a.status == IndexingAgreementStatus::Cancelling) + .map(|a| a.indexer.id) + .collect::>(); + if cancelling.is_empty() { + return; + } + let excluded = declined.entry(deployment_id).or_default(); + for indexer in cancelling { + if !excluded.contains(&indexer) { + excluded.push(indexer); + } + } +} + /// Check if an agreement status represents an active agreement. fn is_active_agreement(status: &IndexingAgreementStatus) -> bool { matches!( @@ -184,7 +212,46 @@ fn is_active_agreement(status: &IndexingAgreementStatus) -> bool { #[cfg(test)] mod tests { use super::*; - use crate::registry::AgreementFeeRate; + use crate::{cancel_dispatch::tests::agreement, registry::AgreementFeeRate}; + + fn indexer(hex_digit: char) -> IndexerId { + format!("0x{}", hex_digit.to_string().repeat(40)) + .parse() + .unwrap() + } + + #[test] + fn an_indexer_still_being_cancelled_cannot_be_picked_again_for_the_deployment() { + let deployment: DeploymentId = "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" + .parse() + .unwrap(); + let mut cancelling = agreement(IndexingAgreementStatus::Cancelling, None); + cancelling.indexer.id = indexer('b'); + let mut accepted = agreement(IndexingAgreementStatus::AcceptedOnChain, None); + accepted.indexer.id = indexer('c'); + let mut declined = HashMap::from([(deployment, vec![indexer('a')])]); + + exclude_cancelling_indexers(&mut declined, deployment, &[cancelling, accepted]); + + assert_eq!(declined[&deployment], vec![indexer('a'), indexer('b')]); + } + + #[test] + fn a_cancelling_indexer_already_declined_is_listed_once_and_none_adds_no_entry() { + let deployment: DeploymentId = "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" + .parse() + .unwrap(); + let mut cancelling = agreement(IndexingAgreementStatus::Cancelling, None); + cancelling.indexer.id = indexer('a'); + let mut declined = HashMap::from([(deployment, vec![indexer('a')])]); + exclude_cancelling_indexers(&mut declined, deployment, &[cancelling]); + assert_eq!(declined[&deployment], vec![indexer('a')]); + + let mut none_declined = HashMap::new(); + let accepted = agreement(IndexingAgreementStatus::AcceptedOnChain, None); + exclude_cancelling_indexers(&mut none_declined, deployment, &[accepted]); + assert!(none_declined.is_empty()); + } #[test] fn test_wei_per_second_to_grt_per_28d() { diff --git a/bin/dipper-service/src/worker/handlers/submit_offer.rs b/bin/dipper-service/src/worker/handlers/submit_offer.rs index e27f3d74..2a24a1a4 100644 --- a/bin/dipper-service/src/worker/handlers/submit_offer.rs +++ b/bin/dipper-service/src/worker/handlers/submit_offer.rs @@ -418,6 +418,12 @@ mod tests { ) -> Result { Ok(self.on_chain.load(Ordering::SeqCst)) } + async fn agreement_ended_by_indexer( + &self, + _agreement_id: &[u8; 16], + ) -> Result { + Ok(false) + } async fn latest_block_timestamp(&self) -> Result { unimplemented!() } diff --git a/dipper-pgregistry/src/postgres.rs b/dipper-pgregistry/src/postgres.rs index d40439b3..fb36f95b 100644 --- a/dipper-pgregistry/src/postgres.rs +++ b/dipper-pgregistry/src/postgres.rs @@ -972,7 +972,9 @@ impl PgRegistry { } /// `Cancelling` agreements marked over `min_age_minutes` ago whose cancel has failed - /// fewer than `max_attempts` times, those checked longest ago first. + /// fewer than `max_attempts` times, those checked longest ago first. One that may be paying + /// an indexer (accepted, or past the offer deadline, which only an accepted one outlives) + /// counts as checked an hour earlier, so it goes first without holding the rest back. pub async fn get_cancelling_agreements( &self, batch_size: i64, @@ -1001,7 +1003,14 @@ impl PgRegistry { WHERE status = $1 AND cancel_attempts < $2 AND updated_at < timezone('UTC', now()) - make_interval(mins => $4) - ORDER BY cancel_checked_at ASC NULLS FIRST, updated_at ASC + ORDER BY + cancel_checked_at - CASE + WHEN accepted_at IS NOT NULL + OR CAST(terms->>'deadline' AS bigint) < EXTRACT(EPOCH FROM now()) + THEN INTERVAL '1 hour' + ELSE INTERVAL '0 seconds' + END ASC NULLS FIRST, + updated_at ASC LIMIT $3 "#, ) diff --git a/dipper-pgregistry/tests/it_registry_postgres.rs b/dipper-pgregistry/tests/it_registry_postgres.rs index 18530ded..ff1de13f 100644 --- a/dipper-pgregistry/tests/it_registry_postgres.rs +++ b/dipper-pgregistry/tests/it_registry_postgres.rs @@ -3361,8 +3361,18 @@ async fn cancelling_agreements_are_listed_until_their_cancel_fails_too_often() { ) .await .expect("Failed to run fixture"); - let registry = PgRegistry::new(db); let created = fixture_agreement(0xaa); + // An offer still open to acceptance, so only the accepted agreement can be paying. + sqlx::query( + "UPDATE dipper_reg_indexing_agreements \ + SET terms = jsonb_set(terms::jsonb, '{deadline}', to_jsonb(4102444800::bigint)) \ + WHERE id = $1", + ) + .bind(created) + .execute(&db) + .await + .expect("Failed to update deadline"); + let registry = PgRegistry::new(db.clone()); let accepted = IndexingAgreementId::from_bytes([0xaa, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2]); let ended = @@ -3421,16 +3431,36 @@ async fn cancelling_agreements_are_listed_until_their_cancel_fails_too_often() { ] ); + assert_eq!(registry.record_cancel_check(&created, 0).await.unwrap(), 0); assert_eq!(registry.record_cancel_check(&accepted, 0).await.unwrap(), 0); let listed = registry .get_cancelling_agreements(100, 2, 0) .await .expect("cancelling query"); let ids: Vec<_> = listed.iter().map(|row| row.agreement.id).collect(); + assert_eq!( + ids, + vec![accepted, created], + "one accepted on-chain may be paying its indexer, so it goes first" + ); + + sqlx::query( + "UPDATE dipper_reg_indexing_agreements \ + SET cancel_checked_at = cancel_checked_at - INTERVAL '2 hours' WHERE id = $1", + ) + .bind(created) + .execute(&db) + .await + .expect("Failed to age the check"); + let listed = registry + .get_cancelling_agreements(100, 2, 0) + .await + .expect("cancelling query"); + let ids: Vec<_> = listed.iter().map(|row| row.agreement.id).collect(); assert_eq!( ids, vec![created, accepted], - "one checked longest ago goes first" + "an offer unchecked for over an hour isn't held back for ever" ); assert_eq!(registry.record_cancel_check(&created, 1).await.unwrap(), 1); From 8c5e5436e87d04b830b8f7d4a61af5c02e9f93d5 Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Sat, 3 Oct 2026 01:12:31 +0300 Subject: [PATCH 09/26] fix: read agreement state from the chain correctly (#719) * fix(chain): stop treating an expired, unaccepted offer as live The contract keeps an offer stored after its deadline, flagging that nothing can be claimed from it, but dipper read any stored offer as live. It then paid to cancel offers nobody could accept any more, and lost their expired status. An offer with that flag now reads as not live. * fix(listener): stop recording a withdrawn offer as accepted When an offer is withdrawn before anyone accepts it, the subgraph reports it as cancelled with no accept time, and dipper still marked it accepted, so accepted and terminated events went out for an agreement that was never live. An offer cancelled with no accept time is no longer an accept. * fix(cancel): time cancel retries by the chain, not the subgraph The cancel retry decided whether an unaccepted offer's deadline had passed using the subgraph's latest block time, which stops moving while the subgraph is down, so those offers stayed cancelling until it recovered. It now reads the chain's own latest block time instead. * fix(chain): never read an agreement from an RPC endpoint that is behind Dipper read an agreement's state from whichever RPC endpoint it was using, so a lagging fallback could report an agreement as not live after dipper had seen it go live. Each read now checks the endpoint has reached the newest block dipper has seen, and moves to the next endpoint if not. * fix(chain): wait for an RPC endpoint a block behind to catch up A read refused because the endpoint was behind, or because its node lacked the block asked for, was treated as a hard failure, so dipper gave up on that endpoint at once. Hosted endpoints are often a block behind for a moment, so both are now retried on the same endpoint with backoff. --- bin/dipper-service/src/chain_client/client.rs | 198 ++++++++++++++++-- .../src/chain_client/rpc_provider.rs | 16 ++ .../src/network/service/cancel_retry.rs | 48 ++++- .../src/network/service/chain_listener.rs | 49 ++++- 4 files changed, 289 insertions(+), 22 deletions(-) diff --git a/bin/dipper-service/src/chain_client/client.rs b/bin/dipper-service/src/chain_client/client.rs index 8aae2df7..9cc58a9d 100644 --- a/bin/dipper-service/src/chain_client/client.rs +++ b/bin/dipper-service/src/chain_client/client.rs @@ -14,21 +14,21 @@ use std::{ use async_trait::async_trait; use dipper_rpc::indexer::indexer_client::sol::RecurringCollectionAgreement; use thegraph_core::alloy::{ - eips::{BlockNumberOrTag, eip2718::Encodable2718}, + eips::{BlockId, BlockNumberOrTag, eip2718::Encodable2718}, network::{EthereumWallet, TransactionBuilder}, primitives::{Address, B256, FixedBytes, U256}, providers::Provider, rpc::types::TransactionRequest, signers::local::PrivateKeySigner, sol_types::{SolCall, SolValue}, - transports::TransportError, + transports::{TransportError, TransportErrorKind}, }; use tokio::sync::Mutex; use super::{ abi::{IRecurringAgreementManager, IRecurringCollector}, gas::{GasEstimator, calculate_max_fee, exceeds_max_gas_price, get_gas_prices}, - rpc_provider::RpcProviderPool, + rpc_provider::{BEHIND_A_SEEN_BLOCK, RpcProviderPool}, }; use crate::{ chain_client::{ @@ -58,21 +58,23 @@ const RECEIPT_POLL_INTERVAL: Duration = Duration::from_millis(500); const VERSION_CURRENT: u64 = 0; /// `AgreementDetails.state` flags from `IAgreementCollector.sol` (REGISTERED=1, -/// ACCEPTED=2, NOTICE_GIVEN=4, BY_PROVIDER=32). `getAgreementDetails` keeps ACCEPTED set on -/// a canceled agreement and ORs in NOTICE_GIVEN, so a cancel must clear it, not just lack -/// it. REGISTERED without ACCEPTED is an offer still waiting. BY_PROVIDER: the indexer cancelled. +/// ACCEPTED=2, NOTICE_GIVEN=4, SETTLED=8, BY_PROVIDER=32). `getAgreementDetails` keeps +/// ACCEPTED set on a canceled agreement and ORs in NOTICE_GIVEN, so a cancel must clear it, +/// not just lack it. SETTLED: nothing left to claim. BY_PROVIDER: the indexer cancelled. const STATE_REGISTERED: u16 = 1; const STATE_ACCEPTED: u16 = 2; const STATE_NOTICE_GIVEN: u16 = 4; +const STATE_SETTLED: u16 = 8; const STATE_BY_PROVIDER: u16 = 32; /// Live iff the terms are accepted and no cancellation notice exists, or an /// offer is still stored for the indexer to accept. A cancel sets NOTICE_GIVEN /// while ACCEPTED stays set, so the notice bit tells a live agreement from a -/// cancelled one; a revoked offer reads as an empty state. +/// cancelled one; a revoked offer reads as an empty state. An offer past its +/// deadline stays stored but is SETTLED, since it can no longer be accepted. fn still_live(state: u16) -> bool { let accepted = state & STATE_ACCEPTED != 0; - let pending_offer = state & STATE_REGISTERED != 0 && !accepted; + let pending_offer = state & STATE_REGISTERED != 0 && !accepted && state & STATE_SETTLED == 0; pending_offer || (accepted && state & STATE_NOTICE_GIVEN == 0) } @@ -238,6 +240,9 @@ struct AlloyChainClientInner { submit_lock: Mutex<()>, /// How long one submission may hold `submit_lock`; see `derive_submit_deadline`. submit_deadline: Duration, + /// The newest block dipper has seen, from reads and receipts. An agreement's state is + /// never read from an endpoint behind it, so a lagging endpoint can't undo what dipper saw. + seen_block: AtomicU64, } impl AlloyChainClient { @@ -288,6 +293,7 @@ impl AlloyChainClient { nonce: AtomicU64::new(NONCE_UNINITIALIZED), submit_lock: Mutex::new(()), submit_deadline, + seen_block: AtomicU64::new(0), }), }) } @@ -611,11 +617,51 @@ impl AlloyChainClient { }; let collector = self.inner.recurring_collector_address; Ok(self - .view(collector, call, "get_agreement_details") + .view_at_seen_block(collector, call, "get_agreement_details") .await? .state) } + /// Run a read-only contract call at the endpoint's latest block, refusing an endpoint + /// whose latest block is older than one dipper has already seen; the pool moves on to + /// the next endpoint instead. + async fn view_at_seen_block( + &self, + to: Address, + call: C, + operation: &'static str, + ) -> Result { + let calldata = call.abi_encode(); + let seen = self.inner.seen_block.load(Ordering::Relaxed); + let (head, output) = self + .inner + .rpc_pool + .execute(operation, |provider| { + let calldata = calldata.clone(); + async move { + let head = provider.get_block_number().await?; + if head < seen { + return Err(TransportErrorKind::custom_str(&format!( + "endpoint is at block {head}, {BEHIND_A_SEEN_BLOCK} ({seen})" + ))); + } + let tx = TransactionRequest::default().to(to).input(calldata.into()); + let output = provider.call(tx).block(BlockId::number(head)).await?; + Ok((head, output)) + } + }) + .await?; + self.note_block(head); + C::abi_decode_returns(&output).map_err(|err| { + ChainClientError::RpcError(anyhow::anyhow!("undecodable {operation} from {to}: {err}")) + }) + } + + /// Remember a block dipper has seen, so later reads are never older. + fn note_block(&self, block: u64) { + self.inner.seen_block.fetch_max(block, Ordering::Relaxed); + } + /// Run a read-only contract call and decode its return value. async fn view( &self, @@ -659,7 +705,12 @@ impl AlloyChainClient { .await; match receipt { - Ok(Some(r)) => return Ok(Some(r.status())), + Ok(Some(r)) => { + if let Some(block) = r.block_number { + self.note_block(block); + } + return Ok(Some(r.status())); + } Ok(None) => {} // not mined yet Err(e) => { // Transient RPC error: log and keep polling. If it persists, the outer @@ -693,6 +744,7 @@ impl ChainClient for AlloyChainClient { .ok_or_else(|| { ChainClientError::RpcError(anyhow::anyhow!("no latest block returned")) })?; + self.note_block(block.header.number); Ok(block.header.timestamp) } @@ -1016,15 +1068,22 @@ mod tests { /// or an id it never saw, as an empty state. #[test] fn still_live_covers_accepted_agreements_and_pending_offers() { - const SETTLED: u16 = 8; const BY_PAYER: u16 = 16; assert!(still_live(STATE_REGISTERED | STATE_ACCEPTED)); assert!( still_live(STATE_REGISTERED), "a pending offer can still be accepted" ); + assert!( + !still_live(STATE_REGISTERED | STATE_SETTLED), + "an offer past its deadline can't be" + ); + assert!( + still_live(STATE_REGISTERED | STATE_ACCEPTED | STATE_SETTLED), + "an accepted agreement just collected from is still live" + ); assert!(!still_live( - STATE_REGISTERED | STATE_ACCEPTED | STATE_NOTICE_GIVEN | BY_PAYER | SETTLED + STATE_REGISTERED | STATE_ACCEPTED | STATE_NOTICE_GIVEN | BY_PAYER | STATE_SETTLED )); assert!(!still_live(0), "revoked or never offered"); } @@ -2021,6 +2080,121 @@ mod tests { } } + /// Answers as an endpoint whose latest block is `head`, reporting `state` for any + /// agreement read at that block. + struct AgreementStateResponder { + head: AtomicU64, + /// Blocks the endpoint gains each time it reports its head. + catch_up: u64, + state: u16, + } + + impl Respond for AgreementStateResponder { + fn respond(&self, request: &Request) -> ResponseTemplate { + let body: serde_json::Value = + serde_json::from_slice(&request.body).expect("JSON-RPC request body"); + let result = match body["method"].as_str().unwrap_or_default() { + "eth_blockNumber" => { + let head = self.head.fetch_add(self.catch_up, Ordering::SeqCst); + format!("{head:#x}") + } + "eth_call" => { + let at = body["params"][1].as_str().expect("a block number"); + let at = u64::from_str_radix(at.trim_start_matches("0x"), 16).expect("hex"); + assert!( + at <= self.head.load(Ordering::SeqCst), + "read at a block the endpoint has" + ); + let details = IRecurringCollector::AgreementDetails { + agreementId: FixedBytes::<16>::ZERO, + payer: Address::ZERO, + dataService: Address::ZERO, + serviceProvider: Address::ZERO, + versionHash: B256::ZERO, + state: self.state, + }; + let output = + IRecurringCollector::getAgreementDetailsCall::abi_encode_returns(&details); + format!( + "0x{}", + thegraph_core::alloy::primitives::hex::encode(output) + ) + } + other => panic!("unexpected call {other}"), + }; + ResponseTemplate::new(200).set_body_json(serde_json::json!({ + "jsonrpc": "2.0", + "id": body["id"], + "result": result, + })) + } + } + + async fn server_at_block(head: u64, state: u16) -> MockServer { + server_catching_up(head, 0, state).await + } + + async fn server_catching_up(head: u64, catch_up: u64, state: u16) -> MockServer { + let server = MockServer::start().await; + Mock::given(method("POST")) + .respond_with(AgreementStateResponder { + head: AtomicU64::new(head), + catch_up, + state, + }) + .mount(&server) + .await; + server + } + + #[tokio::test] + async fn waits_for_an_endpoint_a_block_behind_to_catch_up() { + // Hosted endpoints spread calls across nodes, so one a block behind is routine + // rather than a reason to give up on the endpoint. + let endpoint = server_catching_up(94, 1, STATE_REGISTERED | STATE_ACCEPTED).await; + let client = client_over_retrying(vec![endpoint.uri().parse().expect("provider URL")], 1); + client.note_block(95); + + let live = client + .agreement_still_active(&[0xab; 16]) + .await + .expect("read once the endpoint caught up"); + + assert!(live); + } + + #[tokio::test] + async fn never_reads_an_agreement_from_an_endpoint_behind_a_block_already_seen() { + // The lagging endpoint still shows the offer it hasn't seen accepted and cancelled. + let lagging = server_at_block(90, STATE_REGISTERED).await; + let current = + server_at_block(100, STATE_REGISTERED | STATE_ACCEPTED | STATE_NOTICE_GIVEN).await; + let client = client_over(vec![ + lagging.uri().parse().expect("provider URL"), + current.uri().parse().expect("provider URL"), + ]); + client.note_block(95); + + let live = client + .agreement_still_active(&[0xab; 16]) + .await + .expect("read"); + + assert!(!live, "read from the endpoint that has reached block 95"); + assert_eq!(client.inner.seen_block.load(Ordering::Relaxed), 100); + } + + #[tokio::test] + async fn fails_a_read_rather_than_answer_from_endpoints_all_behind() { + let lagging = server_at_block(90, STATE_REGISTERED).await; + let client = client_over(vec![lagging.uri().parse().expect("provider URL")]); + client.note_block(95); + + let read = client.agreement_still_active(&[0xab; 16]).await; + + assert!(read.is_err(), "got {read:?}"); + } + async fn client_over_manager( responder: ManagerViewsResponder, ) -> (AlloyChainClient, MockServer) { diff --git a/bin/dipper-service/src/chain_client/rpc_provider.rs b/bin/dipper-service/src/chain_client/rpc_provider.rs index 98954d46..b2f280b7 100644 --- a/bin/dipper-service/src/chain_client/rpc_provider.rs +++ b/bin/dipper-service/src/chain_client/rpc_provider.rs @@ -56,6 +56,10 @@ fn describe_failure(url: &Url, error: &TransportError) -> String { /// Error text that indicates a transient failure worth retrying, used only for faults /// that arrive as prose rather than as a status code or JSON-RPC error object. const RETRYABLE_ERROR_PATTERNS: &[&str] = &[ + BEHIND_A_SEEN_BLOCK, + // A node behind the rest of its provider's fleet, asked for a block it hasn't reached. + "header not found", + "unknown block", "connection refused", "connection reset", "connection closed", @@ -68,6 +72,10 @@ const RETRYABLE_ERROR_PATTERNS: &[&str] = &[ "temporary internal error", ]; +/// How a read refused by an endpoint behind a block dipper has already seen describes it. It +/// is retried, since an endpoint a block or so behind catches up within a second or two. +pub(super) const BEHIND_A_SEEN_BLOCK: &str = "behind a block already seen"; + /// Type alias for the provider with default fillers. pub type HttpProvider = FillProvider< JoinFill< @@ -633,6 +641,14 @@ mod tests { ); } + #[test] + fn a_node_that_has_not_reached_a_block_yet_is_retryable() { + let payload = serde_json::from_str(r#"{"code":-32000,"message":"header not found"}"#) + .expect("JSON-RPC error payload"); + let err: TransportError = RpcError::ErrorResp(payload); + assert!(RpcProviderPool::is_retryable(&err)); + } + #[test] fn each_retry_waits_twice_as_long_up_to_a_ceiling() { // 1s, 2s, 4s, 8s, 16s, 32s->30s diff --git a/bin/dipper-service/src/network/service/cancel_retry.rs b/bin/dipper-service/src/network/service/cancel_retry.rs index dc4874b2..4ea06248 100644 --- a/bin/dipper-service/src/network/service/cancel_retry.rs +++ b/bin/dipper-service/src/network/service/cancel_retry.rs @@ -32,17 +32,20 @@ const SETTLE_MINUTES: i32 = 2; /// agreement ended, before the retry marks it ended without those details. const LISTENER_GRACE: time::Duration = time::Duration::HOUR; -/// Retry the cancel of agreements still `Cancelling`. `chain_now`, in chain seconds, -/// decides when an offer that was never accepted no longer can be. +/// Retry the cancel of agreements still `Cancelling`. The chain's own latest block time +/// decides when an offer that was never accepted no longer can be, so a subgraph that has +/// fallen behind doesn't hold that up. pub async fn retry_cancelling_agreements( registry: &R, chain_client: &T, config: &IndexingAgreementConfig, - chain_now: u64, ) where R: AgreementRegistry + Sync, T: ChainClient, { + let Some(chain_now) = chain_time(chain_client).await else { + return; + }; let cancelling = match registry .get_cancelling_agreements(BATCH_SIZE, MAX_CANCEL_ATTEMPTS, SETTLE_MINUTES) .await @@ -66,6 +69,19 @@ pub async fn retry_cancelling_agreements( } } +async fn chain_time(chain_client: &T) -> Option { + match chain_client.latest_block_timestamp().await { + Ok(chain_now) => Some(chain_now), + Err(err) => { + tracing::warn!( + error = %err, + "Failed to read the chain's time; cancels are retried next sweep" + ); + None + } + } +} + async fn retry_cancel( registry: &R, chain_client: &T, @@ -322,7 +338,7 @@ fn failed_attempts(err: &ChainClientError) -> u32 { mod tests { use std::sync::{ Mutex, - atomic::{AtomicBool, AtomicU32, Ordering}, + atomic::{AtomicBool, AtomicU32, AtomicU64, Ordering}, }; use async_trait::async_trait; @@ -411,6 +427,8 @@ mod tests { send_fails: bool, mined_cancel_reverts: bool, cancel_has_no_effect: bool, + clock_fails: bool, + now: AtomicU64, ended_by_indexer: bool, who_read_fails: bool, cancels_sent: AtomicU32, @@ -478,7 +496,10 @@ mod tests { Ok(self.ended_by_indexer) } async fn latest_block_timestamp(&self) -> Result { - unimplemented!() + if self.clock_fails { + return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); + } + Ok(self.now.load(Ordering::SeqCst)) } } @@ -503,7 +524,22 @@ mod tests { async fn retry(registry: &MockRegistry, chain: &MockChain, chain_now: u64) { let config = IndexingAgreementConfig::for_tests(); - retry_cancelling_agreements(registry, chain, &config, chain_now).await; + chain.now.store(chain_now, Ordering::SeqCst); + retry_cancelling_agreements(registry, chain, &config).await; + } + + #[tokio::test] + async fn waits_for_the_next_sweep_when_the_chain_time_cannot_be_read() { + let registry = registry_with_one(true); + let chain = MockChain { + clock_fails: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 0); + assert_eq!(registry.checks.load(Ordering::SeqCst), 0); } #[tokio::test] diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index 145f8cb2..278db264 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -301,13 +301,10 @@ where // finishing a cancel needs only the chain. if last_cancel_retry.is_none_or(|at| at.elapsed() >= CANCEL_RETRY_INTERVAL) { last_cancel_retry = Some(Instant::now()); - let chain_now = - last_persisted_timestamp.unwrap_or_else(dipper_core::time::now_secs); super::cancel_retry::retry_cancelling_agreements( ®istry, &chain_client, &agreement_conf, - chain_now, ) .await; } @@ -901,10 +898,13 @@ where // Both transitions are applied atomically downstream so the // Accept-then-Cancel-in-one-snapshot path can't leak an intermediate // AcceptedOnChain to concurrent readers. + // A withdrawn offer reads as cancelled with no accept time: it never went live. + let withdrawn_offer = snapshot.state.is_canceled() && snapshot.accepted_at == 0; let apply_accept = matches!( agreement.status, IndexingAgreementStatus::Created | IndexingAgreementStatus::Expired, - ) && snapshot.state.reached_accepted(); + ) && snapshot.state.reached_accepted() + && !withdrawn_offer; let already_terminal_cancel = matches!( agreement.status, @@ -2824,6 +2824,47 @@ mod tests { ); } + #[tokio::test] + async fn reconcile_does_not_accept_an_offer_withdrawn_before_anyone_accepted_it() { + // Announcing it would send accepted and terminated events for an agreement that + // was never live; an expired one stays expired. + for status in [ + IndexingAgreementStatus::Created, + IndexingAgreementStatus::Expired, + ] { + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let worker_queue = MockWorkerQueue::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, status); + let mut snapshot = + make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); + snapshot.accepted_at = 0; + + reconcile_agreement( + &snapshot, + ®istry, + &worker_queue, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await + .expect("reconcile ok"); + + assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); + assert_eq!( + registry.was_marked_canceled_by_requester(&agreement_id), + status == IndexingAgreementStatus::Created + ); + assert!( + registry + .audit_writes() + .iter() + .all(|(kind, _)| *kind != "accept") + ); + } + } + #[tokio::test] async fn reconcile_marks_an_agreement_being_cancelled_once_the_chain_shows_it_ended() { for (state, ended_by_dipper) in [ From efd4abb533927d35e87a0511d04c2ae5d2368bca Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Sat, 3 Oct 2026 01:12:32 +0300 Subject: [PATCH 10/26] fix: send every cancel through the cancelling status (#720) * refactor(cancel): share one step for marking a cancel dipper confirmed Both the first cancel attempt and the cancel retry marked an agreement cancelled by dipper, logged it and recorded the cancel, each with its own copy that could drift apart. They now share one step, which records the cancel whenever its transaction is known. * fix(cancel): read the chain before sending a cancel Starting a cancel sent it blind, so cancelling offers that never reached the chain cost gas each, and the reassessment waited on every receipt while holding its lock. The agreement is still marked first, so an offer that lands later withdraws itself, but a cancel now goes out only if it is live. * fix(cancel): finish a live cancelled agreement through the cancel retry An agreement dipper had rejected or cancelled that the subgraph showed accepted went to a separate job with its own 40 minute retry budget, queued again on every poll. The listener now reads the chain and, if it is live, moves it back to cancelling so the one cancel retry ends it. --- .../handlers/indexing_requests.rs | 8 - bin/dipper-service/src/cancel_dispatch.rs | 95 +- bin/dipper-service/src/main.rs | 1 - .../src/network/service/cancel_retry.rs | 27 +- .../src/network/service/chain_listener.rs | 297 ++---- .../src/network/service/expiration.rs | 7 - .../src/network/service/liveness_checker.rs | 7 - bin/dipper-service/src/registry.rs | 10 + bin/dipper-service/src/registry/agreement.rs | 8 + .../src/registry/agreement_stub.rs | 8 + bin/dipper-service/src/worker/context.rs | 1 - .../cancel_rejected_agreement_on_chain.rs | 951 ++---------------- .../handlers/reassess_indexing_request.rs | 42 +- .../send_indexing_agreement_proposal.rs | 15 +- .../src/worker/service_queue.rs | 58 +- dipper-pgregistry/src/postgres.rs | 29 + .../tests/it_registry_postgres.rs | 41 + 17 files changed, 388 insertions(+), 1217 deletions(-) diff --git a/bin/dipper-service/src/admin_rpc_server/handlers/indexing_requests.rs b/bin/dipper-service/src/admin_rpc_server/handlers/indexing_requests.rs index 17666cbd..b846dd8e 100644 --- a/bin/dipper-service/src/admin_rpc_server/handlers/indexing_requests.rs +++ b/bin/dipper-service/src/admin_rpc_server/handlers/indexing_requests.rs @@ -364,14 +364,6 @@ mod tests { Ok(JobId::default()) } - async fn cancel_rejected_agreement_on_chain( - &self, - _agreement_id: IndexingAgreementId, - _priority: crate::worker::queue::JobPriority, - ) -> anyhow::Result { - unimplemented!() - } - async fn submit_offer( &self, _agreement_id: IndexingAgreementId, diff --git a/bin/dipper-service/src/cancel_dispatch.rs b/bin/dipper-service/src/cancel_dispatch.rs index 4f6734a2..b35e8a20 100644 --- a/bin/dipper-service/src/cancel_dispatch.rs +++ b/bin/dipper-service/src/cancel_dispatch.rs @@ -71,9 +71,9 @@ pub enum CancelStarted { Cancelling, } -/// Start ending an agreement that may be live on-chain. It is marked `Cancelling` before -/// its cancel goes out, so an offer for it still in flight withdraws itself on landing. -/// Fails, sending nothing, when the mark can't be written. +/// Start ending an agreement that may be live on-chain. It is marked `Cancelling` first, so an +/// offer for it still in flight withdraws itself on landing, then cancelled only if the chain +/// shows it live. Fails, sending nothing, when the mark can't be written. pub async fn start_cancel( registry: &R, chain_client: &T, @@ -87,9 +87,10 @@ where registry .mark_indexing_agreement_as_cancelling(&agreement.id) .await?; - let tx_hash = match cancel_agreement_on_chain(chain_client, agreement, config).await { - Ok(tx_hash) => tx_hash, - Err(err) => { + let tx_hash = match cancel_if_live(chain_client, agreement, config).await { + LiveCancel::Ended(tx_hash) => tx_hash, + LiveCancel::NotLive => return Ok(CancelStarted::Cancelling), + LiveCancel::ReadFailed(err) | LiveCancel::CancelFailed(err) => { tracing::warn!( agreement_id = %agreement.id, error = %err, @@ -107,17 +108,24 @@ where if agreement.status != IndexingAgreementStatus::AcceptedOnChain { return Ok(CancelStarted::Cancelling); } - Ok(confirm_cancelled(registry, agreement, tx_hash, config).await) + Ok( + if confirm_cancelled(registry, agreement, tx_hash, config).await { + CancelStarted::Ended + } else { + CancelStarted::Cancelling + }, + ) } -/// Mark an accepted agreement whose cancel landed `CanceledByRequester` and record the -/// cancel, so the `terminated` sweep announces it. -async fn confirm_cancelled( +/// Mark an agreement the chain shows dipper ended `CanceledByRequester`, recording the cancel +/// when its transaction is known, so the `terminated` sweep announces it. False, logged, when +/// the mark fails; it stays `Cancelling` for the cancel retry. +pub async fn confirm_cancelled( registry: &R, agreement: &IndexingAgreement, tx_hash: Option, config: &IndexingAgreementConfig, -) -> CancelStarted { +) -> bool { if let Err(err) = registry .mark_indexing_agreement_as_canceled_by_requester(&agreement.id) .await @@ -125,17 +133,27 @@ async fn confirm_cancelled( tracing::warn!( agreement_id = %agreement.id, error = %err, - "Failed to mark a cancelled agreement; the chain listener finishes it" + "Failed to mark an ended agreement cancelled; the cancel retry tries again" ); - return CancelStarted::Cancelling; + return false; } - record_cancel(registry, agreement, tx_hash, config).await; - CancelStarted::Ended + tracing::info!( + agreement_id = %agreement.id, + indexing_request_id = %agreement.indexing_request_id, + old_status = "CANCELLING", + new_status = "CANCELED_BY_REQUESTER", + reason = "cancel_confirmed_on_chain", + "agreement state transition" + ); + if tx_hash.is_some() { + record_cancel(registry, agreement, tx_hash, config).await; + } + true } /// Record dipper's own cancel of an accepted agreement, so the `terminated` sweep /// announces it. -pub async fn record_cancel( +async fn record_cancel( registry: &R, agreement: &IndexingAgreement, tx_hash: Option, @@ -155,6 +173,51 @@ pub async fn record_cancel( } } +/// Move an agreement dipper had already rejected or cancelled back into `Cancelling` when the +/// chain shows it live after all, so the cancel retry ends it. The chain is read first, so a +/// subgraph report from before dipper's cancel landed reopens nothing; an unreadable chain +/// reopens it anyway, as the retry reads again before sending. True if it was reopened. +pub async fn reopen_if_live( + registry: &R, + chain_client: &T, + agreement: &IndexingAgreement, +) -> RegistryResult +where + R: AgreementRegistry + Sync, + T: ChainClient, +{ + match chain_client + .agreement_still_active(agreement.id.as_bytes()) + .await + { + Ok(false) => return Ok(false), + Ok(true) => {} + Err(err) => tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to read an ended agreement reported live; the cancel retry checks it" + ), + } + match registry + .reopen_indexing_agreement_cancel(&agreement.id) + .await + { + Ok(()) => {} + Err(crate::registry::Error::NoRecordsUpdated) => return Ok(false), + Err(err) => return Err(err), + } + tracing::warn!( + agreement_id = %agreement.id, + indexer_id = %agreement.indexer.id, + indexing_request_id = %agreement.indexing_request_id, + old_status = %agreement.status, + new_status = "CANCELLING", + reason = "live_on_chain_after_end", + "agreement state transition" + ); + Ok(true) +} + /// What [`cancel_if_live`] found and did. #[derive(Debug)] pub enum LiveCancel { diff --git a/bin/dipper-service/src/main.rs b/bin/dipper-service/src/main.rs index 8b22c649..e6a243f1 100644 --- a/bin/dipper-service/src/main.rs +++ b/bin/dipper-service/src/main.rs @@ -522,7 +522,6 @@ pub async fn main() -> anyhow::Result<()> { let ctx = network::service::chain_listener::Ctx { registry: registry.clone(), - worker_queue: worker_handle.queue().clone(), event_source, chain_client: chain_client.clone(), agreement_conf: chain_listener_agreement_conf.clone(), diff --git a/bin/dipper-service/src/network/service/cancel_retry.rs b/bin/dipper-service/src/network/service/cancel_retry.rs index 4ea06248..5b37a2dd 100644 --- a/bin/dipper-service/src/network/service/cancel_retry.rs +++ b/bin/dipper-service/src/network/service/cancel_retry.rs @@ -6,7 +6,7 @@ use dipper_core::time::now_secs; use thegraph_core::alloy::primitives::B256; use crate::{ - cancel_dispatch::{LiveCancel, cancel_if_live, record_cancel}, + cancel_dispatch::{LiveCancel, cancel_if_live, confirm_cancelled}, chain_client::{ChainClient, ChainClientError}, config::IndexingAgreementConfig, registry::{AgreementRegistry, CancelKind, CancellingAgreement, IndexingAgreement}, @@ -154,30 +154,7 @@ where Some(false) => {} } } - if let Err(err) = registry - .mark_indexing_agreement_as_canceled_by_requester(&agreement.id) - .await - { - tracing::warn!( - agreement_id = %agreement.id, - error = %err, - "Failed to mark an ended agreement cancelled, will retry" - ); - return false; - } - tracing::info!( - agreement_id = %agreement.id, - indexing_request_id = %agreement.indexing_request_id, - old_status = "CANCELLING", - new_status = "CANCELED_BY_REQUESTER", - reason = "cancel_confirmed_on_chain", - "agreement state transition" - ); - // Without its own transaction, a late read by the chain listener fills in the cancel. - if row.accepted_on_chain && tx_hash.is_some() { - record_cancel(registry, agreement, tx_hash, config).await; - } - true + confirm_cancelled(registry, agreement, tx_hash, config).await } /// Whether the chain shows the indexer ended the agreement, or `None` when it can't be read, diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index 278db264..585626ca 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -56,7 +56,6 @@ use crate::{ AgreementRegistry, CancelKind, IndexingAgreement, IndexingAgreementStatus, PendingCancellationRegistry, ReconciliationItem, }, - worker::service::{JobPriority, WorkerQueue}, }; /// Idle interval used when no `Created` agreements are awaiting acceptance. @@ -106,11 +105,9 @@ impl Handle { } /// Context required by the chain listener service -pub struct Ctx { +pub struct Ctx { /// Registry for querying and updating agreements pub registry: R, - /// Worker queue (still used by reconciliation paths that hand work back to the worker) - pub worker_queue: W, /// Chain event source (subgraph) pub event_source: E, /// Chain client used to cancel on-chain via the RecurringAgreementManager @@ -150,10 +147,9 @@ pub struct ChainListenerState { clippy::too_many_lines, reason = "predates this lint; fix when next touched" )] -pub fn new(ctx: Ctx) -> (Handle, impl Future>) +pub fn new(ctx: Ctx) -> (Handle, impl Future>) where R: AgreementRegistry + ChainListenerStateRegistry + PendingCancellationRegistry + Send + Sync, - W: WorkerQueue + Send + Sync, E: ChainEventSource, T: ChainClient + Send + Sync, { @@ -161,7 +157,6 @@ where let Ctx { registry, - worker_queue, event_source, chain_client, agreement_conf, @@ -321,7 +316,6 @@ where chain_ts_drift_tolerance_secs, bypass_chain_clock_defenses, ®istry, - &worker_queue, &chain_client, &event_source, &mut rx_stop, @@ -440,7 +434,7 @@ struct DrainOutcome { clippy::too_many_lines, reason = "predates this lint; fix when next touched" )] -async fn drain_once( +async fn drain_once( cursor: &mut Cursor, last_persisted_timestamp: &mut Option, last_chain_ts_persist_wall: &mut std::time::Instant, @@ -452,7 +446,6 @@ async fn drain_once( chain_ts_drift_tolerance_secs: u64, bypass_chain_clock_defenses: bool, registry: &R, - worker_queue: &W, chain_client: &T, event_source: &E, rx_stop: &mut mpsc::Receiver<()>, @@ -460,7 +453,6 @@ async fn drain_once( ) -> Result where R: AgreementRegistry + ChainListenerStateRegistry + PendingCancellationRegistry + Send + Sync, - W: WorkerQueue + Send + Sync, E: ChainEventSource, T: ChainClient + Send + Sync, { @@ -603,7 +595,7 @@ where } let agreement = agreements_by_id.remove(&snapshot.agreement_id); - match prepare_reconciliation(&snapshot, agreement, registry, worker_queue).await { + match prepare_reconciliation(&snapshot, agreement, registry, chain_client).await { Ok(Some(prep)) => prepared.push(prep), Ok(None) => {} Err(err) => { @@ -823,15 +815,15 @@ fn apply_chain_ts_drift_cap( clippy::cognitive_complexity, reason = "predates this lint; fix when next touched" )] -async fn prepare_reconciliation( +async fn prepare_reconciliation( snapshot: &AgreementStateSnapshot, agreement: Option, registry: &R, - worker_queue: &W, + chain_client: &T, ) -> anyhow::Result> where R: AgreementRegistry + Sync, - W: WorkerQueue, + T: ChainClient, { tracing::debug!( agreement_id = %snapshot.agreement_id, @@ -868,12 +860,9 @@ where tracing::warn!( agreement_id = %agreement.id, indexer = %snapshot.indexer, - "Rejected agreement accepted on-chain, queuing cancellation" + "Rejected agreement accepted on-chain, cancelling it" ); - worker_queue - // Background: on-chain cleanup of a rejected-then-accepted agreement. - .cancel_rejected_agreement_on_chain(agreement.id, JobPriority::Background) - .await?; + crate::cancel_dispatch::reopen_if_live(registry, chain_client, &agreement).await?; // This row goes Rejected -> Canceled without ever transiting // AcceptedOnChain, so `apply_reconciliation` never records the accept. @@ -892,7 +881,7 @@ where } } - queue_cancel_if_cancelled_but_accepted(snapshot, &agreement, worker_queue).await?; + reopen_if_cancelled_but_accepted(snapshot, &agreement, registry, chain_client).await?; record_accept_of_cancelling(snapshot, &agreement, registry).await; // Both transitions are applied atomically downstream so the @@ -1008,25 +997,23 @@ fn created_after_events_started(agreement: &IndexingAgreement) -> bool { /// Safety net for an agreement dipper cancelled whose offer the indexer accepted /// anyway, such as one that landed after dipper's cancel. Nothing else would end -/// it: reconciliation ignores an accept on a cancelled row. The job reads the -/// chain before acting, so a stale snapshot of an agreement already ended is free. -async fn queue_cancel_if_cancelled_but_accepted( +/// it: reconciliation ignores an accept on a cancelled row. It goes back to +/// `Cancelling` for the cancel retry, unless the chain shows it already ended. +async fn reopen_if_cancelled_but_accepted( snapshot: &AgreementStateSnapshot, agreement: &IndexingAgreement, - worker_queue: &W, -) -> anyhow::Result<()> { + registry: &R, + chain_client: &T, +) -> anyhow::Result<()> +where + R: AgreementRegistry + Sync, + T: ChainClient, +{ if agreement.status == IndexingAgreementStatus::CanceledByRequester && snapshot.state.reached_accepted() && !snapshot.state.is_canceled() { - tracing::warn!( - agreement_id = %agreement.id, - indexer = %snapshot.indexer, - "Cancelled agreement accepted on-chain, queuing cancellation" - ); - worker_queue - .cancel_rejected_agreement_on_chain(agreement.id, JobPriority::Background) - .await?; + crate::cancel_dispatch::reopen_if_live(registry, chain_client, agreement).await?; } Ok(()) } @@ -1131,22 +1118,20 @@ where /// and applies whatever transitions the diff implies. See the module-level /// transition table for the full mapping. #[cfg(test)] -async fn reconcile_agreement( +async fn reconcile_agreement( snapshot: &AgreementStateSnapshot, registry: &R, - worker_queue: &W, chain_client: &T, config: &crate::config::IndexingAgreementConfig, ) -> anyhow::Result<()> where R: AgreementRegistry + PendingCancellationRegistry + Sync, - W: WorkerQueue, T: ChainClient, { let agreement = registry .get_indexing_agreement_by_id(&snapshot.agreement_id) .await?; - let Some(prep) = prepare_reconciliation(snapshot, agreement, registry, worker_queue).await? + let Some(prep) = prepare_reconciliation(snapshot, agreement, registry, chain_client).await? else { return Ok(()); }; @@ -1817,7 +1802,6 @@ mod tests { use dipper_core::ids::{IndexingAgreementId, IndexingRequestId}; use thegraph_core::{DeploymentId, IndexerId, alloy::primitives::ChainId}; use time::OffsetDateTime; - use url::Url; use super::{super::chain_events::AgreementState, *}; use crate::registry::{ @@ -1902,6 +1886,7 @@ mod tests { marked_accepted_on_chain: Vec, marked_canceled_by_requester: Vec, marked_cancelling: Vec, + reopened: Vec, marked_canceled_by_indexer: Vec, /// Ids passed to `record_cancel_audit` -- the signal a cancel path drives /// the terminated event (the sweep emits from this audit). @@ -2012,6 +1997,10 @@ mod tests { self.state.lock().unwrap().marked_cancelling.contains(id) } + fn was_reopened(&self, id: &IndexingAgreementId) -> bool { + self.state.lock().unwrap().reopened.contains(id) + } + fn was_marked_canceled_by_requester(&self, id: &IndexingAgreementId) -> bool { self.state .lock() @@ -2203,6 +2192,14 @@ mod tests { Ok(()) } + async fn reopen_indexing_agreement_cancel( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()> { + self.state.lock().unwrap().reopened.push(*id); + Ok(()) + } + async fn get_cancelling_agreements( &self, _batch_size: i64, @@ -2545,18 +2542,6 @@ mod tests { } } - // Mock worker queue - #[derive(Clone, Default)] - struct MockWorkerQueue { - cancel_jobs: Arc>>, - } - - impl MockWorkerQueue { - fn was_cancellation_queued(&self, id: &IndexingAgreementId) -> bool { - self.cancel_jobs.lock().unwrap().contains(id) - } - } - /// Minimal `ChainClient` mock for chain_listener tests. Records every /// on-chain cancel attempt. Tests can mark specific agreements as /// already-canceled-on-chain (cancel returns `Ok(None)`); unmarked @@ -2573,6 +2558,14 @@ mod tests { live_until_cancelled: bool, } + /// A chain on which every agreement is live until a cancel is sent for it. + fn live_chain() -> MockChainClient { + MockChainClient { + live_until_cancelled: true, + ..MockChainClient::default() + } + } + impl MockChainClient { fn was_on_chain_cancel_attempted(&self, id: &IndexingAgreementId) -> bool { self.cancels.lock().unwrap().contains(id.as_bytes()) @@ -2671,60 +2664,12 @@ mod tests { } } - #[async_trait::async_trait] - impl crate::worker::service::WorkerQueue for MockWorkerQueue { - async fn send_indexing_agreement_proposal( - &self, - _candidate_url: Url, - _agreement_id: IndexingAgreementId, - _indexing_request_id: IndexingRequestId, - _deployment_id: DeploymentId, - _deployment_chain_id: ChainId, - _priority: JobPriority, - ) -> anyhow::Result { - Ok(dipper_pgmq::JobId::default()) - } - - async fn reassess_indexing_request( - &self, - _indexing_request_id: IndexingRequestId, - _deployment_id: DeploymentId, - _deployment_chain_id: ChainId, - _num_candidates: usize, - _priority: JobPriority, - ) -> anyhow::Result { - Ok(dipper_pgmq::JobId::default()) - } - - async fn cancel_rejected_agreement_on_chain( - &self, - agreement_id: IndexingAgreementId, - _priority: JobPriority, - ) -> anyhow::Result { - self.cancel_jobs.lock().unwrap().push(agreement_id); - Ok(dipper_pgmq::JobId::default()) - } - - async fn submit_offer( - &self, - _agreement_id: IndexingAgreementId, - _indexing_request_id: IndexingRequestId, - _indexer_url: Url, - _deployment_id: DeploymentId, - _deployment_chain_id: ChainId, - _priority: JobPriority, - ) -> anyhow::Result { - Ok(dipper_pgmq::JobId::default()) - } - } - // -- reconcile_agreement tests -- #[tokio::test] async fn test_reconcile_transitions_created_to_accepted() { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::Created); @@ -2733,7 +2678,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2741,7 +2685,7 @@ mod tests { assert!(result.is_ok()); assert!(registry.was_marked_accepted_on_chain(&agreement_id)); - assert!(!worker_queue.was_cancellation_queued(&agreement_id)); + assert!(!registry.was_reopened(&agreement_id)); } #[tokio::test] @@ -2749,7 +2693,6 @@ mod tests { // It stays cancelling, and the recorded accept lets its end be announced. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::Cancelling); @@ -2757,7 +2700,6 @@ mod tests { reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2766,14 +2708,13 @@ mod tests { assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); assert_eq!(registry.audit_writes(), vec![("accept", agreement_id)]); - assert!(!worker_queue.was_cancellation_queued(&agreement_id)); + assert!(!registry.was_reopened(&agreement_id)); } #[tokio::test] async fn reconcile_records_no_accept_of_an_agreement_from_before_events_existed() { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::Cancelling); registry.set_agreement_created_at(agreement_id, LIFECYCLE_EVENTS_START - 1); @@ -2782,7 +2723,6 @@ mod tests { reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2798,7 +2738,6 @@ mod tests { // time; recording it would announce an agreement that was never live. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::Cancelling); let mut snapshot = @@ -2808,7 +2747,6 @@ mod tests { reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2834,7 +2772,6 @@ mod tests { ] { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, status); let mut snapshot = @@ -2844,7 +2781,6 @@ mod tests { reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2873,7 +2809,6 @@ mod tests { ] { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::Cancelling); @@ -2881,7 +2816,6 @@ mod tests { reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2905,7 +2839,6 @@ mod tests { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let events = CapturingEventsProducer::new(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::Created); @@ -2914,7 +2847,6 @@ mod tests { reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2937,7 +2869,6 @@ mod tests { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let events = CapturingEventsProducer::new(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::AcceptedOnChain); @@ -2954,7 +2885,6 @@ mod tests { reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2972,10 +2902,12 @@ mod tests { } #[tokio::test] - async fn test_reconcile_queues_cancellation_for_rejected() { + async fn test_reconcile_reopens_the_cancel_of_a_rejected_agreement_live_on_chain() { let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); + let chain_client = MockChainClient { + live_until_cancelled: true, + ..MockChainClient::default() + }; let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::Rejected); @@ -2984,7 +2916,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -2992,7 +2923,7 @@ mod tests { assert!(result.is_ok()); assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); - assert!(worker_queue.was_cancellation_queued(&agreement_id)); + assert!(registry.was_reopened(&agreement_id)); } #[tokio::test] @@ -3004,7 +2935,6 @@ mod tests { // the canceler address. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let signer_address: Address = "0xaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" .parse() @@ -3020,14 +2950,13 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) .await; assert!(result.is_ok()); - assert!(!worker_queue.was_cancellation_queued(&agreement_id)); + assert!(!registry.was_reopened(&agreement_id)); assert!(registry.was_marked_canceled_by_requester(&agreement_id)); assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); } @@ -3036,7 +2965,6 @@ mod tests { async fn test_reconcile_ignores_unknown_agreement() { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); // Don't add the agreement to the registry @@ -3044,7 +2972,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -3052,14 +2979,13 @@ mod tests { assert!(result.is_ok()); assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); - assert!(!worker_queue.was_cancellation_queued(&agreement_id)); + assert!(!registry.was_reopened(&agreement_id)); } #[tokio::test] async fn test_reconcile_recovers_expired_agreement() { let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); + let chain_client = live_chain(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let old_agreement_id = IndexingAgreementId::from_bytes(rand::random()); @@ -3071,7 +2997,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -3087,7 +3012,6 @@ mod tests { async fn test_reconcile_marks_canceled_by_indexer() { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let indexer_address: Address = "0x1234567890123456789012345678901234567890" .parse() @@ -3103,7 +3027,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -3118,7 +3041,6 @@ mod tests { async fn test_reconcile_marks_canceled_by_requester() { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let signer_address: Address = "0xaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" .parse() @@ -3134,7 +3056,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -3152,7 +3073,6 @@ mod tests { // the state: CanceledByPayer -> ByRequester, the only kind allowed here. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); // A payer address deliberately distinct from dipper's signer key. let payer_address: Address = "0xcccccccccccccccccccccccccccccccccccccccc" @@ -3165,7 +3085,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -3176,17 +3095,19 @@ mod tests { assert!(!registry.was_marked_canceled_by_indexer(&agreement_id)); assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); // Already canceled on-chain: must not queue a fresh cancel job. - assert!(!worker_queue.was_cancellation_queued(&agreement_id)); + assert!(!registry.was_reopened(&agreement_id)); } #[tokio::test] - async fn test_reconcile_queues_cancel_for_cancelled_agreement_accepted_on_chain() { + async fn test_reconcile_reopens_the_cancel_of_a_cancelled_agreement_live_on_chain() { // Dipper cancelled the agreement locally, but the indexer accepted its offer // (for example one that landed after dipper's cancel). Nothing else would end - // it, so the listener queues an on-chain cancel and leaves the row cancelled. + // it, so it goes back to cancelling for the cancel retry to end. let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); + let chain_client = MockChainClient { + live_until_cancelled: true, + ..MockChainClient::default() + }; let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); @@ -3194,22 +3115,42 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) .await; assert!(result.is_ok()); - assert!(worker_queue.was_cancellation_queued(&agreement_id)); + assert!(registry.was_reopened(&agreement_id)); assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); } + #[tokio::test] + async fn test_reconcile_leaves_a_cancelled_agreement_the_chain_shows_ended() { + // The subgraph can still report an agreement accepted for a few polls after + // dipper's cancel lands; reopening it would only bring it back to cancelling. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); + + let snapshot = make_snapshot(agreement_id, AgreementState::Accepted, Address::ZERO); + reconcile_agreement( + &snapshot, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await + .expect("reconcile ok"); + + assert!(!registry.was_reopened(&agreement_id)); + } + #[tokio::test] async fn test_reconcile_cancelled_agreement_already_cancelled_on_chain_queues_nothing() { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); @@ -3217,14 +3158,13 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) .await; assert!(result.is_ok()); - assert!(!worker_queue.was_cancellation_queued(&agreement_id)); + assert!(!registry.was_reopened(&agreement_id)); } #[tokio::test] @@ -3235,7 +3175,6 @@ mod tests { // waits for the accept. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); // The listener lagged: dipper only cancelled it locally long after the offer's @@ -3246,7 +3185,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -3265,7 +3203,6 @@ mod tests { // out with fallback cancel fields. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); registry.state.lock().unwrap().fail_cancel_audit = true; @@ -3274,7 +3211,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -3290,7 +3226,6 @@ mod tests { // and are never announced, even when a replay of the chain reads them again. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); registry.set_agreement_created_at(agreement_id, LIFECYCLE_EVENTS_START - 1); @@ -3299,7 +3234,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -3314,7 +3248,6 @@ mod tests { // A withdrawn offer was never accepted, so there is nothing to announce. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); @@ -3324,7 +3257,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -3340,7 +3272,6 @@ mod tests { // agreement has actually ended on-chain. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); @@ -3348,7 +3279,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -3362,7 +3292,6 @@ mod tests { async fn test_reconcile_ignores_already_canceled() { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByIndexer); @@ -3375,7 +3304,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -3399,7 +3327,6 @@ mod tests { // rather than incrementing its `errors` counter. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let signer_address: Address = "0xaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" .parse() @@ -3415,7 +3342,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -3434,8 +3360,7 @@ mod tests { // We should run the acceptance-side bookkeeping (pending cancellations) // AND mark the agreement as CanceledByRequester. let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); - let worker_queue = MockWorkerQueue::default(); + let chain_client = live_chain(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let old_agreement_id = IndexingAgreementId::from_bytes(rand::random()); let signer_address: Address = "0xaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" @@ -3454,7 +3379,6 @@ mod tests { let result = reconcile_agreement( &snapshot, ®istry, - &worker_queue, &chain_client, test_agreement_conf().as_ref(), ) @@ -3474,7 +3398,7 @@ mod tests { #[tokio::test] async fn test_pending_cancellations_all_succeed_records_deleted() { let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); + let chain_client = live_chain(); let new_id = IndexingAgreementId::from_bytes(rand::random()); let old_id_1 = IndexingAgreementId::from_bytes(rand::random()); let old_id_2 = IndexingAgreementId::from_bytes(rand::random()); @@ -3510,7 +3434,7 @@ mod tests { // execute_pending no longer emits `terminated` directly; it records the // cancel audit and the chain_listener sweep announces it durably. let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); + let chain_client = live_chain(); let new_id = IndexingAgreementId::from_bytes(rand::random()); let old_id = IndexingAgreementId::from_bytes(rand::random()); @@ -3568,6 +3492,7 @@ mod tests { let registry = MockRegistry::new(); let chain_client = MockChainClient { registry: Some(registry.clone()), + live_until_cancelled: true, ..MockChainClient::default() }; let new_id = IndexingAgreementId::from_bytes(rand::random()); @@ -3696,7 +3621,7 @@ mod tests { #[tokio::test] async fn test_pending_cancellations_transient_failure_retains_record() { let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); + let chain_client = live_chain(); let new_id = IndexingAgreementId::from_bytes(rand::random()); let old_ok = IndexingAgreementId::from_bytes(rand::random()); let old_fail = IndexingAgreementId::from_bytes(rand::random()); @@ -3782,13 +3707,10 @@ mod tests { #[tokio::test] async fn test_pending_cancellations_already_canceled_on_chain_succeeds() { - // Crash-recovery edge case: the cancel tx confirmed on-chain on a - // prior pass, but dipper crashed before deleting the pending row. - // On the next sweep the chain call surfaces as Ok(None) (the - // SubgraphService contract reverts with IndexingAgreementNotActive; - // the chain client translates that into "already canceled"). The - // handler must still flip the local row to CanceledByRequester and - // delete the pending row, not loop forever. + // Crash-recovery edge case: the cancel landed on a prior pass, but dipper + // crashed before deleting the pending row. The chain shows nothing live, so no + // cancel is sent; the row is left cancelling for the cancel retry to confirm + // and the pending row is deleted rather than retried forever. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); let new_id = IndexingAgreementId::from_bytes(rand::random()); @@ -3811,8 +3733,8 @@ mod tests { result.is_ok(), "expected idempotent success, got {result:?}" ); - assert!(chain_client.was_on_chain_cancel_attempted(&old_id)); - assert!(registry.was_marked_canceled_by_requester(&old_id)); + assert!(!chain_client.was_on_chain_cancel_attempted(&old_id)); + assert!(registry.was_marked_cancelling(&old_id)); assert!(registry.was_pending_cancellation_deleted(&new_id, &old_id)); } @@ -3838,7 +3760,7 @@ mod tests { ) .await; - assert!(registry.was_marked_canceled_by_requester(&old_id)); + assert!(registry.was_marked_cancelling(&old_id)); assert!(registry.was_pending_cancellation_deleted(&new_id, &old_id)); let remaining = registry .get_pending_cancellations_by_new_agreement(new_id) @@ -3857,7 +3779,7 @@ mod tests { // and the old agreement is still alive. The sweep must complete // the cancellation without needing another snapshot to arrive. let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); + let chain_client = live_chain(); let new_id = IndexingAgreementId::from_bytes(rand::random()); let old_id = IndexingAgreementId::from_bytes(rand::random()); @@ -4012,7 +3934,6 @@ mod tests { let ctx = Ctx { registry: registry.clone(), - worker_queue: MockWorkerQueue::default(), chain_client: MockChainClient::default(), event_source, config, @@ -4096,7 +4017,6 @@ mod tests { let ctx = Ctx { registry: registry.clone(), - worker_queue: MockWorkerQueue::default(), chain_client: MockChainClient::default(), event_source, config, @@ -4177,7 +4097,6 @@ mod tests { let ctx = Ctx { registry: registry.clone(), - worker_queue: MockWorkerQueue::default(), chain_client: MockChainClient::default(), event_source, config, @@ -4262,7 +4181,6 @@ mod tests { 10, false, ®istry, - &MockWorkerQueue::default(), &chain_client, &event_source, &mut rx_stop, @@ -4327,7 +4245,6 @@ mod tests { 10, false, ®istry, - &MockWorkerQueue::default(), &chain_client, &event_source, &mut rx_stop, @@ -4355,7 +4272,6 @@ mod tests { 10, false, ®istry, - &MockWorkerQueue::default(), &chain_client, &event_source, &mut rx_stop, @@ -4459,7 +4375,6 @@ mod tests { 10, false, ®istry, - &MockWorkerQueue::default(), &chain_client, &event_source, &mut rx_stop, @@ -4500,7 +4415,6 @@ mod tests { let ctx = Ctx { registry: MockRegistry::new(), - worker_queue: MockWorkerQueue::default(), chain_client: MockChainClient::default(), event_source: TimingEventSource { poll_times: poll_times.clone(), @@ -4539,7 +4453,7 @@ mod tests { // Canceled is the orphan signature: reassessment fired the chain // cancel and failed, then bailed out. The sweep must pick it up. let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); + let chain_client = live_chain(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let request_id = IndexingRequestId::new(); @@ -4586,7 +4500,7 @@ mod tests { // The orphan sweep no longer emits `terminated` directly; it records the // cancel audit and `sweep_pending_terminated_events` announces it. let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); + let chain_client = live_chain(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); let request_id = IndexingRequestId::new(); @@ -4606,8 +4520,8 @@ mod tests { #[tokio::test] async fn test_orphan_sweep_handles_already_canceled_on_chain() { - // Idempotency check: the chain reports the agreement is already - // canceled (Ok(None)). The sweep must still clean up the local row. + // The chain shows the agreement already ended, so no cancel is sent; it is + // left cancelling for the cancel retry to confirm who ended it. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); @@ -4621,8 +4535,9 @@ mod tests { sweep_orphan_canceled_agreements(®istry, &chain_client, test_agreement_conf().as_ref()) .await; - assert!(chain_client.was_on_chain_cancel_attempted(&agreement_id)); - assert!(registry.was_marked_canceled_by_requester(&agreement_id)); + assert!(!chain_client.was_on_chain_cancel_attempted(&agreement_id)); + assert!(registry.was_marked_cancelling(&agreement_id)); + assert!(!registry.was_marked_canceled_by_requester(&agreement_id)); } #[tokio::test] diff --git a/bin/dipper-service/src/network/service/expiration.rs b/bin/dipper-service/src/network/service/expiration.rs index 14652149..12802fcf 100644 --- a/bin/dipper-service/src/network/service/expiration.rs +++ b/bin/dipper-service/src/network/service/expiration.rs @@ -538,13 +538,6 @@ mod tests { ) -> anyhow::Result { Ok(JobId::default()) } - async fn cancel_rejected_agreement_on_chain( - &self, - _agreement_id: IndexingAgreementId, - _priority: JobPriority, - ) -> anyhow::Result { - unimplemented!() - } async fn submit_offer( &self, _agreement_id: IndexingAgreementId, diff --git a/bin/dipper-service/src/network/service/liveness_checker.rs b/bin/dipper-service/src/network/service/liveness_checker.rs index 9e520241..367f3a98 100644 --- a/bin/dipper-service/src/network/service/liveness_checker.rs +++ b/bin/dipper-service/src/network/service/liveness_checker.rs @@ -1098,13 +1098,6 @@ mod tests { self.calls.reassessments.lock().unwrap().push(req_id); Ok(JobId::default()) } - async fn cancel_rejected_agreement_on_chain( - &self, - _agr_id: IndexingAgreementId, - _priority: JobPriority, - ) -> anyhow::Result { - unimplemented!() - } async fn submit_offer( &self, _agreement_id: IndexingAgreementId, diff --git a/bin/dipper-service/src/registry.rs b/bin/dipper-service/src/registry.rs index 7d30552c..eb5c7185 100644 --- a/bin/dipper-service/src/registry.rs +++ b/bin/dipper-service/src/registry.rs @@ -401,6 +401,16 @@ impl AgreementRegistry for RegistryProvider { .map_err(Into::into) } + async fn reopen_indexing_agreement_cancel( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()> { + self.inner + .reopen_indexing_agreement_cancel(id) + .await + .map_err(Into::into) + } + async fn get_cancelling_agreements( &self, batch_size: i64, diff --git a/bin/dipper-service/src/registry/agreement.rs b/bin/dipper-service/src/registry/agreement.rs index 1309d1d3..762848a8 100644 --- a/bin/dipper-service/src/registry/agreement.rs +++ b/bin/dipper-service/src/registry/agreement.rs @@ -273,6 +273,14 @@ pub trait AgreementRegistry { id: &IndexingAgreementId, ) -> RegistryResult<()>; + /// Move a `CANCELED_BY_REQUESTER` or `REJECTED` agreement the chain shows live back to + /// `CANCELLING`, its cancel attempts reset; [`NoRecordUpdated`](Error::NoRecordsUpdated) + /// otherwise. + async fn reopen_indexing_agreement_cancel( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()>; + /// `CANCELLING` agreements marked over `min_age_minutes` ago whose cancel has failed /// fewer than `max_attempts` times, those checked longest ago first. One that may be paying /// an indexer (accepted, or past the offer deadline, which only an accepted one outlives) diff --git a/bin/dipper-service/src/registry/agreement_stub.rs b/bin/dipper-service/src/registry/agreement_stub.rs index 3e7477be..9e516c4b 100644 --- a/bin/dipper-service/src/registry/agreement_stub.rs +++ b/bin/dipper-service/src/registry/agreement_stub.rs @@ -137,6 +137,10 @@ pub trait StubAgreementRegistry: Send + Sync { unimplemented!("mark_indexing_agreement_as_cancelling") } + async fn reopen_indexing_agreement_cancel(&self, _id: &IndexingAgreementId) -> Result<()> { + unimplemented!("reopen_indexing_agreement_cancel") + } + async fn get_cancelling_agreements( &self, _batch_size: i64, @@ -440,6 +444,10 @@ impl AgreementRegistry for T { StubAgreementRegistry::mark_indexing_agreement_as_cancelling(self, id).await } + async fn reopen_indexing_agreement_cancel(&self, id: &IndexingAgreementId) -> Result<()> { + StubAgreementRegistry::reopen_indexing_agreement_cancel(self, id).await + } + async fn get_cancelling_agreements( &self, batch_size: i64, diff --git a/bin/dipper-service/src/worker/context.rs b/bin/dipper-service/src/worker/context.rs index 5e1b7e60..73175fc5 100644 --- a/bin/dipper-service/src/worker/context.rs +++ b/bin/dipper-service/src/worker/context.rs @@ -234,7 +234,6 @@ impl_from_state!(SendIndexingAgreementProposalCtx { impl_from_state!(CancelRejectedAgreementOnChainCtx { registry, chain_client, - agreement_conf, }); impl_from_state!(SubmitOfferCtx { diff --git a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs index 651245c0..700cb26a 100644 --- a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs +++ b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs @@ -1,27 +1,21 @@ -//! Cancel on-chain, via the RecurringAgreementManager, an agreement dipper doesn't want -//! that was accepted anyway: one the indexer rejected off-chain, or one dipper had already -//! cancelled. The chain listener queues it. +//! Kept so jobs queued before an upgrade still run. The chain listener now moves an +//! agreement dipper rejected or cancelled that went live on-chain anyway back into +//! `Cancelling` itself, and the cancel retry ends it; this job does the same. -use std::{ - collections::HashSet, - sync::{Arc, LazyLock, Mutex, PoisonError}, - time::Duration, -}; +use std::time::Duration; use dipper_core::ids::IndexingAgreementId; use crate::{ - cancel_dispatch::{LiveCancel, cancel_agreement_on_chain, cancel_if_live}, - chain_client::{ChainClient, ChainClientError}, - config::IndexingAgreementConfig, - registry::{AgreementRegistry, IndexingAgreement, IndexingAgreementStatus}, + cancel_dispatch::reopen_if_live, + chain_client::ChainClient, + registry::{AgreementRegistry, IndexingAgreementStatus}, worker::result::{JobError, JobResult}, }; pub struct Ctx { pub registry: R, pub chain_client: T, - pub agreement_conf: Arc, } /// Cancel on-chain an agreement dipper rejected or cancelled that was accepted anyway. @@ -30,597 +24,97 @@ pub struct Message { pub agreement_id: IndexingAgreementId, } -/// Cancel on-chain an agreement dipper rejected or cancelled that was accepted -/// anyway, via `cancelIndexingAgreementByPayer`, so the indexer isn't paid for -/// work dipper didn't want. -#[expect( - clippy::cognitive_complexity, - reason = "predates this lint; fix when next touched" -)] +/// Hand an agreement dipper rejected or cancelled that is live on-chain to the cancel retry. pub async fn handle(ctx: Ctx, Message { agreement_id }: &Message) -> JobResult<()> where R: AgreementRegistry + Sync, T: ChainClient, { - // Look up the agreement - let agreement = ctx + let Some(agreement) = ctx .registry .get_indexing_agreement_by_id(agreement_id) .await - .map_err(|err| JobError::Fatal(err.into()))?; - - let agreement = match agreement { - Some(a) => a, - None => { - tracing::error!( - agreement_id = %agreement_id, - "Agreement not found for on-chain cancellation" - ); - return Ok(()); - } - }; - - // This job is only queued for these 2 statuses. - match agreement.status { - IndexingAgreementStatus::Rejected => {} - IndexingAgreementStatus::CanceledByRequester => { - return cancel_live_agreement_dipper_cancelled(&ctx, &agreement).await; - } - status => { - tracing::warn!( - agreement_id = %agreement_id, - status = %status, - "Agreement neither Rejected nor CanceledByRequester, skipping on-chain cancellation" - ); - return Ok(()); - } - } - - tracing::info!( - agreement_id = %agreement_id, - indexer_id = %agreement.indexer.id, - "Canceling rejected agreement on-chain" - ); - - // Send the cancellation transaction (mode-aware dispatch). - let on_chain_cancel_tx: Option = - match cancel_agreement_on_chain(&ctx.chain_client, &agreement, &ctx.agreement_conf).await { - Ok(Some(tx_hash)) => { - tracing::info!( - agreement_id = %agreement_id, - tx_hash = %tx_hash, - "Successfully submitted on-chain cancellation" - ); - Some(tx_hash.to_string()) - } - Ok(None) => { - tracing::info!( - agreement_id = %agreement_id, - "Rejected agreement already canceled on-chain; reconciling local state" - ); - None - } - Err(err @ ChainClientError::MissingTermsVersionHash { .. }) => { - // Permanent: the hash never appears, so retrying can't help. Fail - // terminally and leave the live agreement for operator action. - tracing::error!( - agreement_id = %agreement_id, - error = %err, - "Cannot cancel rejected agreement: missing terms_version_hash" - ); - return Err(JobError::Fatal(err.into())); - } - Err(err) => { - tracing::warn!( - agreement_id = %agreement_id, - error = %err, - "Failed to cancel agreement on-chain, will retry" - ); - // Retry with backoff - on-chain transactions can fail due to gas issues, nonce, etc. - return Err(JobError::Retryable(err.into(), Duration::from_secs(30))); - } - }; - - // Once the row is terminal, the cancel audit lets the `terminated` sweep - // announce it (the accept was recorded when the listener queued this job). - // If the mark failed, the listener sees the on-chain cancel and flips it. - if mark_cancellation_complete(&ctx.registry, agreement_id).await { - let manager = ctx.agreement_conf.recurring_agreement_manager().to_string(); - if let Err(err) = ctx - .registry - .record_cancel_audit( - agreement_id, - dipper_core::time::now_secs(), - &manager, - on_chain_cancel_tx.as_deref(), - ) - .await - { - tracing::warn!( - agreement_id = %agreement_id, - error = %err, - "failed to record cancel audit; terminated event may emit with fallback fields" - ); - } - } - - Ok(()) -} - -/// Agreements a job in this process is cancelling right now. The listener can -/// queue one twice before the chain shows it ended, and 2 jobs at once would -/// both cancel it and both alert. -static CANCELLING: LazyLock>> = LazyLock::new(Mutex::default); - -/// A job's claim on cancelling one agreement, released when the job ends. -struct Cancelling(IndexingAgreementId); - -impl Cancelling { - fn claim(agreement_id: IndexingAgreementId) -> Option { - // Unlock before a claim exists: dropping one locks the set again. - let claimed = CANCELLING - .lock() - .unwrap_or_else(PoisonError::into_inner) - .insert(agreement_id); - claimed.then(|| Self(agreement_id)) - } -} - -impl Drop for Cancelling { - fn drop(&mut self) { - let mut cancelling = CANCELLING.lock().unwrap_or_else(PoisonError::into_inner); - cancelling.remove(&self.0); - } -} - -/// Cancel on-chain an agreement dipper had already cancelled that the indexer accepted -/// anyway; the row is already terminal. The chain is read first, so a stale snapshot of -/// one dipper has since ended raises no alert. -async fn cancel_live_agreement_dipper_cancelled( - ctx: &Ctx, - agreement: &IndexingAgreement, -) -> JobResult<()> -where - T: ChainClient, -{ - let Some(_claim) = Cancelling::claim(agreement.id) else { - tracing::info!( - agreement_id = %agreement.id, - "Another job is already cancelling this agreement" - ); + .map_err(|err| JobError::Fatal(err.into()))? + else { + tracing::warn!(%agreement_id, "Agreement not found for on-chain cancellation"); return Ok(()); }; - match cancel_if_live(&ctx.chain_client, agreement, &ctx.agreement_conf).await { - LiveCancel::NotLive => { - tracing::info!( - agreement_id = %agreement.id, - "Cancelled agreement is no longer live on-chain; nothing to cancel" - ); - Ok(()) - } - LiveCancel::ReadFailed(err) => { - tracing::warn!( - agreement_id = %agreement.id, - error = %err, - "Failed to read whether a cancelled agreement is live on-chain, will retry" - ); - Err(JobError::Retryable(err.into(), Duration::from_secs(30))) - } - LiveCancel::Ended(tx_hash) => { - let tx = tx_hash.map_or_else(|| "none".to_owned(), |hash| hash.to_string()); - log_caught_live_agreement(agreement, "cancelled", &tx); - Ok(()) - } - LiveCancel::CancelFailed(err @ ChainClientError::MissingTermsVersionHash { .. }) => { - log_caught_live_agreement(agreement, "cancel_impossible", &err.to_string()); - Err(JobError::Fatal(err.into())) - } - LiveCancel::CancelFailed(err) => { - log_caught_live_agreement(agreement, "cancel_failed", &err.to_string()); - Err(JobError::Retryable(err.into(), Duration::from_secs(30))) - } + if !matches!( + agreement.status, + IndexingAgreementStatus::Rejected | IndexingAgreementStatus::CanceledByRequester + ) { + return Ok(()); } -} - -/// ERROR line with stable `event` and `outcome` for alerting: `cancelled` once per -/// agreement ended, `cancel_failed` per failed attempt (so running out of retries -/// is never silent), `cancel_impossible` when it never can be. -fn log_caught_live_agreement(agreement: &IndexingAgreement, outcome: &str, detail: &str) { - tracing::error!( - event = "cancelled_agreement_live_on_chain", - outcome, - agreement_id = %agreement.id, - indexer_id = %agreement.indexer.id, - indexing_request_id = %agreement.indexing_request_id, - detail, - "Agreement dipper had cancelled is live on-chain" - ); -} - -/// Flip the row to CanceledByRequester once the chain shows it cancelled. A failure -/// is logged, not fatal: the chain is already right and the listener retries the -/// DB update. Returns whether the row is now terminal, gating the cancel audit. -async fn mark_cancellation_complete(registry: &R, agreement_id: &IndexingAgreementId) -> bool -where - R: AgreementRegistry + Sync, -{ - match registry - .mark_indexing_agreement_as_canceled_by_requester(agreement_id) + reopen_if_live(&ctx.registry, &ctx.chain_client, &agreement) .await - { - Ok(()) => { - tracing::info!( - agreement_id = %agreement_id, - old_status = "REJECTED", - new_status = "CANCELED_BY_REQUESTER", - reason = "canceled_on_chain_after_rejection", - "agreement state transition" - ); - true - } - Err(err) => { - tracing::error!( - agreement_id = %agreement_id, - error = %err, - "Failed to update agreement status after on-chain cancellation" - ); - false - } - } + .map(|_| ()) + .map_err(|err| JobError::Retryable(err.into(), Duration::from_secs(30))) } #[cfg(test)] mod tests { - use std::sync::{ - Mutex, - atomic::{AtomicBool, Ordering}, - }; + use std::sync::{Arc, Mutex}; use async_trait::async_trait; - use dipper_core::ids::IndexingRequestId; use dipper_rpc::indexer::indexer_client::sol::RecurringCollectionAgreement; - use thegraph_core::{ - DeploymentId, IndexerId, - alloy::primitives::{Address, B256, U256}, - }; - use time::OffsetDateTime; - use url::Url; + use thegraph_core::alloy::primitives::{Address, B256}; use super::*; use crate::{ - chain_client::{ChainClient, ChainClientError}, - registry::{ - IndexingAgreement, IndexingAgreementStatus, IndexingAgreementTerms, - IndexingAgreementTermsMetadata, - }, + cancel_dispatch::tests::agreement, + chain_client::ChainClientError, + registry::{IndexingAgreement, StubAgreementRegistry}, }; - // ========================================================================= - // Mock implementations - // ========================================================================= - - /// Returns one configurable agreement and records the terminal-cancel and - /// cancel-audit calls. Clones share state, so a test can still assert on its - /// copy after `handle` consumes the ctx. - #[derive(Clone)] struct MockRegistry { - agreement: Arc>>, - marked_canceled: Arc>>, - /// Ids passed to `record_cancel_audit` -- the signal the handler drives - /// the terminated event (the chain_listener sweep emits from this audit). - recorded_cancel_audit: Arc>>, - /// When true, `mark_indexing_agreement_as_canceled_by_requester` errors. - fail_mark: bool, - } - - impl MockRegistry { - fn new(agreement: IndexingAgreement) -> Self { - Self { - agreement: Arc::new(Mutex::new(Some(agreement))), - marked_canceled: Arc::new(Mutex::new(Vec::new())), - recorded_cancel_audit: Arc::new(Mutex::new(Vec::new())), - fail_mark: false, - } - } - - fn with_mark_failure(agreement: IndexingAgreement) -> Self { - Self { - fail_mark: true, - ..Self::new(agreement) - } - } + agreement: IndexingAgreement, + reopened: Arc>>, } #[async_trait] - impl AgreementRegistry for MockRegistry { + impl StubAgreementRegistry for MockRegistry { async fn get_indexing_agreement_by_id( &self, _id: &IndexingAgreementId, ) -> crate::registry::Result> { - Ok(self.agreement.lock().unwrap().clone()) - } - - async fn get_indexing_agreements_by_deployment_id( - &self, - _deployment_id: &DeploymentId, - ) -> crate::registry::Result> { - Ok(vec![]) - } - - async fn get_indexing_agreements_by_indexer_id( - &self, - _indexer_id: &IndexerId, - ) -> crate::registry::Result> { - Ok(vec![]) - } - - async fn get_pending_agreement_indexers_by_deployment( - &self, - _indexer_ids: &[IndexerId], - ) -> crate::registry::Result>> - { - Ok(std::collections::HashMap::new()) + Ok(Some(self.agreement.clone())) } - - async fn get_declined_indexers_by_deployment( - &self, - _default_lookback_days: i32, - _price_lookback_days: i32, - _transient_lookback_minutes: i32, - _uncertain_lookback_days: i32, - ) -> crate::registry::Result>> - { - Ok(std::collections::HashMap::new()) - } - - async fn get_indexing_agreements_by_indexing_request_id( - &self, - _request_id: &IndexingRequestId, - ) -> crate::registry::Result> { - Ok(vec![]) - } - - async fn get_active_indexing_agreements_by_indexing_request_id( - &self, - _request_id: &IndexingRequestId, - ) -> crate::registry::Result> { - Ok(vec![]) - } - - async fn count_accepted_agreements_by_deployment( - &self, - _deployment_id: &DeploymentId, - ) -> crate::registry::Result { - Ok(0) - } - - async fn record_cancel_audit( - &self, - agreement_id: &IndexingAgreementId, - _canceled_at: u64, - _canceled_by: &str, - _canceled_tx: Option<&str>, - ) -> crate::registry::Result<()> { - self.recorded_cancel_audit - .lock() - .unwrap() - .push(*agreement_id); - Ok(()) - } - - async fn register_new_indexing_agreement( - &self, - _params: crate::registry::NewAgreementParams, - ) -> crate::registry::Result { - Ok(IndexingAgreementId::from_bytes(rand::random())) - } - - async fn register_agreement_with_pending_cancellation( - &self, - _params: crate::registry::NewAgreementParams, - _old_agreement_id: IndexingAgreementId, - ) -> crate::registry::Result { - Ok(IndexingAgreementId::from_bytes(rand::random())) - } - - async fn get_unresponsive_indexers( - &self, - _lookback_days: i32, - _chain_id: thegraph_core::alloy::primitives::ChainId, - ) -> crate::registry::Result> { - unimplemented!() - } - - async fn mark_indexing_agreement_as_unresponsive( - &self, - _id: &IndexingAgreementId, - ) -> crate::registry::Result<()> { - unimplemented!() - } - - async fn count_created_agreements_by_indexer( - &self, - ) -> crate::registry::Result<(std::collections::HashMap, u64)> { - unimplemented!() - } - - async fn update_offer_tx_hash( - &self, - _id: &IndexingAgreementId, - _tx_hash: &[u8; 32], - ) -> crate::registry::Result<()> { - Ok(()) - } - - async fn mark_indexing_agreement_as_canceled_by_requester( + async fn reopen_indexing_agreement_cancel( &self, id: &IndexingAgreementId, ) -> crate::registry::Result<()> { - if self.fail_mark { - return Err(crate::registry::Error::NoRecordsUpdated); - } - self.marked_canceled.lock().unwrap().push(*id); + self.reopened.lock().unwrap().push(*id); Ok(()) } - - async fn mark_indexing_agreement_as_cancelling( - &self, - _id: &IndexingAgreementId, - ) -> crate::registry::Result<()> { - Ok(()) - } - - async fn get_cancelling_agreements( - &self, - _batch_size: i64, - _max_attempts: u32, - _min_age_minutes: i32, - ) -> crate::registry::Result> { - Ok(Vec::new()) - } - - async fn record_cancel_check( - &self, - _id: &IndexingAgreementId, - failed_attempts: u32, - ) -> crate::registry::Result { - Ok(failed_attempts) - } - - async fn apply_reconciliation( - &self, - _id: &IndexingAgreementId, - _apply_accept: bool, - _cancel: Option, - ) -> crate::registry::Result { - Ok(crate::registry::ReconciliationOutcome { - did_accept: false, - did_cancel: false, - }) - } - - async fn get_expired_created_agreements( - &self, - _batch_size: i64, - _chain_timestamp: u64, - ) -> crate::registry::Result> { - Ok(vec![]) - } - - async fn mark_indexing_agreement_as_expired( - &self, - _id: &IndexingAgreementId, - ) -> crate::registry::Result<()> { - Ok(()) - } - - async fn mark_indexing_agreement_as_rejected( - &self, - _id: &IndexingAgreementId, - _rejection_reason: Option<&str>, - ) -> crate::registry::Result<()> { - Ok(()) - } - - async fn get_accepted_on_chain_agreements( - &self, - _batch_size: i64, - ) -> crate::registry::Result> { - Ok(vec![]) - } - - async fn get_agreements_pending_chain_cancel( - &self, - _batch_size: i64, - ) -> crate::registry::Result> { - Ok(vec![]) - } - - async fn update_agreement_sync_progress( - &self, - _id: &IndexingAgreementId, - _block_height: u64, - _progress_at: time::OffsetDateTime, - ) -> crate::registry::Result<()> { - Ok(()) - } - - async fn count_active_agreements_by_deployment( - &self, - ) -> crate::registry::Result> { - Ok(std::collections::HashMap::new()) - } - - async fn mark_indexing_agreement_as_abandoned( - &self, - _id: &IndexingAgreementId, - ) -> crate::registry::Result { - Err(crate::registry::Error::NoRecordsUpdated) - } - - async fn get_agreement_fee_rates( - &self, - ) -> crate::registry::Result> { - Ok(vec![]) - } } - /// Chain client whose manager cancel always mines (returns a tx hash) and - /// ends the agreement, so the post-cancel liveness read confirms it. `live` - /// is whether the agreement is live on-chain before any cancel lands. - #[derive(Default, Clone)] - struct MockChainClient { - live: Arc, - cancelled: Arc>>, - fail_liveness_read: bool, - fail_cancel: bool, - } - - impl MockChainClient { - fn live() -> Self { - Self { - live: Arc::new(AtomicBool::new(true)), - ..Self::default() - } - } + struct MockChain { + live: bool, } #[async_trait] - impl ChainClient for MockChainClient { - async fn latest_block_timestamp(&self) -> Result { - Err(ChainClientError::RpcError(anyhow::anyhow!( - "latest_block_timestamp not mocked" - ))) - } - + impl ChainClient for MockChain { async fn offer_via_manager( &self, _rca: &RecurringCollectionAgreement, ) -> Result, ChainClientError> { - Ok(None) + unimplemented!() } - async fn cancel_via_manager( &self, _collector: Address, - agreement_id: &[u8; 16], + _agreement_id: &[u8; 16], _version_hash: B256, _options: u16, ) -> Result, ChainClientError> { - if self.fail_cancel { - return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); - } - self.cancelled.lock().unwrap().push(*agreement_id); - self.live.store(false, Ordering::SeqCst); - Ok(Some(B256::ZERO)) + panic!("the cancel retry sends cancels, not this job") } - async fn reconcile_provider( &self, _collector: Address, _provider: Address, ) -> Result, ChainClientError> { - Ok(None) + unimplemented!() } async fn reconcile_agreement( &self, @@ -629,370 +123,63 @@ mod tests { ) -> Result, ChainClientError> { unimplemented!() } - async fn agreement_still_active( &self, _agreement_id: &[u8; 16], ) -> Result { - if self.fail_liveness_read { - return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); - } - Ok(self.live.load(Ordering::SeqCst)) + Ok(self.live) } async fn agreement_ended_by_indexer( &self, _agreement_id: &[u8; 16], ) -> Result { - Ok(false) - } - } - - fn test_agreement_conf() -> Arc { - Arc::new(IndexingAgreementConfig::for_tests()) - } - - fn test_deployment_id() -> DeploymentId { - "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" - .parse() - .unwrap() - } - - fn make_agreement(status: IndexingAgreementStatus) -> IndexingAgreement { - IndexingAgreement { - id: IndexingAgreementId::from_bytes(rand::random()), - nonce_uuid: uuid::Uuid::now_v7(), - created_at: OffsetDateTime::now_utc(), - updated_at: OffsetDateTime::now_utc(), - status, - indexing_request_id: IndexingRequestId::new(), - indexer: crate::registry::Indexer { - id: IndexerId::from(Address::ZERO), - url: Url::parse("https://indexer.example").unwrap(), - }, - terms: IndexingAgreementTerms { - payer: Address::ZERO, - service_provider: Address::ZERO, - data_service: Address::ZERO, - deadline: 0, - ends_at: 0, - max_initial_tokens: U256::ZERO, - max_ongoing_tokens_per_second: U256::ZERO, - min_seconds_per_collection: 0, - max_seconds_per_collection: 0, - conditions: 0, - metadata: IndexingAgreementTermsMetadata { - tokens_per_second: U256::ZERO, - tokens_per_entity_per_second: U256::ZERO, - subgraph_deployment_id: test_deployment_id(), - protocol_network: 1u64, - chain_id: 1u64, - proposed_at: 0, - }, - }, - last_block_height: None, - last_progress_at: None, - rejection_reason: None, - // 32-byte hash so the on-chain cancel path is exercised. - terms_version_hash: Some(vec![0u8; 32]), - } - } - - // ========================================================================= - // Tests - // ========================================================================= - - fn ctx_for(registry: MockRegistry) -> Ctx { - ctx_with_chain(registry, MockChainClient::default()) - } - - fn ctx_with_chain( - registry: MockRegistry, - chain_client: MockChainClient, - ) -> Ctx { - Ctx { - registry, - chain_client, - agreement_conf: test_agreement_conf(), - } - } - - /// Records the `outcome` of every ERROR line carrying the alert's `event`. - #[derive(Clone, Default)] - struct AlertLines(Arc>>); - - impl AlertLines { - fn outcomes(&self) -> Vec { - self.0.lock().unwrap().clone() + unimplemented!() } - } - - impl tracing_subscriber::Layer for AlertLines { - fn on_event( - &self, - event: &tracing::Event<'_>, - _ctx: tracing_subscriber::layer::Context<'_, S>, - ) { - #[derive(Default)] - struct Fields { - event: Option, - outcome: Option, - } - impl tracing::field::Visit for Fields { - fn record_str(&mut self, field: &tracing::field::Field, value: &str) { - match field.name() { - "event" => self.event = Some(value.to_owned()), - "outcome" => self.outcome = Some(value.to_owned()), - _ => {} - } - } - fn record_debug( - &mut self, - _field: &tracing::field::Field, - _value: &dyn std::fmt::Debug, - ) { - } - } - if *event.metadata().level() != tracing::Level::ERROR { - return; - } - let mut fields = Fields::default(); - event.record(&mut fields); - if fields.event.as_deref() == Some("cancelled_agreement_live_on_chain") { - self.0 - .lock() - .unwrap() - .push(fields.outcome.unwrap_or_default()); - } + async fn latest_block_timestamp(&self) -> Result { + unimplemented!() } } - /// Run `handle` with alert lines captured; tokio tests run on one thread. - async fn handle_capturing_alerts( - ctx: Ctx, - agreement_id: IndexingAgreementId, - ) -> (JobResult<()>, Vec) { - use tracing_subscriber::layer::SubscriberExt; - let alerts = AlertLines::default(); - let _guard = - tracing::subscriber::set_default(tracing_subscriber::registry().with(alerts.clone())); - let result = handle(ctx, &Message { agreement_id }).await; - (result, alerts.outcomes()) - } - - #[tokio::test] - async fn logs_one_alert_line_for_a_live_agreement_it_ends() { - let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); - let agreement_id = agreement.id; - let ctx = ctx_with_chain(MockRegistry::new(agreement), MockChainClient::live()); - - let (result, alerts) = handle_capturing_alerts(ctx, agreement_id).await; - - assert!(result.is_ok(), "got {result:?}"); - assert_eq!(alerts, vec!["cancelled"]); - } - - #[tokio::test] - async fn logs_no_alert_line_for_an_agreement_that_already_ended() { - let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); - let agreement_id = agreement.id; - let ctx = ctx_with_chain(MockRegistry::new(agreement), MockChainClient::default()); - - let (result, alerts) = handle_capturing_alerts(ctx, agreement_id).await; - - assert!(result.is_ok(), "got {result:?}"); - assert!( - alerts.is_empty(), - "a stale snapshot must not alert: {alerts:?}" - ); - } - - #[tokio::test] - async fn logs_an_alert_line_for_each_failed_attempt_at_a_live_agreement() { - // The job can run out of retries; each failure is visible, so a live - // agreement dipper couldn't end is never silent. - let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); - let agreement_id = agreement.id; - let chain = MockChainClient { - fail_cancel: true, - ..MockChainClient::live() - }; - let ctx = ctx_with_chain(MockRegistry::new(agreement), chain); - - let (result, alerts) = handle_capturing_alerts(ctx, agreement_id).await; - - assert!( - matches!(result, Err(JobError::Retryable(_, _))), - "got {result:?}" - ); - assert_eq!(alerts, vec!["cancel_failed"]); - } - - #[tokio::test] - async fn fails_with_an_alert_line_when_a_live_agreement_cannot_be_cancelled() { - let mut agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); - agreement.terms_version_hash = None; - let agreement_id = agreement.id; - let chain = MockChainClient::live(); - let ctx = ctx_with_chain(MockRegistry::new(agreement), chain.clone()); - - let (result, alerts) = handle_capturing_alerts(ctx, agreement_id).await; - - assert!(matches!(result, Err(JobError::Fatal(_))), "got {result:?}"); - assert_eq!(alerts, vec!["cancel_impossible"]); - assert!(chain.cancelled.lock().unwrap().is_empty()); - } - - #[tokio::test] - async fn a_second_job_for_the_same_agreement_leaves_it_to_the_first() { - // The listener can queue an agreement twice before the chain shows it - // ended; 2 jobs at once would both cancel it and both alert. - let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); - let agreement_id = agreement.id; - let chain = MockChainClient::live(); - let ctx = ctx_with_chain(MockRegistry::new(agreement), chain.clone()); - let _first_job = Cancelling::claim(agreement_id).expect("not yet claimed"); - - let (result, alerts) = handle_capturing_alerts(ctx, agreement_id).await; - - assert!(result.is_ok(), "got {result:?}"); - assert!(alerts.is_empty(), "{alerts:?}"); - assert!(chain.cancelled.lock().unwrap().is_empty()); - } - - #[tokio::test] - async fn cancels_a_live_agreement_dipper_had_cancelled() { - // Dipper cancelled the agreement locally, but the indexer accepted its offer - // anyway. The job ends it on-chain and leaves the already-terminal row alone. - let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); - let agreement_id = agreement.id; - let registry = MockRegistry::new(agreement); - let chain = MockChainClient::live(); - - let result = handle( - ctx_with_chain(registry.clone(), chain.clone()), - &Message { agreement_id }, - ) - .await; - - assert!(result.is_ok(), "got {result:?}"); - assert_eq!( - *chain.cancelled.lock().unwrap(), - vec![*agreement_id.as_bytes()] - ); - assert!( - registry.marked_canceled.lock().unwrap().is_empty(), - "the row is already cancelled; it must not be marked again" - ); - } - - #[tokio::test] - async fn leaves_alone_an_agreement_dipper_cancelled_that_already_ended() { - // A stale snapshot can report an accept after dipper's own cancel already - // ended the agreement. Reading the chain first avoids a pointless cancel. - let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); - let agreement_id = agreement.id; - let chain = MockChainClient::default(); - - let result = handle( - ctx_with_chain(MockRegistry::new(agreement), chain.clone()), - &Message { agreement_id }, - ) - .await; - - assert!(result.is_ok(), "got {result:?}"); - assert!( - chain.cancelled.lock().unwrap().is_empty(), - "nothing to cancel" - ); - } - - #[tokio::test] - async fn retries_without_cancelling_when_the_chain_cannot_be_read() { - let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); - let agreement_id = agreement.id; - let chain = MockChainClient { - fail_liveness_read: true, - ..MockChainClient::live() + async fn run(status: IndexingAgreementStatus, live: bool) -> Vec { + let reopened = Arc::new(Mutex::new(Vec::new())); + let agreement = agreement(status, Some(vec![7u8; 32])); + let message = Message { + agreement_id: agreement.id, }; - - let result = handle( - ctx_with_chain(MockRegistry::new(agreement), chain.clone()), - &Message { agreement_id }, - ) - .await; - - assert!( - matches!(result, Err(JobError::Retryable(_, _))), - "got {result:?}" - ); - assert!(chain.cancelled.lock().unwrap().is_empty()); - } - - #[tokio::test] - async fn retries_when_cancelling_a_live_agreement_fails() { - let agreement = make_agreement(IndexingAgreementStatus::CanceledByRequester); - let agreement_id = agreement.id; - let chain = MockChainClient { - fail_cancel: true, - ..MockChainClient::live() + let ctx = Ctx { + registry: MockRegistry { + agreement, + reopened: Arc::clone(&reopened), + }, + chain_client: MockChain { live }, }; - let result = handle( - ctx_with_chain(MockRegistry::new(agreement), chain), - &Message { agreement_id }, - ) - .await; + handle(ctx, &message).await.expect("job ok"); - assert!( - matches!(result, Err(JobError::Retryable(_, _))), - "got {result:?}" - ); + reopened.lock().unwrap().clone() } #[tokio::test] - async fn rejected_agreement_records_cancel_audit_once() { - // The handler no longer emits `terminated` directly: it records the cancel - // audit, and the chain_listener sweep announces it durably. Assert exactly - // one audit was recorded for the agreement. - let agreement = make_agreement(IndexingAgreementStatus::Rejected); - let agreement_id = agreement.id; - let registry = MockRegistry::new(agreement); - - let result = handle(ctx_for(registry.clone()), &Message { agreement_id }).await; - assert!(result.is_ok(), "handle should succeed: {result:?}"); - - let recorded = registry.recorded_cancel_audit.lock().unwrap().clone(); - assert_eq!(recorded, vec![agreement_id], "exactly one cancel audit"); + async fn hands_a_live_agreement_dipper_ended_to_the_cancel_retry() { + for status in [ + IndexingAgreementStatus::Rejected, + IndexingAgreementStatus::CanceledByRequester, + ] { + assert_eq!(run(status, true).await.len(), 1, "{status}"); + } } #[tokio::test] - async fn failed_local_mark_records_no_cancel_audit() { - // The on-chain cancel succeeds but the DB mark fails, so the row stays - // non-terminal and no cancel audit is recorded: the listener sees the - // on-chain cancel, flips the row, and the sweep emits from there. - let agreement = make_agreement(IndexingAgreementStatus::Rejected); - let agreement_id = agreement.id; - let registry = MockRegistry::with_mark_failure(agreement); - - let result = handle(ctx_for(registry.clone()), &Message { agreement_id }).await; - assert!(result.is_ok(), "handle should still return Ok: {result:?}"); + async fn leaves_an_agreement_that_already_ended_or_is_still_wanted() { assert!( - registry.recorded_cancel_audit.lock().unwrap().is_empty(), - "no cancel audit recorded when the local mark failed" + run(IndexingAgreementStatus::CanceledByRequester, false) + .await + .is_empty() ); - } - - #[tokio::test] - async fn non_rejected_agreement_records_nothing() { - let agreement = make_agreement(IndexingAgreementStatus::Created); - let agreement_id = agreement.id; - let registry = MockRegistry::new(agreement); - - let result = handle(ctx_for(registry.clone()), &Message { agreement_id }).await; - assert!(result.is_ok(), "handle should return Ok: {result:?}"); assert!( - registry.recorded_cancel_audit.lock().unwrap().is_empty(), - "no cancel audit recorded for a non-Rejected agreement" + run(IndexingAgreementStatus::AcceptedOnChain, true) + .await + .is_empty() ); } } diff --git a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs index 51b46811..b9897a4f 100644 --- a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs +++ b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs @@ -932,15 +932,6 @@ mod lifecycle_event_tests { unimplemented!("not exercised by reassess handler") } - async fn cancel_rejected_agreement_on_chain( - &self, - agreement_id: IndexingAgreementId, - _priority: crate::worker::queue::JobPriority, - ) -> anyhow::Result { - self.cancels_queued.lock().unwrap().push(agreement_id); - Ok(crate::worker::queue::JobId::default()) - } - async fn submit_offer( &self, _agreement_id: IndexingAgreementId, @@ -956,11 +947,12 @@ mod lifecycle_event_tests { // ---- Mock: chain client -------------------------------------------------- - /// Always reports a successful cancel that the post-cancel read confirms, and - /// records the id of every agreement it was asked to cancel. + /// Shows every agreement live until a cancel is sent for it, unless nothing is on + /// chain, and records the id of every agreement it was asked to cancel. #[derive(Default, Clone)] struct MockChainClient { cancelled: Arc>>, + nothing_on_chain: bool, /// When set, each cancel records whether its agreement was already marked cancelling. marked_cancelling: Option>>>, marked_at_cancel: Arc>>, @@ -1018,10 +1010,9 @@ mod lifecycle_event_tests { async fn agreement_still_active( &self, - _agreement_id: &[u8; 16], + agreement_id: &[u8; 16], ) -> std::result::Result { - // Cancel confirmed: agreement is no longer active on-chain. - Ok(false) + Ok(!self.nothing_on_chain && !self.cancelled.lock().unwrap().contains(agreement_id)) } async fn agreement_ended_by_indexer( &self, @@ -1209,6 +1200,13 @@ mod lifecycle_event_tests { Ok(()) } + async fn reopen_indexing_agreement_cancel( + &self, + _id: &IndexingAgreementId, + ) -> crate::registry::Result<()> { + unimplemented!() + } + async fn get_cancelling_agreements( &self, _batch_size: i64, @@ -1921,6 +1919,22 @@ mod lifecycle_event_tests { assert!(cancelled.lock().unwrap().is_empty()); } + #[tokio::test] + async fn sends_no_cancel_for_an_offer_that_never_reached_the_chain() { + // A cancel of nothing still mines and costs gas. The mark made first means an + // offer still in flight withdraws itself when it lands. + let (mut ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::Created); + ctx.chain_client.nothing_on_chain = true; + let chain = ctx.chain_client.clone(); + let marked = ctx.registry.marked_cancelling.clone(); + + let result = handle(ctx, &test_message(0)).await; + + assert!(result.is_ok(), "got {result:?}"); + assert!(chain.cancelled.lock().unwrap().is_empty()); + assert_eq!(*marked.lock().unwrap(), vec![leaving.id]); + } + #[tokio::test] async fn marks_an_accepted_agreement_cancelled_once_its_cancel_lands() { let (ctx, leaving) = ctx_cancelling_one(IndexingAgreementStatus::AcceptedOnChain); diff --git a/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs b/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs index af9cbf83..a3139803 100644 --- a/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs +++ b/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs @@ -529,6 +529,13 @@ mod tests { Ok(()) } + async fn reopen_indexing_agreement_cancel( + &self, + _id: &IndexingAgreementId, + ) -> crate::registry::Result<()> { + unimplemented!() + } + async fn get_cancelling_agreements( &self, _batch_size: i64, @@ -753,14 +760,6 @@ mod tests { Ok(JobId::default()) } - async fn cancel_rejected_agreement_on_chain( - &self, - _agreement_id: IndexingAgreementId, - _priority: JobPriority, - ) -> anyhow::Result { - Ok(JobId::default()) - } - async fn submit_offer( &self, _agreement_id: IndexingAgreementId, diff --git a/bin/dipper-service/src/worker/service_queue.rs b/bin/dipper-service/src/worker/service_queue.rs index 2a0195cd..49ad37e1 100644 --- a/bin/dipper-service/src/worker/service_queue.rs +++ b/bin/dipper-service/src/worker/service_queue.rs @@ -4,19 +4,11 @@ use thegraph_core::{DeploymentId, alloy::primitives::ChainId}; use url::Url; use super::{ - handlers::{ - CancelRejectedAgreementOnChain, ReassessIndexingRequest, SendIndexingAgreementProposal, - SubmitOffer, - }, + handlers::{ReassessIndexingRequest, SendIndexingAgreementProposal, SubmitOffer}, messages::Message, queue::{JobId, JobPriority, Queue}, }; -/// Retries for the job that cancels on-chain an agreement dipper doesn't want but -/// that went live anyway: about 40 minutes of attempts at its 30 s backoff base, -/// since giving up leaves the indexer paid. -const CANCEL_ON_CHAIN_MAX_RETRIES: u32 = 10; - #[async_trait] pub trait WorkerQueue { async fn send_indexing_agreement_proposal( @@ -38,15 +30,6 @@ pub trait WorkerQueue { priority: JobPriority, ) -> anyhow::Result; - /// Cancel a rejected agreement on-chain. When an indexer rejected off-chain - /// but accepted on-chain, this cancels the agreement via - /// `cancelIndexingAgreementByPayer`. - async fn cancel_rejected_agreement_on_chain( - &self, - agreement_id: IndexingAgreementId, - priority: JobPriority, - ) -> anyhow::Result; - /// Submit an RCA offer on-chain as the first step of a new proposal. The /// job retries until the indexer's window to accept has closed, so a /// provider outage costs one offer only if it outlasts that window. @@ -129,22 +112,6 @@ where .await } - async fn cancel_rejected_agreement_on_chain( - &self, - agreement_id: IndexingAgreementId, - priority: JobPriority, - ) -> anyhow::Result { - self.queue - .push_with_max_retries( - Message::CancelRejectedAgreementOnChain(CancelRejectedAgreementOnChain { - agreement_id, - }), - priority, - CANCEL_ON_CHAIN_MAX_RETRIES, - ) - .await - } - async fn submit_offer( &self, agreement_id: IndexingAgreementId, @@ -275,27 +242,4 @@ mod tests { //* Assert assert_eq!(*queue.queue.pushes.lock().unwrap(), vec![None]); } - - /// Giving up on cancelling a live agreement dipper doesn't want leaves the - /// indexer paid, so that job keeps trying well past the queue default. - #[tokio::test] - async fn an_on_chain_cancel_carries_its_longer_retry_budget() { - //* Arrange - let queue = handle(4); - - //* Act - queue - .cancel_rejected_agreement_on_chain( - IndexingAgreementId::from_bytes([0; 16]), - JobPriority::Background, - ) - .await - .unwrap(); - - //* Assert - assert_eq!( - *queue.queue.pushes.lock().unwrap(), - vec![Some(CANCEL_ON_CHAIN_MAX_RETRIES)] - ); - } } diff --git a/dipper-pgregistry/src/postgres.rs b/dipper-pgregistry/src/postgres.rs index fb36f95b..485ab10e 100644 --- a/dipper-pgregistry/src/postgres.rs +++ b/dipper-pgregistry/src/postgres.rs @@ -971,6 +971,35 @@ impl PgRegistry { .await } + /// Move an agreement dipper had already ended, cancelled or rejected, back to `Cancelling` + /// once the chain shows it live after all, with its cancel attempts started afresh. + pub async fn reopen_indexing_agreement_cancel( + &self, + agreement_id: &IndexingAgreementId, + ) -> Result<(), Error> { + let updated = sqlx::query( + r#" + UPDATE dipper_reg_indexing_agreements + SET + status = $1, + cancel_attempts = 0, + cancel_checked_at = NULL, + updated_at = timezone('UTC', now()) + WHERE id = $2 AND status IN ($3, $4) + "#, + ) + .bind(IndexingAgreementStatus::Cancelling) + .bind(agreement_id) + .bind(IndexingAgreementStatus::CanceledByRequester) + .bind(IndexingAgreementStatus::Rejected) + .execute(&self.pool) + .await?; + if updated.rows_affected() == 0 { + return Err(Error::NoRecordsUpdated); + } + Ok(()) + } + /// `Cancelling` agreements marked over `min_age_minutes` ago whose cancel has failed /// fewer than `max_attempts` times, those checked longest ago first. One that may be paying /// an indexer (accepted, or past the offer deadline, which only an accepted one outlives) diff --git a/dipper-pgregistry/tests/it_registry_postgres.rs b/dipper-pgregistry/tests/it_registry_postgres.rs index ff1de13f..e82c5617 100644 --- a/dipper-pgregistry/tests/it_registry_postgres.rs +++ b/dipper-pgregistry/tests/it_registry_postgres.rs @@ -3482,6 +3482,47 @@ async fn cancelling_agreements_are_listed_until_their_cancel_fails_too_often() { assert!(matches!(not_cancelling, Err(Error::NoRecordsUpdated))); } +#[tokio::test] +async fn an_ended_agreement_found_live_on_chain_goes_back_to_cancelling() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let registry = PgRegistry::new(db); + let ended = fixture_agreement(0xaa); + let accepted = + IndexingAgreementId::from_bytes([0xaa, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2]); + registry + .mark_indexing_agreement_as_cancelling(&ended) + .await + .expect("mark cancelling"); + assert_eq!(registry.record_cancel_check(&ended, 2).await.unwrap(), 2); + registry + .mark_indexing_agreement_as_canceled_by_requester(&ended) + .await + .expect("mark ended"); + + registry + .reopen_indexing_agreement_cancel(&ended) + .await + .expect("an ended agreement can be reopened"); + + let listed = registry + .get_cancelling_agreements(100, 1, 0) + .await + .expect("cancelling query"); + let ids: Vec<_> = listed.iter().map(|row| row.agreement.id).collect(); + assert_eq!(ids, vec![ended], "its cancel attempts start afresh"); + let still_wanted = registry.reopen_indexing_agreement_cancel(&accepted).await; + assert!( + matches!(still_wanted, Err(Error::NoRecordsUpdated)), + "got {still_wanted:?}" + ); +} + #[tokio::test] async fn a_cancelling_agreement_stays_live_and_unannounced_until_it_ends() { let (db, _temp_db) = temp_registry_db().await; From ede3979f4ad13615d75b67bb2380b7abe227db37 Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Sat, 3 Oct 2026 01:12:32 +0300 Subject: [PATCH 11/26] fix: alert on a failing cancel and slow it to hourly (#721) * fix(cancel): count a cancel the contract refuses before it is sent A cancel the contract rejected while it was being prepared was never counted as a failed attempt, so one that always failed that way was retried, with an error logged, every 5 minutes for ever. It now counts like a cancel that reverts once mined, so it reaches the give-up limit and alert. * fix(cancel): keep checking cancels dipper gave up on until they end Once a cancel failed 10 times dipper never looked at the agreement again, so if it later ended it stayed cancelling for good, its fees counted and its indexer kept out of selection. It is now read hourly, without sending more cancels, and closed once the chain shows it ended. * fix(cancel): slow a failing cancel to hourly instead of stopping it Counting refusals before sending meant a paused manager used up every agreement's 10 attempts in under an hour, and dipper then never sent their cancels again once it was unpaused. It now alerts once at the limit and keeps retrying hourly, so those cancels resume by themselves. --- .../src/network/service/cancel_retry.rs | 74 ++++++++++++------- bin/dipper-service/src/registry/agreement.rs | 4 +- dipper-pgregistry/src/postgres.rs | 9 ++- .../tests/it_registry_postgres.rs | 19 ++++- 4 files changed, 74 insertions(+), 32 deletions(-) diff --git a/bin/dipper-service/src/network/service/cancel_retry.rs b/bin/dipper-service/src/network/service/cancel_retry.rs index 5b37a2dd..49a651a6 100644 --- a/bin/dipper-service/src/network/service/cancel_retry.rs +++ b/bin/dipper-service/src/network/service/cancel_retry.rs @@ -12,8 +12,8 @@ use crate::{ registry::{AgreementRegistry, CancelKind, CancellingAgreement, IndexingAgreement}, }; -/// Retried cancels mined without ending an agreement before dipper stops retrying it and -/// leaves it to an operator. Other failures don't count (see `failed_attempts`). +/// Failed cancels before dipper alerts an operator and retries the agreement only hourly, so a +/// paused manager recovers once unpaused. Outages don't count (see `failed_attempts`). pub const MAX_CANCEL_ATTEMPTS: u32 = 10; /// Agreements a sweep takes on, those that may be paying an indexer first; the time budget @@ -250,7 +250,7 @@ async fn note_check( { Ok(attempts) => { if let Some(err) = failure.filter(|_| failed_attempts > 0) { - log_failed_cancel(agreement, attempts, err); + log_failed_cancel(agreement, attempts, failed_attempts, err); } } Err(err) => tracing::warn!( @@ -261,26 +261,25 @@ async fn note_check( } } -/// A revert before sending is logged as an ERROR: a paused or misconfigured manager makes -/// every cancel revert, so it is retried rather than counted against the agreement. fn log_uncounted_failure(agreement: &IndexingAgreement, err: &ChainClientError) { - if matches!(err, ChainClientError::ContractRevert { .. }) { - tracing::error!( - agreement_id = %agreement.id, - error = %err, - "Cancel of an agreement reverted before it was sent, will retry" - ); - } else { - tracing::warn!( - agreement_id = %agreement.id, - error = %err, - "Cancel of an agreement failed or could not be confirmed, will retry" - ); - } + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Cancel of an agreement failed or could not be confirmed, will retry" + ); } -fn log_failed_cancel(agreement: &IndexingAgreement, attempts: u32, err: &ChainClientError) { - if attempts < MAX_CANCEL_ATTEMPTS { +/// One ERROR as an agreement reaches the limit, for an operator to look into; a WARN for +/// every other failed cancel. +fn log_failed_cancel( + agreement: &IndexingAgreement, + attempts: u32, + failed: u32, + err: &ChainClientError, +) { + let reached_limit = + attempts >= MAX_CANCEL_ATTEMPTS && attempts.saturating_sub(failed) < MAX_CANCEL_ATTEMPTS; + if !reached_limit { tracing::warn!( agreement_id = %agreement.id, attempts, @@ -290,22 +289,24 @@ fn log_failed_cancel(agreement: &IndexingAgreement, attempts: u32, err: &ChainCl return; } tracing::error!( - event = "agreement_cancel_abandoned", + event = "agreement_cancel_stuck", agreement_id = %agreement.id, indexer_id = %agreement.indexer.id, indexing_request_id = %agreement.indexing_request_id, attempts, error = %err, - "Gave up cancelling an agreement on-chain; it may still be live" + "Cancelling an agreement keeps failing; it may still be live. Dipper now retries it hourly" ); } -/// How many of an agreement's cancel attempts a failure uses up. A cancel mined without -/// ending it, or reverted, counts, and one that can never be sent uses them all. An -/// unreachable chain or a revert before sending is retried freely. +/// How many of an agreement's cancel attempts a failure uses up. A cancel the contract +/// refused, before sending or once mined, or that mined without ending the agreement counts, +/// and one that can never be sent uses them all. An unreachable chain is retried freely. fn failed_attempts(err: &ChainClientError) -> u32 { match err { - ChainClientError::CancelNotConfirmed { .. } | ChainClientError::TxReverted { .. } => 1, + ChainClientError::CancelNotConfirmed { .. } + | ChainClientError::TxReverted { .. } + | ChainClientError::ContractRevert { .. } => 1, ChainClientError::MissingTermsVersionHash { .. } => MAX_CANCEL_ATTEMPTS, _ => 0, } @@ -403,6 +404,7 @@ mod tests { read_fails: bool, send_fails: bool, mined_cancel_reverts: bool, + reverts_before_sending: bool, cancel_has_no_effect: bool, clock_fails: bool, now: AtomicU64, @@ -429,6 +431,12 @@ mod tests { if self.send_fails { return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); } + if self.reverts_before_sending { + return Err(ChainClientError::ContractRevert { + selector: [0xde, 0xad, 0xbe, 0xef], + data: Default::default(), + }); + } if self.mined_cancel_reverts { return Err(ChainClientError::TxReverted { tx_hash: B256::repeat_byte(0xee), @@ -721,6 +729,20 @@ mod tests { assert_eq!(registry.attempts.load(Ordering::SeqCst), 1); } + #[tokio::test] + async fn counts_a_cancel_the_contract_refuses_before_it_is_sent() { + // Otherwise one that always reverts is retried, and alerted on, for ever. + let registry = registry_with_one(true); + let chain = MockChain { + reverts_before_sending: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(registry.attempts.load(Ordering::SeqCst), 1); + } + #[tokio::test] async fn counts_a_cancel_that_mines_without_ending_the_agreement() { let registry = registry_with_one(true); diff --git a/bin/dipper-service/src/registry/agreement.rs b/bin/dipper-service/src/registry/agreement.rs index 762848a8..24122084 100644 --- a/bin/dipper-service/src/registry/agreement.rs +++ b/bin/dipper-service/src/registry/agreement.rs @@ -281,8 +281,8 @@ pub trait AgreementRegistry { id: &IndexingAgreementId, ) -> RegistryResult<()>; - /// `CANCELLING` agreements marked over `min_age_minutes` ago whose cancel has failed - /// fewer than `max_attempts` times, those checked longest ago first. One that may be paying + /// `CANCELLING` agreements marked over `min_age_minutes` ago, those checked longest ago + /// first; one whose cancel has failed `max_attempts` times only once an hour. One that may be paying /// an indexer (accepted, or past the offer deadline, which only an accepted one outlives) /// counts as checked an hour earlier, so it goes first without holding the rest back. async fn get_cancelling_agreements( diff --git a/dipper-pgregistry/src/postgres.rs b/dipper-pgregistry/src/postgres.rs index 485ab10e..3e7d062b 100644 --- a/dipper-pgregistry/src/postgres.rs +++ b/dipper-pgregistry/src/postgres.rs @@ -1000,8 +1000,8 @@ impl PgRegistry { Ok(()) } - /// `Cancelling` agreements marked over `min_age_minutes` ago whose cancel has failed - /// fewer than `max_attempts` times, those checked longest ago first. One that may be paying + /// `Cancelling` agreements marked over `min_age_minutes` ago, those checked longest ago + /// first; one whose cancel has failed `max_attempts` times only once an hour. One that may be paying /// an indexer (accepted, or past the offer deadline, which only an accepted one outlives) /// counts as checked an hour earlier, so it goes first without holding the rest back. pub async fn get_cancelling_agreements( @@ -1030,7 +1030,10 @@ impl PgRegistry { accepted_at FROM dipper_reg_indexing_agreements WHERE status = $1 - AND cancel_attempts < $2 + AND ( + cancel_attempts < $2 + OR cancel_checked_at < timezone('UTC', now()) - INTERVAL '1 hour' + ) AND updated_at < timezone('UTC', now()) - make_interval(mins => $4) ORDER BY cancel_checked_at - CASE diff --git a/dipper-pgregistry/tests/it_registry_postgres.rs b/dipper-pgregistry/tests/it_registry_postgres.rs index e82c5617..215e946c 100644 --- a/dipper-pgregistry/tests/it_registry_postgres.rs +++ b/dipper-pgregistry/tests/it_registry_postgres.rs @@ -3473,7 +3473,24 @@ async fn cancelling_agreements_are_listed_until_their_cancel_fails_too_often() { assert_eq!( ids, vec![accepted], - "one that failed too often is left alone" + "one that failed too often waits an hour between checks" + ); + + sqlx::query( + "UPDATE dipper_reg_indexing_agreements \ + SET cancel_checked_at = cancel_checked_at - INTERVAL '2 hours' WHERE id = $1", + ) + .bind(created) + .execute(&db) + .await + .expect("Failed to age the check"); + let listed = registry + .get_cancelling_agreements(100, 2, 0) + .await + .expect("cancelling query"); + assert!( + listed.iter().any(|row| row.agreement.id == created), + "and is tried again after it" ); let not_cancelling = registry From 0eced9a60f3050ea35df80b3b73a0d73a5ca9a3c Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Sat, 3 Oct 2026 01:12:33 +0300 Subject: [PATCH 12/26] test: build registry test mocks on the shared stub (#722) * docs(config): write the test settings' zero value as a numeral The doc comment on the shared test settings spelled the configured value 0 as a word, against the house rule that quantities and configured values are written as numerals. * test(registry): build 3 test mocks on the shared registry stub Three test mocks implemented every registry method by hand, most as placeholders, so each new registry method meant editing all of them. They now build on the shared stub and keep only the methods their tests rely on. --- bin/dipper-service/src/config.rs | 2 +- .../src/network/service/chain_listener.rs | 29 ++-- .../handlers/reassess_indexing_request.rs | 130 +----------------- .../send_indexing_agreement_proposal.rs | 26 +--- 4 files changed, 15 insertions(+), 172 deletions(-) diff --git a/bin/dipper-service/src/config.rs b/bin/dipper-service/src/config.rs index 740974f3..73ee033a 100644 --- a/bin/dipper-service/src/config.rs +++ b/bin/dipper-service/src/config.rs @@ -1204,7 +1204,7 @@ pub struct IndexingAgreementConfig { #[cfg(test)] impl IndexingAgreementConfig { - /// Zero addresses and limits, with permissive breaker and cache settings, for tests + /// Addresses and limits all set to 0, with permissive breaker and cache settings, for tests /// to adjust the fields they care about. pub fn for_tests() -> Self { Self { diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index 585626ca..80111bad 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -2063,7 +2063,7 @@ mod tests { } #[async_trait::async_trait] - impl AgreementRegistry for MockRegistry { + impl crate::registry::StubAgreementRegistry for MockRegistry { async fn get_indexing_agreement_by_id( &self, id: &IndexingAgreementId, @@ -2200,23 +2200,6 @@ mod tests { Ok(()) } - async fn get_cancelling_agreements( - &self, - _batch_size: i64, - _max_attempts: u32, - _min_age_minutes: i32, - ) -> RegistryResult> { - Ok(Vec::new()) - } - - async fn record_cancel_check( - &self, - _id: &IndexingAgreementId, - failed_attempts: u32, - ) -> RegistryResult { - Ok(failed_attempts) - } - async fn record_cancel_audit( &self, agreement_id: &IndexingAgreementId, @@ -2345,9 +2328,13 @@ mod tests { } let mut outcomes = std::collections::HashMap::with_capacity(items.len()); for item in items { - let outcome = self - .apply_reconciliation(&item.agreement_id, item.apply_accept, item.cancel) - .await?; + let outcome = crate::registry::StubAgreementRegistry::apply_reconciliation( + self, + &item.agreement_id, + item.apply_accept, + item.cancel, + ) + .await?; outcomes.insert(item.agreement_id, outcome); } Ok(outcomes) diff --git a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs index b9897a4f..d2cf1d2a 100644 --- a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs +++ b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs @@ -839,11 +839,10 @@ mod lifecycle_event_tests { }, }, registry::{ - AgreementFeeRate, AgreementRegistry, CancelKind, Indexer, IndexerDenylistRegistry, - IndexingAgreement, IndexingAgreementStatus, IndexingAgreementTerms, - IndexingAgreementTermsMetadata, IndexingRequest, IndexingRequestRegistry, - NewAgreementParams, PendingCancellation, PendingCancellationRegistry, - ReconciliationItem, ReconciliationOutcome, Result as RegistryResult, SetTargetOutcome, + AgreementFeeRate, Indexer, IndexerDenylistRegistry, IndexingAgreement, + IndexingAgreementStatus, IndexingAgreementTerms, IndexingAgreementTermsMetadata, + IndexingRequest, IndexingRequestRegistry, NewAgreementParams, PendingCancellation, + PendingCancellationRegistry, Result as RegistryResult, SetTargetOutcome, }, signing::eip712::Eip712Signer, test_support::{CapturedEvent, CapturingEventsProducer}, @@ -1089,13 +1088,7 @@ mod lifecycle_event_tests { } #[async_trait] - impl AgreementRegistry for MockRegistry { - async fn get_indexing_agreement_by_id( - &self, - _id: &IndexingAgreementId, - ) -> RegistryResult> { - unimplemented!() - } + impl crate::registry::StubAgreementRegistry for MockRegistry { // gather_selection_context: all active agreements for the deployment. async fn get_indexing_agreements_by_deployment_id( &self, @@ -1103,12 +1096,6 @@ mod lifecycle_event_tests { ) -> RegistryResult> { Ok(self.active_agreements.clone()) } - async fn get_indexing_agreements_by_indexer_id( - &self, - _indexer_id: &IndexerId, - ) -> RegistryResult> { - unimplemented!() - } // gather_selection_context: no pending agreements. async fn get_pending_agreement_indexers_by_deployment( &self, @@ -1126,12 +1113,6 @@ mod lifecycle_event_tests { ) -> RegistryResult>> { Ok(HashMap::new()) } - async fn get_indexing_agreements_by_indexing_request_id( - &self, - _request_id: &IndexingRequestId, - ) -> RegistryResult> { - unimplemented!() - } // The handler's current-state baseline for the diff. async fn get_active_indexing_agreements_by_indexing_request_id( &self, @@ -1166,24 +1147,11 @@ mod lifecycle_event_tests { ) -> RegistryResult> { Ok(Vec::new()) } - async fn mark_indexing_agreement_as_unresponsive( - &self, - _id: &IndexingAgreementId, - ) -> RegistryResult<()> { - unimplemented!() - } async fn count_created_agreements_by_indexer( &self, ) -> RegistryResult<(std::collections::HashMap, u64)> { Ok((std::collections::HashMap::new(), 0)) } - async fn update_offer_tx_hash( - &self, - _id: &IndexingAgreementId, - _tx_hash: &[u8; 32], - ) -> RegistryResult<()> { - unimplemented!() - } // Cancel path: pre-mark the local row terminal. async fn mark_indexing_agreement_as_canceled_by_requester( &self, @@ -1200,94 +1168,6 @@ mod lifecycle_event_tests { Ok(()) } - async fn reopen_indexing_agreement_cancel( - &self, - _id: &IndexingAgreementId, - ) -> crate::registry::Result<()> { - unimplemented!() - } - - async fn get_cancelling_agreements( - &self, - _batch_size: i64, - _max_attempts: u32, - _min_age_minutes: i32, - ) -> RegistryResult> { - Ok(Vec::new()) - } - - async fn record_cancel_check( - &self, - _id: &IndexingAgreementId, - failed_attempts: u32, - ) -> RegistryResult { - Ok(failed_attempts) - } - async fn apply_reconciliation( - &self, - _id: &IndexingAgreementId, - _apply_accept: bool, - _cancel: Option, - ) -> RegistryResult { - unimplemented!() - } - async fn apply_reconciliation_batch( - &self, - _items: &[ReconciliationItem], - ) -> RegistryResult> { - unimplemented!() - } - async fn get_expired_created_agreements( - &self, - _batch_size: i64, - _chain_timestamp: u64, - ) -> RegistryResult> { - unimplemented!() - } - async fn mark_indexing_agreement_as_expired( - &self, - _id: &IndexingAgreementId, - ) -> RegistryResult<()> { - unimplemented!() - } - async fn mark_indexing_agreement_as_rejected( - &self, - _id: &IndexingAgreementId, - _rejection_reason: Option<&str>, - ) -> RegistryResult<()> { - unimplemented!() - } - async fn get_accepted_on_chain_agreements( - &self, - _batch_size: i64, - ) -> RegistryResult> { - unimplemented!() - } - async fn get_agreements_pending_chain_cancel( - &self, - _batch_size: i64, - ) -> RegistryResult> { - unimplemented!() - } - async fn update_agreement_sync_progress( - &self, - _id: &IndexingAgreementId, - _block_height: u64, - _progress_at: OffsetDateTime, - ) -> RegistryResult<()> { - unimplemented!() - } - async fn count_active_agreements_by_deployment( - &self, - ) -> RegistryResult> { - unimplemented!() - } - async fn mark_indexing_agreement_as_abandoned( - &self, - _id: &IndexingAgreementId, - ) -> RegistryResult { - unimplemented!() - } // gather_selection_context: optimistic DIPs fees (none). async fn get_agreement_fee_rates(&self) -> RegistryResult> { Ok(Vec::new()) diff --git a/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs b/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs index a3139803..1022faff 100644 --- a/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs +++ b/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs @@ -415,7 +415,7 @@ mod tests { } #[async_trait] - impl AgreementRegistry for MockRegistry { + impl crate::registry::StubAgreementRegistry for MockRegistry { async fn get_indexing_agreement_by_id( &self, _id: &IndexingAgreementId, @@ -529,30 +529,6 @@ mod tests { Ok(()) } - async fn reopen_indexing_agreement_cancel( - &self, - _id: &IndexingAgreementId, - ) -> crate::registry::Result<()> { - unimplemented!() - } - - async fn get_cancelling_agreements( - &self, - _batch_size: i64, - _max_attempts: u32, - _min_age_minutes: i32, - ) -> crate::registry::Result> { - Ok(Vec::new()) - } - - async fn record_cancel_check( - &self, - _id: &IndexingAgreementId, - failed_attempts: u32, - ) -> crate::registry::Result { - Ok(failed_attempts) - } - async fn apply_reconciliation( &self, _id: &IndexingAgreementId, From 95852d6b6020af0636bf5ee2f2ec455894338729 Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Mon, 5 Oct 2026 12:10:51 +0100 Subject: [PATCH 13/26] fix: keep a faulty RPC endpoint from blocking chain reads (#726) * fix(chain): ignore RPC blocks too far ahead to be real Dipper never reads from an endpoint behind the newest block it has seen, so one faulty endpoint reporting a far-off block refused every later read. Blocks over a week ahead, plus the time since, are now ignored, and an ERROR fires when every endpoint is far off. * fix(chain): pass over a lagging RPC endpoint quickly and quietly An endpoint a block behind one dipper has seen is routine, yet each read backed off from it for seconds with a warning every time. It is now asked once more after half a second, then the next endpoint is tried, with nothing above debug in the logs. * fix(chain): trust the newest block only once the chain is seen moving An endpoint stuck far behind that answered first after a restart set the newest block, and then every endpoint that was right looked too far ahead. It is now trusted only after 2 blocks show the chain moving, and the ERROR fires only when every endpoint, not an outage, is far off. * fix(chain): wait a little longer for a lagging RPC endpoint The read just after a transaction mines often lands on a node a few blocks behind, and 1 retry half a second later could still find it behind, failing the read in a pool of 1. It is now asked up to twice more before the next endpoint is tried. * fix(chain): confirm the newest block by receipt or agreeing endpoints Two rising reads from one endpoint confirmed its block, so one moving but far behind shut out the rest, and until then a lagging endpoint wasn't refused. The block now never goes back, and only a receipt or 2 endpoints agreeing confirm it before far-ahead blocks are refused. --- bin/dipper-service/src/chain_client/client.rs | 460 +++++++++++++++++- .../src/chain_client/rpc_provider.rs | 211 ++++++-- 2 files changed, 600 insertions(+), 71 deletions(-) diff --git a/bin/dipper-service/src/chain_client/client.rs b/bin/dipper-service/src/chain_client/client.rs index 9cc58a9d..9fe31298 100644 --- a/bin/dipper-service/src/chain_client/client.rs +++ b/bin/dipper-service/src/chain_client/client.rs @@ -8,7 +8,7 @@ use std::{ Arc, atomic::{AtomicU64, Ordering}, }, - time::Duration, + time::{Duration, Instant}, }; use async_trait::async_trait; @@ -57,6 +57,197 @@ const RECEIPT_POLL_INTERVAL: Duration = Duration::from_millis(500); /// pre-acceptance) terms. `getAgreementDetails(id, 0)` reports their state. const VERSION_CURRENT: u64 = 0; +/// Blocks Arbitrum adds in a week, at its 4 a second. +const WEEK_OF_BLOCKS: u64 = 2_419_200; + +/// The most blocks Arbitrum adds a second. +const BLOCKS_PER_SECOND: u64 = 4; + +/// An hour of blocks. A read every endpoint fails while the nearest is further than this from +/// the newest block dipper has seen raises an ERROR; nearer is lag that clears by itself. +const ALERT_GAP_BLOCKS: u64 = 14_400; + +/// How a read refused by an endpoint whose latest block is too far ahead to be real says so. +const FAR_AHEAD_OF_A_SEEN_BLOCK: &str = "too far ahead of the newest block seen"; + +/// How often, at most, dipper asks every endpoint for its latest block to confirm the newest +/// block it has seen, while that block is unconfirmed. +const CROSS_CHECK_INTERVAL: Duration = Duration::from_secs(60); + +/// The newest block dipper has seen, from reads, receipts and latest-block lookups, and when it +/// last moved. It never goes backwards, so a lagging endpoint can't show state from before it. +/// Blocks too far ahead of it are refused only once it is confirmed, by the receipt for one of +/// dipper's transactions or by 2 endpoints agreeing, so an endpoint stuck far behind, or far +/// ahead, that answers first after a restart can't shut out the endpoints that are right. +#[derive(Debug)] +struct SeenBlock { + number: u64, + moved_at: Instant, + confirmed: bool, + cross_checked_at: Option, +} + +impl SeenBlock { + fn new() -> Self { + Self { + number: 0, + moved_at: Instant::now(), + confirmed: false, + cross_checked_at: None, + } + } + + /// The highest block that believably follows the newest one seen: a week of blocks past + /// it, plus what the chain can have added since. + fn believable_limit(&self, now: Instant) -> u64 { + let since = now.saturating_duration_since(self.moved_at).as_secs(); + self.number + .saturating_add(WEEK_OF_BLOCKS) + .saturating_add(since.saturating_mul(BLOCKS_PER_SECOND)) + } + + /// The lowest and highest latest block an endpoint may report to be read from: from the + /// newest seen to its believable limit once confirmed, with no limit until then. + fn bounds(&self, now: Instant) -> (u64, u64) { + let highest = if self.confirmed { + self.believable_limit(now) + } else { + u64::MAX + }; + (self.number, highest) + } + + /// Take `block` as seen, unless it is too far ahead of a confirmed one to be real; false + /// when it is. + fn advance(&mut self, block: u64, now: Instant) -> bool { + if self.confirmed && block > self.believable_limit(now) { + return false; + } + if block > self.number { + self.number = block; + self.moved_at = now; + } + true + } + + /// Take `block` as one the chain is known to have reached. An unconfirmed newest block more + /// than an hour of blocks past it came from a faulty endpoint, so it is replaced. + fn confirm(&mut self, block: u64, now: Instant) { + if self.confirmed { + self.advance(block, now); + return; + } + if block > self.number || self.number > block.saturating_add(ALERT_GAP_BLOCKS) { + self.number = block; + self.moved_at = now; + } + self.confirmed = true; + } + + /// Whether to ask every endpoint for its latest block: only while unconfirmed, and at most + /// once every [`CROSS_CHECK_INTERVAL`]. + fn cross_check_due(&mut self, now: Instant) -> bool { + let checked_lately = self + .cross_checked_at + .is_some_and(|at| now.saturating_duration_since(at) < CROSS_CHECK_INTERVAL); + if self.confirmed || checked_lately { + return false; + } + self.cross_checked_at = Some(now); + true + } +} + +/// The newest of the latest blocks that the most endpoints agree on, at least 2 of them within +/// an hour of blocks of each other, or `None` when no 2 agree. +fn agreed_head(mut heads: Vec) -> Option { + heads.sort_unstable(); + let mut agreed: Option<(usize, u64)> = None; + let mut start = 0; + for (end, &head) in heads.iter().enumerate() { + while head - heads[start] > ALERT_GAP_BLOCKS { + start += 1; + } + let agreeing = end - start + 1; + if agreeing >= 2 && agreed.is_none_or(|(most, _)| agreeing >= most) { + agreed = Some((agreeing, head)); + } + } + agreed.map(|(_, head)| head) +} + +/// Why a read's attempts were refused, to tell a wrong newest block, or endpoints all far out +/// of step, from an outage. +struct Refusals { + /// How far, in blocks, the nearest endpoint refused for its latest block was off. + nearest: AtomicU64, + /// Attempts that failed for any other reason. + other_failures: AtomicU64, +} + +impl Refusals { + fn new() -> Self { + Self { + nearest: AtomicU64::new(u64::MAX), + other_failures: AtomicU64::new(0), + } + } + + /// Refuse an endpoint whose latest block is older than the newest one seen. + fn behind(&self, head: u64, seen: u64) -> Result<(), TransportError> { + if head >= seen { + return Ok(()); + } + self.nearest.fetch_min(seen - head, Ordering::Relaxed); + Err(TransportErrorKind::custom_str(&format!( + "endpoint is at block {head}, {BEHIND_A_SEEN_BLOCK} ({seen})" + ))) + } + + /// Refuse an endpoint whose latest block is too far ahead of the newest one seen to be real. + fn far_ahead(&self, head: u64, seen: u64, highest: u64) -> Result<(), TransportError> { + if head <= highest { + return Ok(()); + } + self.nearest.fetch_min(head - seen, Ordering::Relaxed); + Err(TransportErrorKind::custom_str(&format!( + "endpoint is at block {head}, {FAR_AHEAD_OF_A_SEEN_BLOCK} ({seen})" + ))) + } + + /// Count an attempt that failed for another reason. + fn other(&self, result: Result) -> Result { + if result.is_err() { + self.other_failures.fetch_add(1, Ordering::Relaxed); + } + result + } + + /// Whether every attempt was refused for its block, the nearest over an hour of blocks + /// off: then the endpoints, or the newest block seen, are wrong, not briefly behind. + fn all_far_off(&self) -> bool { + let nearest = self.nearest.load(Ordering::Relaxed); + self.other_failures.load(Ordering::Relaxed) == 0 + && nearest != u64::MAX + && nearest > ALERT_GAP_BLOCKS + } + + /// An ERROR for a read that failed because every endpoint was far out of step. + fn alert(&self, operation: &str, seen: u64) { + if self.all_far_off() { + tracing::error!( + event = "rpc_blocks_refused", + operation, + seen_block = seen, + blocks_off = self.nearest.load(Ordering::Relaxed), + "Every RPC endpoint is over an hour of blocks from the newest block dipper has \ + seen, so chain reads are refused; check the endpoints are in sync and on the \ + right chain, then restart dipper" + ); + } + } +} + /// `AgreementDetails.state` flags from `IAgreementCollector.sol` (REGISTERED=1, /// ACCEPTED=2, NOTICE_GIVEN=4, SETTLED=8, BY_PROVIDER=32). `getAgreementDetails` keeps /// ACCEPTED set on a canceled agreement and ORs in NOTICE_GIVEN, so a cancel must clear it, @@ -240,9 +431,10 @@ struct AlloyChainClientInner { submit_lock: Mutex<()>, /// How long one submission may hold `submit_lock`; see `derive_submit_deadline`. submit_deadline: Duration, - /// The newest block dipper has seen, from reads and receipts. An agreement's state is - /// never read from an endpoint behind it, so a lagging endpoint can't undo what dipper saw. - seen_block: AtomicU64, + /// The newest block dipper has seen. An agreement's state is never read from an endpoint + /// behind it, so a lagging endpoint can't undo what dipper saw, nor from one too far + /// ahead of it to be real, so a faulty endpoint can't refuse every read after it. + seen_block: std::sync::Mutex, } impl AlloyChainClient { @@ -293,7 +485,7 @@ impl AlloyChainClient { nonce: AtomicU64::new(NONCE_UNINITIALIZED), submit_lock: Mutex::new(()), submit_deadline, - seen_block: AtomicU64::new(0), + seen_block: std::sync::Mutex::new(SeenBlock::new()), }), }) } @@ -623,8 +815,8 @@ impl AlloyChainClient { } /// Run a read-only contract call at the endpoint's latest block, refusing an endpoint - /// whose latest block is older than one dipper has already seen; the pool moves on to - /// the next endpoint instead. + /// whose latest block is older than one dipper has already seen, or too far ahead of it + /// to be real; the pool moves on to the next endpoint instead. async fn view_at_seen_block( &self, to: Address, @@ -632,34 +824,75 @@ impl AlloyChainClient { operation: &'static str, ) -> Result { let calldata = call.abi_encode(); - let seen = self.inner.seen_block.load(Ordering::Relaxed); - let (head, output) = self + self.cross_check_seen_block().await; + let (seen, highest) = self.block_bounds(); + let refusals = Refusals::new(); + let read = self .inner .rpc_pool .execute(operation, |provider| { let calldata = calldata.clone(); + let refusals = &refusals; async move { - let head = provider.get_block_number().await?; - if head < seen { - return Err(TransportErrorKind::custom_str(&format!( - "endpoint is at block {head}, {BEHIND_A_SEEN_BLOCK} ({seen})" - ))); - } + let head = refusals.other(provider.get_block_number().await)?; + refusals.behind(head, seen)?; + refusals.far_ahead(head, seen, highest)?; let tx = TransactionRequest::default().to(to).input(calldata.into()); - let output = provider.call(tx).block(BlockId::number(head)).await?; + let output = + refusals.other(provider.call(tx).block(BlockId::number(head)).await)?; Ok((head, output)) } }) - .await?; + .await; + let (head, output) = read.inspect_err(|_| refusals.alert(operation, seen))?; self.note_block(head); C::abi_decode_returns(&output).map_err(|err| { ChainClientError::RpcError(anyhow::anyhow!("undecodable {operation} from {to}: {err}")) }) } - /// Remember a block dipper has seen, so later reads are never older. + fn seen_block(&self) -> std::sync::MutexGuard<'_, SeenBlock> { + self.inner + .seen_block + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + } + + /// The lowest and highest latest block an endpoint may report to be read from. + fn block_bounds(&self) -> (u64, u64) { + self.seen_block().bounds(Instant::now()) + } + + /// Confirm the newest block seen from the endpoints' own agreement, if it isn't yet and + /// there are at least 2 to agree. + async fn cross_check_seen_block(&self) { + let pool = &self.inner.rpc_pool; + if pool.endpoint_count() < 2 || !self.seen_block().cross_check_due(Instant::now()) { + return; + } + match agreed_head(pool.latest_blocks().await) { + Some(head) => self.seen_block().confirm(head, Instant::now()), + None => tracing::warn!( + "No 2 RPC endpoints agree on the chain's latest block; reads aren't checked \ + against blocks too far ahead until they do" + ), + } + } + + /// Remember a block dipper has seen, so later reads are never older. One too far ahead to + /// be real is ignored, so it can't refuse every read after it. fn note_block(&self, block: u64) { - self.inner.seen_block.fetch_max(block, Ordering::Relaxed); + let (taken, seen) = { + let mut seen = self.seen_block(); + (seen.advance(block, Instant::now()), seen.number) + }; + if !taken { + tracing::warn!( + block, + seen_block = seen, + "Ignoring a block too far ahead of the newest block dipper has seen" + ); + } } /// Run a read-only contract call and decode its return value. @@ -707,7 +940,7 @@ impl AlloyChainClient { match receipt { Ok(Some(r)) => { if let Some(block) = r.block_number { - self.note_block(block); + self.seen_block().confirm(block, Instant::now()); } return Ok(Some(r.status())); } @@ -734,13 +967,27 @@ impl AlloyChainClient { #[async_trait] impl ChainClient for AlloyChainClient { async fn latest_block_timestamp(&self) -> Result { - let block = self + self.cross_check_seen_block().await; + let (seen, highest) = self.block_bounds(); + let refusals = Refusals::new(); + let read = self .inner .rpc_pool - .execute("get_latest_block", |provider| async move { - provider.get_block_by_number(BlockNumberOrTag::Latest).await + .execute("get_latest_block", |provider| { + let refusals = &refusals; + async move { + let block = refusals + .other(provider.get_block_by_number(BlockNumberOrTag::Latest).await)?; + // A lagging endpoint's time is only a little early, which brings nothing forward. + if let Some(block) = &block { + refusals.far_ahead(block.header.number, seen, highest)?; + } + Ok(block) + } }) - .await? + .await; + let block = read + .inspect_err(|_| refusals.alert("get_latest_block", seen))? .ok_or_else(|| { ChainClientError::RpcError(anyhow::anyhow!("no latest block returned")) })?; @@ -2173,6 +2420,7 @@ mod tests { lagging.uri().parse().expect("provider URL"), current.uri().parse().expect("provider URL"), ]); + // Not yet confirmed, as after a restart: a lagging endpoint is refused all the same. client.note_block(95); let live = client @@ -2181,7 +2429,7 @@ mod tests { .expect("read"); assert!(!live, "read from the endpoint that has reached block 95"); - assert_eq!(client.inner.seen_block.load(Ordering::Relaxed), 100); + assert_eq!(client.seen_block().number, 100); } #[tokio::test] @@ -2195,6 +2443,168 @@ mod tests { assert!(read.is_err(), "got {read:?}"); } + #[tokio::test] + async fn skips_an_endpoint_too_far_ahead_to_be_real() { + // Otherwise its block would refuse every read from the endpoints that are right. + let faulty = server_at_block(100 + WEEK_OF_BLOCKS + 1_000, STATE_REGISTERED).await; + let current = + server_at_block(150, STATE_REGISTERED | STATE_ACCEPTED | STATE_NOTICE_GIVEN).await; + let client = client_over(vec![ + faulty.uri().parse().expect("provider URL"), + current.uri().parse().expect("provider URL"), + ]); + trust_block(&client, 100); + + let live = client + .agreement_still_active(&[0xab; 16]) + .await + .expect("read"); + + assert!(!live, "read from the endpoint at block 150"); + assert_eq!(client.seen_block().number, 150); + } + + /// Set the newest block seen as a confirmed one, as a receipt for dipper's transaction does. + fn trust_block(client: &AlloyChainClient, block: u64) { + client.seen_block().confirm(block, Instant::now()); + } + + #[tokio::test] + async fn takes_the_first_block_after_a_restart_however_far_ahead() { + // Dipper may have been down for over a week. + let endpoint = server_at_block(WEEK_OF_BLOCKS * 3, STATE_REGISTERED).await; + let client = client_over(vec![endpoint.uri().parse().expect("provider URL")]); + + client + .agreement_still_active(&[0xab; 16]) + .await + .expect("read"); + + assert_eq!(client.seen_block().number, WEEK_OF_BLOCKS * 3); + } + + #[test] + fn ignores_a_receipt_block_too_far_ahead_to_be_real() { + let client = client_over(vec!["http://127.0.0.1:1".parse().expect("provider URL")]); + trust_block(&client, 100); + + client.note_block(100 + WEEK_OF_BLOCKS + 1); + assert_eq!(client.seen_block().number, 100); + + client.note_block(100 + WEEK_OF_BLOCKS); + assert_eq!(client.seen_block().number, 100 + WEEK_OF_BLOCKS); + } + + #[test] + fn believes_a_week_of_blocks_ahead_plus_what_the_chain_added_since() { + // A quiet spell with no reads mustn't make the chain's real progress look faulty. + let now = Instant::now(); + let seen = SeenBlock { + number: 100, + moved_at: now.checked_sub(Duration::from_secs(600)).expect("instant"), + confirmed: true, + cross_checked_at: None, + }; + + assert_eq!(seen.bounds(now), (100, 100 + WEEK_OF_BLOCKS + 2_400)); + assert_eq!(SeenBlock::new().bounds(now), (0, u64::MAX)); + } + + #[test] + fn never_goes_back_even_before_it_is_confirmed() { + // So a read just after dipper's cancel mined can't come from an endpoint behind it. + let now = Instant::now(); + let mut seen = SeenBlock::new(); + assert!(seen.advance(100, now)); + assert!(seen.advance(90, now)); + + assert_eq!(seen.bounds(now), (100, u64::MAX)); + } + + #[test] + fn refuses_blocks_far_ahead_only_once_confirmed() { + // An endpoint stuck far behind may answer first after a restart; refusing what is + // far ahead of its block would shut out every endpoint that is right. + let now = Instant::now(); + let mut seen = SeenBlock::new(); + let real = 1_000 + WEEK_OF_BLOCKS * 3; + assert!(seen.advance(1_000, now)); + assert!(seen.advance(real, now), "taken while unconfirmed"); + + seen.confirm(real, now); + + assert!(!seen.advance(real + WEEK_OF_BLOCKS * 2, now)); + assert_eq!(seen.bounds(now).0, real); + } + + #[test] + fn a_confirmed_block_replaces_one_far_ahead_that_was_never_confirmed() { + let now = Instant::now(); + let mut seen = SeenBlock::new(); + let real = 5_000_000; + assert!(seen.advance(real + WEEK_OF_BLOCKS * 10, now)); + + seen.confirm(real, now); + + assert_eq!(seen.bounds(now).0, real); + } + + #[test] + fn agrees_on_the_head_most_endpoints_share() { + let real = 5_000_000; + assert_eq!(agreed_head(vec![1_000, real, real + 3]), Some(real + 3)); + assert_eq!( + agreed_head(vec![real + WEEK_OF_BLOCKS, real + 2, real]), + Some(real + 2) + ); + assert_eq!(agreed_head(vec![1_000, real]), None, "2 that disagree"); + assert_eq!(agreed_head(vec![real]), None, "1 can't agree with itself"); + } + + #[tokio::test] + async fn confirms_the_newest_block_from_endpoints_that_agree() { + // The first endpoint is stuck far behind; the other 2 agree, so it is passed over + // rather than read from. + let stuck = server_at_block(1_000, STATE_REGISTERED).await; + let ended = STATE_REGISTERED | STATE_ACCEPTED | STATE_NOTICE_GIVEN; + let current = server_at_block(5_000_002, ended).await; + let also_current = server_at_block(5_000_002, ended).await; + let client = client_over(vec![ + stuck.uri().parse().expect("provider URL"), + current.uri().parse().expect("provider URL"), + also_current.uri().parse().expect("provider URL"), + ]); + + let live = client + .agreement_still_active(&[0xab; 16]) + .await + .expect("read"); + + assert!(!live, "read from an endpoint that is right"); + assert!(client.seen_block().confirmed); + assert_eq!(client.seen_block().number, 5_000_002); + } + + #[test] + fn alerts_only_when_every_attempt_is_refused_far_off() { + let far = Refusals::new(); + assert!(far.behind(100, 100 + ALERT_GAP_BLOCKS + 1).is_err()); + assert!(far.all_far_off()); + + let near = Refusals::new(); + assert!(near.behind(97, 100).is_err()); + assert!(!near.all_far_off(), "a few blocks of lag clears by itself"); + + // A primary that is down while a fallback lags is an outage, not a wrong block. + let outage = Refusals::new(); + assert!(outage.behind(100, 100 + ALERT_GAP_BLOCKS + 1).is_err()); + let failed: Result<(), TransportError> = Err(TransportErrorKind::custom_str("down")); + assert!(outage.other(failed).is_err()); + assert!(!outage.all_far_off()); + + assert!(!Refusals::new().all_far_off(), "none refused"); + } + async fn client_over_manager( responder: ManagerViewsResponder, ) -> (AlloyChainClient, MockServer) { diff --git a/bin/dipper-service/src/chain_client/rpc_provider.rs b/bin/dipper-service/src/chain_client/rpc_provider.rs index b2f280b7..836c6b98 100644 --- a/bin/dipper-service/src/chain_client/rpc_provider.rs +++ b/bin/dipper-service/src/chain_client/rpc_provider.rs @@ -10,7 +10,7 @@ use std::{ use thegraph_core::alloy::{ providers::{ - ProviderBuilder, RootProvider, + Provider, ProviderBuilder, RootProvider, fillers::{BlobGasFiller, ChainIdFiller, FillProvider, GasFiller, JoinFill, NonceFiller}, }, transports::{RpcError, TransportError, TransportErrorKind}, @@ -56,7 +56,6 @@ fn describe_failure(url: &Url, error: &TransportError) -> String { /// Error text that indicates a transient failure worth retrying, used only for faults /// that arrive as prose rather than as a status code or JSON-RPC error object. const RETRYABLE_ERROR_PATTERNS: &[&str] = &[ - BEHIND_A_SEEN_BLOCK, // A node behind the rest of its provider's fleet, asked for a block it hasn't reached. "header not found", "unknown block", @@ -73,9 +72,17 @@ const RETRYABLE_ERROR_PATTERNS: &[&str] = &[ ]; /// How a read refused by an endpoint behind a block dipper has already seen describes it. It -/// is retried, since an endpoint a block or so behind catches up within a second or two. +/// gets quick retries, since an endpoint a few blocks behind catches up within a second, then +/// the next endpoint, rather than the backoff for a failing one. pub(super) const BEHIND_A_SEEN_BLOCK: &str = "behind a block already seen"; +/// How long a read waits before asking an endpoint behind a block already seen again: 2 blocks. +const LAG_PAUSE: Duration = Duration::from_millis(500); + +/// How many times an endpoint behind a block already seen is asked again: about 4 blocks of +/// lag in all, so the read just after a transaction mines can wait out a node a little behind. +const LAG_RETRIES: u32 = 2; + /// Type alias for the provider with default fillers. pub type HttpProvider = FillProvider< JoinFill< @@ -154,6 +161,35 @@ impl RpcProviderPool { self.worst_case_walk } + /// How many endpoints the pool has. + pub fn endpoint_count(&self) -> usize { + self.providers.len() + } + + /// Each endpoint's latest block, all asked at once with no retries. Endpoints that fail are + /// left out, so dipper can see whether the ones that answer agree. + pub async fn latest_blocks(&self) -> Vec { + let mut asks = tokio::task::JoinSet::new(); + for url in &self.providers { + let provider = ProviderBuilder::new().connect_reqwest(self.http.clone(), url.clone()); + let endpoint = endpoint_name(url); + asks.spawn(async move { (endpoint, provider.get_block_number().await) }); + } + let mut heads = Vec::with_capacity(self.providers.len()); + while let Some(answer) = asks.join_next().await { + match answer { + Ok((_, Ok(head))) => heads.push(head), + Ok((endpoint, Err(err))) => tracing::debug!( + provider = %endpoint, + error = %err, + "RPC endpoint didn't give its latest block for a cross-check" + ), + Err(err) => tracing::warn!(error = %err, "Latest-block cross-check task failed"), + } + } + heads + } + /// Rotate to the next provider. /// /// Returns the new provider URL after rotation. @@ -207,41 +243,15 @@ impl RpcProviderPool { let current_url = self.url_at(start + providers_tried).clone(); let endpoint = endpoint_name(¤t_url); - // Reused across this endpoint's attempts, so a retry does not pay for a fresh - // TLS handshake on the path that is already running out of time. - let provider = - ProviderBuilder::new().connect_reqwest(self.http.clone(), current_url.clone()); - - // Retry loop for current provider - let mut endpoint_error: Option = None; - for attempt in 0..=max_retries { - match f(provider.clone()).await { - Ok(result) => return Ok(result), - Err(e) if Self::is_retryable(&e) && attempt < max_retries => { - let delay = Self::backoff_delay(attempt); - tracing::warn!( - operation, - provider = %endpoint, - attempt = attempt + 1, - max_retries, - delay_ms = delay.as_millis(), - error = %describe_failure(¤t_url, &e), - "Retryable RPC error, backing off" - ); - tokio::time::sleep(delay).await; - endpoint_error = Some(e); - } - Err(e) => { - endpoint_error = Some(e); - break; - } - } - } - // Every attempt records why it failed before stopping, so the fallback only - // covers a configuration that somehow allows no attempt at all. - let endpoint_error = endpoint_error - .unwrap_or_else(|| TransportErrorKind::custom_str("no attempt was made")); + let endpoint_error = match self + .ask_endpoint(operation, ¤t_url, max_retries, &f) + .await + { + Ok(result) => return Ok(result), + Err(err) => err, + }; + let lagging = Self::is_behind(&endpoint_error); let reason = describe_failure(¤t_url, &endpoint_error); reasons.push(format!("{endpoint}: {reason}")); providers_tried += 1; @@ -269,16 +279,89 @@ impl RpcProviderPool { // back round to a failing endpoint, costing them the one wasted first ask. self.rotate(); let next_url = self.url_at(start + providers_tried); - tracing::warn!( - operation, - old_provider = %endpoint, - new_provider = %endpoint_name(next_url), - providers_tried, - total_providers = self.providers.len(), - error = %reason, - "Rotating RPC provider after failures" - ); + if lagging { + tracing::debug!( + operation, + old_provider = %endpoint, + new_provider = %endpoint_name(next_url), + error = %reason, + "Rotating RPC provider past one behind a block already seen" + ); + } else { + tracing::warn!( + operation, + old_provider = %endpoint, + new_provider = %endpoint_name(next_url), + providers_tried, + total_providers = self.providers.len(), + error = %reason, + "Rotating RPC provider after failures" + ); + } + } + } + + /// Run a call against one endpoint, retrying a fault worth another go, and return what it + /// answered or why it last failed. + async fn ask_endpoint( + &self, + operation: &str, + url: &Url, + max_retries: u32, + f: &F, + ) -> Result + where + F: Fn(HttpProvider) -> Fut, + Fut: Future>, + { + let endpoint = endpoint_name(url); + // Reused across this endpoint's attempts, so a retry does not pay for a fresh + // TLS handshake on the path that is already running out of time. + let provider = ProviderBuilder::new().connect_reqwest(self.http.clone(), url.clone()); + let mut endpoint_error: Option = None; + let mut lag_retries = 0; + for attempt in 0..=max_retries { + match f(provider.clone()).await { + Ok(result) => return Ok(result), + Err(e) + if Self::is_behind(&e) + && lag_retries < LAG_RETRIES + && attempt < max_retries => + { + tracing::debug!( + operation, + provider = %endpoint, + error = %describe_failure(url, &e), + "RPC endpoint behind a block already seen, asking again" + ); + lag_retries += 1; + tokio::time::sleep(LAG_PAUSE).await; + endpoint_error = Some(e); + } + Err(e) if Self::is_retryable(&e) && attempt < max_retries => { + let delay = Self::backoff_delay(attempt); + tracing::warn!( + operation, + provider = %endpoint, + attempt = attempt + 1, + max_retries, + delay_ms = delay.as_millis(), + error = %describe_failure(url, &e), + "Retryable RPC error, backing off" + ); + tokio::time::sleep(delay).await; + endpoint_error = Some(e); + } + Err(e) => return Err(e), + } } + // Every attempt records why it failed before stopping, so the fallback only + // covers a configuration that somehow allows no attempt at all. + Err(endpoint_error.unwrap_or_else(|| TransportErrorKind::custom_str("no attempt was made"))) + } + + fn is_behind(error: &TransportError) -> bool { + error.to_string().contains(BEHIND_A_SEEN_BLOCK) } /// Whether an error is worth trying again rather than giving up on. Each check can only @@ -641,6 +724,42 @@ mod tests { ); } + /// Hosted endpoints spread calls across nodes, so one a block behind is routine: it is + /// asked again after short pauses, then passed over, never backed off from. + #[tokio::test] + async fn an_endpoint_behind_a_block_already_seen_gets_quick_retries() { + let lagging = server_answering_block(1).await; + let current = server_answering_block(2).await; + let pool = RpcProviderPool::new( + vec![ + lagging.uri().parse().expect("lagging URL"), + current.uri().parse().expect("current URL"), + ], + Duration::from_secs(5), + 3, + ) + .expect("pool"); + let started = std::time::Instant::now(); + + let block = pool + .execute("get_block_number", |provider| async move { + let block = provider.get_block_number().await?; + if block < 2 { + return Err(TransportErrorKind::custom_str(BEHIND_A_SEEN_BLOCK)); + } + Ok(block) + }) + .await + .expect("read from the endpoint that has the block"); + + assert_eq!(block, 2); + assert_eq!( + lagging.received_requests().await.unwrap_or_default().len(), + 3 + ); + assert!(started.elapsed() < Duration::from_secs(2), "no backoff"); + } + #[test] fn a_node_that_has_not_reached_a_block_yet_is_retryable() { let payload = serde_json::from_str(r#"{"code":-32000,"message":"header not found"}"#) From 20d599f8c9d0d3e72741d4c734626ab42614ce80 Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Mon, 5 Oct 2026 12:10:51 +0100 Subject: [PATCH 14/26] fix: credit each cancel to whoever ended it (#727) * perf(chain): read each cancelling agreement once per check Checking an agreement dipper is cancelling read its on-chain state, then read it again to learn whether the indexer had ended it. One read now answers both, halving the calls for every agreement that has already ended. * fix(cancel): credit an end to the indexer when it beat dipper's cancel If the indexer cancelled just before dipper's cancel mined, dipper's did nothing but was still recorded as the end, naming dipper and its transaction. The read after a cancel now also says who ended it, and an end by the indexer is left to be recorded as theirs. * fix(cancel): keep a mined cancel's hash when it can't be read back When the read after a mined cancel failed, the cancel was reported as failed and its transaction hash dropped. It is now reported as unconfirmed with its hash in the log, is not counted as a failed attempt, and is read again on the next check. * fix(cancel): record dipper's cancel before marking the agreement ended An end is announced once the agreement is marked ended, and the cancel's transaction was saved just after, so the announcement could go out without it and never be resent. The transaction is now saved first. * fix(cancel): count a cancel that never mines as a failed attempt A cancel an endpoint accepted but that never mined wasn't counted, so one dropped every time was resent every few minutes for ever and the stuck-cancel alert never fired. It now counts like a cancel the contract refused. * perf(cancel): skip the chain when nothing is being cancelled The cancel retry, which runs every few minutes, read the chain's latest block before checking whether any agreement was being cancelled. It now checks first, so a quiet dipper makes no call. * fix(cancel): describe dropped and unconfirmed cancels accurately in logs A cancel that never mined was logged as one that didn't end the agreement, and withdrawing an offer whose cancel mined but couldn't be read back logged no transaction. Both now say what happened, the second with its transaction hash. * test(chain): read through the single agreement read in a merged test A test brought in from the branch below still called the 2 separate chain reads this branch replaced with 1, so it no longer compiled here. * fix(cancel): count a dropped cancel only when its receipt was checked A cancel counted as dropped when no receipt appeared in 15 seconds, even if every receipt check failed, so an RPC outage could use up attempts and raise the stuck alert. It now counts only when an endpoint answered that it had no receipt. --- bin/dipper-service/src/cancel_dispatch.rs | 207 +++++++++----- bin/dipper-service/src/chain_client.rs | 58 ++-- bin/dipper-service/src/chain_client/client.rs | 156 ++++++++--- .../src/network/service/cancel_retry.rs | 260 ++++++++++++------ .../src/network/service/chain_listener.rs | 19 +- .../src/network/service/escrow_reconciler.rs | 11 +- .../src/network/service/liveness_checker.rs | 33 ++- .../cancel_rejected_agreement_on_chain.rs | 14 +- .../handlers/reassess_indexing_request.rs | 30 +- .../src/worker/handlers/submit_offer.rs | 23 +- 10 files changed, 510 insertions(+), 301 deletions(-) diff --git a/bin/dipper-service/src/cancel_dispatch.rs b/bin/dipper-service/src/cancel_dispatch.rs index b35e8a20..049dc105 100644 --- a/bin/dipper-service/src/cancel_dispatch.rs +++ b/bin/dipper-service/src/cancel_dispatch.rs @@ -5,7 +5,7 @@ use dipper_core::time::now_secs; use thegraph_core::alloy::primitives::B256; use crate::{ - chain_client::{ChainClient, ChainClientError}, + chain_client::{AgreementOnChain, ChainClient, ChainClientError}, config::IndexingAgreementConfig, registry::{ AgreementRegistry, IndexingAgreement, IndexingAgreementStatus, Result as RegistryResult, @@ -27,39 +27,45 @@ pub async fn cancel_agreement_on_chain( chain_client: &T, agreement: &IndexingAgreement, config: &IndexingAgreementConfig, -) -> Result, ChainClientError> { - let version_hash = agreement +) -> LiveCancel { + let Some(version_hash) = agreement .terms_version_hash .as_deref() .filter(|h| h.len() == 32) .map(B256::from_slice) - .ok_or_else(|| ChainClientError::MissingTermsVersionHash { + else { + return LiveCancel::CancelFailed(ChainClientError::MissingTermsVersionHash { agreement_id: agreement.id.to_string(), - })?; - // Hazard: the manager's cancel mines successfully even when it does nothing - // (stale/wrong hash, unknown id, already-terminal). So after a submitted - // cancel we re-read on-chain and surface CancelNotConfirmed if still active. - let outcome = chain_client + }); + }; + let tx_hash = match chain_client .cancel_via_manager( config.recurring_collector(), agreement.id.as_bytes(), version_hash, SCOPE_BOTH, ) - .await?; - - // cancel_via_manager only returns Ok(Some) (its tx always submits); - // Ok(None) is reserved. Verify only when a cancel actually mined. - if outcome.is_some() - && chain_client - .agreement_still_active(agreement.id.as_bytes()) - .await? + .await { - return Err(ChainClientError::CancelNotConfirmed { - agreement_id: agreement.id.to_string(), - }); + Ok(tx_hash) => tx_hash, + Err(err) => return LiveCancel::CancelFailed(err), + }; + // The manager's cancel mines even when it does nothing (a stale hash, an unknown id, an + // agreement already ended), and the indexer may have ended it first, so only a read + // afterwards says whether this cancel is what ended it. + match chain_client + .agreement_on_chain(agreement.id.as_bytes()) + .await + { + Ok(AgreementOnChain::NotLive) => LiveCancel::Ended(tx_hash), + Ok(AgreementOnChain::EndedByIndexer) => LiveCancel::NotLive { by_indexer: true }, + Ok(AgreementOnChain::Live) => { + LiveCancel::CancelFailed(ChainClientError::CancelNotConfirmed { + agreement_id: agreement.id.to_string(), + }) + } + Err(err) => LiveCancel::Unconfirmed { tx_hash, err }, } - Ok(outcome) } /// What [`start_cancel`] left an agreement as. @@ -89,7 +95,7 @@ where .await?; let tx_hash = match cancel_if_live(chain_client, agreement, config).await { LiveCancel::Ended(tx_hash) => tx_hash, - LiveCancel::NotLive => return Ok(CancelStarted::Cancelling), + LiveCancel::NotLive { .. } => return Ok(CancelStarted::Cancelling), LiveCancel::ReadFailed(err) | LiveCancel::CancelFailed(err) => { tracing::warn!( agreement_id = %agreement.id, @@ -98,6 +104,10 @@ where ); return Ok(CancelStarted::Cancelling); } + LiveCancel::Unconfirmed { tx_hash, err } => { + log_unconfirmed(agreement, tx_hash, &err); + return Ok(CancelStarted::Cancelling); + } }; tracing::info!( agreement_id = %agreement.id, @@ -126,6 +136,10 @@ pub async fn confirm_cancelled( tx_hash: Option, config: &IndexingAgreementConfig, ) -> bool { + // First, so the sweep, which announces the end once the mark lands, finds the transaction. + if tx_hash.is_some() { + record_cancel(registry, agreement, tx_hash, config).await; + } if let Err(err) = registry .mark_indexing_agreement_as_canceled_by_requester(&agreement.id) .await @@ -145,9 +159,6 @@ pub async fn confirm_cancelled( reason = "cancel_confirmed_on_chain", "agreement state transition" ); - if tx_hash.is_some() { - record_cancel(registry, agreement, tx_hash, config).await; - } true } @@ -187,11 +198,11 @@ where T: ChainClient, { match chain_client - .agreement_still_active(agreement.id.as_bytes()) + .agreement_on_chain(agreement.id.as_bytes()) .await { - Ok(false) => return Ok(false), - Ok(true) => {} + Ok(AgreementOnChain::Live) => {} + Ok(_) => return Ok(false), Err(err) => tracing::warn!( agreement_id = %agreement.id, error = %err, @@ -218,15 +229,37 @@ where Ok(true) } +/// Log a cancel that mined but could not be read back, naming its transaction so it isn't lost; +/// the cancel retry reads the agreement again, and the chain listener records the end. +pub fn log_unconfirmed( + agreement: &IndexingAgreement, + tx_hash: Option, + err: &ChainClientError, +) { + tracing::warn!( + agreement_id = %agreement.id, + tx_hash = ?tx_hash, + error = %err, + "On-chain cancel mined, but whether it ended the agreement couldn't be read; will check again" + ); +} + /// What [`cancel_if_live`] found and did. #[derive(Debug)] pub enum LiveCancel { - /// The chain showed nothing live, so no cancel was sent. - NotLive, + /// The chain showed nothing live, so no cancel was sent or the one sent did nothing; + /// `by_indexer` when the indexer ended it. + NotLive { by_indexer: bool }, /// A cancel went out and the chain confirmed the agreement ended. Ended(Option), /// The chain could not be read, so nothing was sent. ReadFailed(ChainClientError), + /// A cancel mined, but the read after it failed, so whether it ended the agreement, and + /// who did, is not known yet. + Unconfirmed { + tx_hash: Option, + err: ChainClientError, + }, /// The cancel failed or did not end the agreement. CancelFailed(ChainClientError), } @@ -240,14 +273,15 @@ pub async fn cancel_if_live( config: &IndexingAgreementConfig, ) -> LiveCancel { match chain_client - .agreement_still_active(agreement.id.as_bytes()) + .agreement_on_chain(agreement.id.as_bytes()) .await { Err(err) => LiveCancel::ReadFailed(err), - Ok(false) => LiveCancel::NotLive, - Ok(true) => match cancel_agreement_on_chain(chain_client, agreement, config).await { - Ok(tx_hash) => LiveCancel::Ended(tx_hash), - Err(err) => LiveCancel::CancelFailed(err), + Ok(AgreementOnChain::Live) => { + cancel_agreement_on_chain(chain_client, agreement, config).await + } + Ok(ended) => LiveCancel::NotLive { + by_indexer: ended == AgreementOnChain::EndedByIndexer, }, } } @@ -266,9 +300,9 @@ pub(crate) mod tests { use time::OffsetDateTime; use url::Url; - use super::{SCOPE_BOTH, cancel_agreement_on_chain}; + use super::{LiveCancel, SCOPE_BOTH, cancel_agreement_on_chain}; use crate::{ - chain_client::{ChainClient, ChainClientError}, + chain_client::{AgreementOnChain, ChainClient, ChainClientError}, config::IndexingAgreementConfig, registry::{ IndexingAgreement, IndexingAgreementStatus, IndexingAgreementTerms, @@ -286,6 +320,8 @@ pub(crate) mod tests { struct RecordingChainClient { manager_cancels: Mutex>, still_active_after_cancel: bool, + ended_by_indexer: bool, + read_back_fails: bool, active_reads: Mutex, } @@ -336,18 +372,18 @@ pub(crate) mod tests { unimplemented!() } - async fn agreement_still_active( + async fn agreement_on_chain( &self, _agreement_id: &[u8; 16], - ) -> Result { + ) -> Result { *self.active_reads.lock().unwrap() += 1; - Ok(self.still_active_after_cancel) - } - async fn agreement_ended_by_indexer( - &self, - _agreement_id: &[u8; 16], - ) -> Result { - Ok(false) + if self.read_back_fails { + return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); + } + if self.ended_by_indexer { + return Ok(AgreementOnChain::EndedByIndexer); + } + Ok(AgreementOnChain::live_if(self.still_active_after_cancel)) } } @@ -416,9 +452,7 @@ pub(crate) mod tests { Some(vec![7u8; 32]), ); - cancel_agreement_on_chain(&client, &ag, &manager_conf(collector)) - .await - .expect("cancel dispatch"); + cancel_agreement_on_chain(&client, &ag, &manager_conf(collector)).await; let calls = client.manager_cancels.lock().unwrap(); assert_eq!(calls.len(), 1); @@ -438,9 +472,7 @@ pub(crate) mod tests { let client = RecordingChainClient::default(); let ag = agreement(IndexingAgreementStatus::Rejected, Some(vec![9u8; 32])); - cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)) - .await - .expect("cancel dispatch"); + cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)).await; let calls = client.manager_cancels.lock().unwrap(); assert_eq!(calls.len(), 1); @@ -455,13 +487,11 @@ pub(crate) mod tests { let client = RecordingChainClient::default(); let ag = agreement(IndexingAgreementStatus::AcceptedOnChain, None); - let err = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)) - .await - .unwrap_err(); + let out = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)).await; assert!(matches!( - err, - ChainClientError::MissingTermsVersionHash { .. } + out, + LiveCancel::CancelFailed(ChainClientError::MissingTermsVersionHash { .. }) )); assert!(client.manager_cancels.lock().unwrap().is_empty()); } @@ -475,13 +505,11 @@ pub(crate) mod tests { Some(vec![1u8; 16]), ); - let err = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)) - .await - .unwrap_err(); + let out = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)).await; assert!(matches!( - err, - ChainClientError::MissingTermsVersionHash { .. } + out, + LiveCancel::CancelFailed(ChainClientError::MissingTermsVersionHash { .. }) )); assert!(client.manager_cancels.lock().unwrap().is_empty()); } @@ -500,11 +528,12 @@ pub(crate) mod tests { Some(vec![7u8; 32]), ); - let err = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)) - .await - .unwrap_err(); + let out = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)).await; - assert!(matches!(err, ChainClientError::CancelNotConfirmed { .. })); + assert!(matches!( + out, + LiveCancel::CancelFailed(ChainClientError::CancelNotConfirmed { .. }) + )); assert_eq!(client.manager_cancels.lock().unwrap().len(), 1); assert_eq!(*client.active_reads.lock().unwrap(), 1, "verified once"); } @@ -522,12 +551,50 @@ pub(crate) mod tests { Some(vec![7u8; 32]), ); - let out = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)) - .await - .expect("cancel confirmed"); + let out = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)).await; - assert!(out.is_some()); + assert!(matches!(out, LiveCancel::Ended(Some(_)))); assert_eq!(client.manager_cancels.lock().unwrap().len(), 1); assert_eq!(*client.active_reads.lock().unwrap(), 1, "verified once"); } + + #[tokio::test] + async fn a_cancel_the_indexer_beat_to_it_is_not_dipper_s() { + // The indexer's cancel landed first, so dipper's mined as a no-op: crediting the end + // to dipper would announce the wrong canceller. + let client = RecordingChainClient { + ended_by_indexer: true, + ..Default::default() + }; + let ag = agreement( + IndexingAgreementStatus::AcceptedOnChain, + Some(vec![7u8; 32]), + ); + + let out = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)).await; + + assert!(matches!(out, LiveCancel::NotLive { by_indexer: true })); + } + + #[tokio::test] + async fn a_mined_cancel_that_cannot_be_read_back_keeps_its_transaction() { + let client = RecordingChainClient { + read_back_fails: true, + ..Default::default() + }; + let ag = agreement( + IndexingAgreementStatus::AcceptedOnChain, + Some(vec![7u8; 32]), + ); + + let out = cancel_agreement_on_chain(&client, &ag, &manager_conf(Address::ZERO)).await; + + assert!(matches!( + out, + LiveCancel::Unconfirmed { + tx_hash: Some(B256::ZERO), + .. + } + )); + } } diff --git a/bin/dipper-service/src/chain_client.rs b/bin/dipper-service/src/chain_client.rs index f9f1583f..6bc5e3ee 100644 --- a/bin/dipper-service/src/chain_client.rs +++ b/bin/dipper-service/src/chain_client.rs @@ -65,7 +65,12 @@ pub enum ChainClientError { /// tx claimed the nonce with a higher fee. Callers re-sync the nonce and /// resubmit; there is no idempotency guard, so a replay re-sends the call. #[error("tx {tx_hash} did not mine within the receipt-poll window")] - TxDropped { tx_hash: B256 }, + TxDropped { + tx_hash: B256, + /// Whether any receipt check got an answer. When none did, an outage may have hidden a + /// transaction that mined, rather than the chain not mining it. + receipt_checked: bool, + }, /// Tx was mined but reverted on-chain (receipt status = 0). #[error("tx {tx_hash} reverted on-chain")] @@ -80,6 +85,28 @@ pub enum ChainClientError { ContractRevert { selector: [u8; 4], data: Bytes }, } +/// What the chain shows of an agreement's current terms. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum AgreementOnChain { + /// Accepted with no cancellation notice, or an offer still waiting to be accepted. + Live, + /// The indexer ended it. + EndedByIndexer, + /// Neither: never offered, withdrawn, past its offer deadline, or ended by dipper. + NotLive, +} + +impl AgreementOnChain { + pub fn is_live(self) -> bool { + self == Self::Live + } + + #[cfg(test)] + pub fn live_if(live: bool) -> Self { + if live { Self::Live } else { Self::NotLive } + } +} + /// Trait for sending on-chain transactions related to indexing agreements #[async_trait] pub trait ChainClient { @@ -120,20 +147,12 @@ pub trait ChainClient { agreement_id: &[u8; 16], ) -> Result, ChainClientError>; - /// Read whether the agreement is still live on-chain (terms accepted and no - /// cancellation notice given, or an offer still waiting to be accepted) via - /// the RecurringCollector's `getAgreementDetails(id, VERSION_CURRENT)`. - async fn agreement_still_active( + /// Read what the chain shows of the agreement, via the RecurringCollector's + /// `getAgreementDetails(id, VERSION_CURRENT)`. + async fn agreement_on_chain( &self, agreement_id: &[u8; 16], - ) -> Result; - - /// Read whether the indexer ended the agreement on-chain, which `getAgreementDetails` - /// reports with its BY_PROVIDER flag. False for one still live or ended by dipper. - async fn agreement_ended_by_indexer( - &self, - agreement_id: &[u8; 16], - ) -> Result; + ) -> Result; /// Read the latest block's unix timestamp from the chain. Lets agreement /// deadlines be stamped from live chain time when the chain-clock bypass is @@ -182,18 +201,11 @@ impl ChainClient for Arc { (**self).reconcile_agreement(collector, agreement_id).await } - async fn agreement_still_active( - &self, - agreement_id: &[u8; 16], - ) -> Result { - (**self).agreement_still_active(agreement_id).await - } - - async fn agreement_ended_by_indexer( + async fn agreement_on_chain( &self, agreement_id: &[u8; 16], - ) -> Result { - (**self).agreement_ended_by_indexer(agreement_id).await + ) -> Result { + (**self).agreement_on_chain(agreement_id).await } async fn latest_block_timestamp(&self) -> Result { diff --git a/bin/dipper-service/src/chain_client/client.rs b/bin/dipper-service/src/chain_client/client.rs index 9fe31298..7ca8a9c5 100644 --- a/bin/dipper-service/src/chain_client/client.rs +++ b/bin/dipper-service/src/chain_client/client.rs @@ -32,7 +32,8 @@ use super::{ }; use crate::{ chain_client::{ - ChainClient, ChainClientError, EscrowAccount, ManagerEscrowReader, TrackedProviders, + AgreementOnChain, ChainClient, ChainClientError, EscrowAccount, ManagerEscrowReader, + TrackedProviders, }, config::ChainClientConfig, worker::service::PROCESS_JOB_TIMEOUT, @@ -53,6 +54,15 @@ const RECEIPT_POLL_TIMEOUT: Duration = Duration::from_secs(15); /// loose enough to avoid hammering the RPC. const RECEIPT_POLL_INTERVAL: Duration = Duration::from_millis(500); +/// What waiting for a transaction's receipt found. +enum ReceiptWait { + /// It mined; whether it succeeded. + Mined(bool), + /// No receipt appeared in time; `checked` when an endpoint did answer that it had none, + /// rather than every check failing. + NotSeen { checked: bool }, +} + /// VERSION_CURRENT index from `IAgreementCollector.sol`: the active (or /// pre-acceptance) terms. `getAgreementDetails(id, 0)` reports their state. const VERSION_CURRENT: u64 = 0; @@ -269,6 +279,16 @@ fn still_live(state: u16) -> bool { pending_offer || (accepted && state & STATE_NOTICE_GIVEN == 0) } +fn on_chain(state: u16) -> AgreementOnChain { + if still_live(state) { + AgreementOnChain::Live + } else if state & STATE_BY_PROVIDER != 0 { + AgreementOnChain::EndedByIndexer + } else { + AgreementOnChain::NotLive + } +} + /// Error patterns that indicate a nonce-related issue. /// /// These errors can be resolved by refreshing the nonce and retrying. @@ -784,9 +804,9 @@ impl AlloyChainClient { .await?; match self.wait_for_receipt(tx_hash, RECEIPT_POLL_TIMEOUT).await? { - Some(true) => Ok(Some(tx_hash)), - Some(false) => Err(ChainClientError::TxReverted { tx_hash }), - None => { + ReceiptWait::Mined(true) => Ok(Some(tx_hash)), + ReceiptWait::Mined(false) => Err(ChainClientError::TxReverted { tx_hash }), + ReceiptWait::NotSeen { checked } => { tracing::warn!( reconciling = subject, tx_hash = %tx_hash, @@ -796,7 +816,10 @@ impl AlloyChainClient { if let Err(err) = self.fill_nonce_gap(dropped_nonce).await { tracing::warn!(nonce = dropped_nonce, error = %err, "Failed to fill mempool nonce gap"); } - Err(ChainClientError::TxDropped { tx_hash }) + Err(ChainClientError::TxDropped { + tx_hash, + receipt_checked: checked, + }) } } } @@ -919,15 +942,16 @@ impl AlloyChainClient { }) } - /// Poll `eth_getTransactionReceipt` until the tx has mined or the timeout elapses. - /// `Ok(Some(status))` reports the receipt's success flag; `Ok(None)` says the tx never - /// appeared in time (dropped from the mempool). Transient RPC errors keep polling. + /// Poll `eth_getTransactionReceipt` until the tx has mined or the timeout elapses, saying + /// whether it mined and succeeded, or never appeared in time (dropped from the mempool), and + /// then whether any check got an answer. Transient RPC errors keep polling. async fn wait_for_receipt( &self, tx_hash: B256, timeout: Duration, - ) -> Result, ChainClientError> { + ) -> Result { let deadline = tokio::time::Instant::now() + timeout; + let mut checked = false; loop { let receipt = self .inner @@ -942,9 +966,9 @@ impl AlloyChainClient { if let Some(block) = r.block_number { self.seen_block().confirm(block, Instant::now()); } - return Ok(Some(r.status())); + return Ok(ReceiptWait::Mined(r.status())); } - Ok(None) => {} // not mined yet + Ok(None) => checked = true, // not mined yet Err(e) => { // Transient RPC error: log and keep polling. If it persists, the outer // handler sees the timeout as `Ok(None)` and resubmits, the safe default. @@ -957,7 +981,7 @@ impl AlloyChainClient { } if tokio::time::Instant::now() >= deadline { - return Ok(None); + return Ok(ReceiptWait::NotSeen { checked }); } tokio::time::sleep(RECEIPT_POLL_INTERVAL).await; } @@ -1028,9 +1052,9 @@ impl ChainClient for AlloyChainClient { nonce: dropped_nonce, } = submitted; match self.wait_for_receipt(tx_hash, RECEIPT_POLL_TIMEOUT).await? { - Some(true) => Ok(Some(tx_hash)), - Some(false) => Err(ChainClientError::TxReverted { tx_hash }), - None => { + ReceiptWait::Mined(true) => Ok(Some(tx_hash)), + ReceiptWait::Mined(false) => Err(ChainClientError::TxReverted { tx_hash }), + ReceiptWait::NotSeen { checked } => { tracing::warn!( agreement_id = %format_args!("0x{}", agreement_id.iter().map(|b| format!("{b:02x}")).collect::()), tx_hash = %tx_hash, @@ -1040,7 +1064,10 @@ impl ChainClient for AlloyChainClient { if let Err(err) = self.fill_nonce_gap(dropped_nonce).await { tracing::warn!(nonce = dropped_nonce, error = %err, "Failed to fill mempool nonce gap"); } - Err(ChainClientError::TxDropped { tx_hash }) + Err(ChainClientError::TxDropped { + tx_hash, + receipt_checked: checked, + }) } } } @@ -1081,9 +1108,9 @@ impl ChainClient for AlloyChainClient { nonce: dropped_nonce, } = submitted; match self.wait_for_receipt(tx_hash, RECEIPT_POLL_TIMEOUT).await? { - Some(true) => Ok(Some(tx_hash)), - Some(false) => Err(ChainClientError::TxReverted { tx_hash }), - None => { + ReceiptWait::Mined(true) => Ok(Some(tx_hash)), + ReceiptWait::Mined(false) => Err(ChainClientError::TxReverted { tx_hash }), + ReceiptWait::NotSeen { checked } => { tracing::warn!( agreement_id = %format_args!("0x{}", agreement_id.iter().map(|b| format!("{b:02x}")).collect::()), tx_hash = %tx_hash, @@ -1093,23 +1120,19 @@ impl ChainClient for AlloyChainClient { if let Err(err) = self.fill_nonce_gap(dropped_nonce).await { tracing::warn!(nonce = dropped_nonce, error = %err, "Failed to fill mempool nonce gap"); } - Err(ChainClientError::TxDropped { tx_hash }) + Err(ChainClientError::TxDropped { + tx_hash, + receipt_checked: checked, + }) } } } - async fn agreement_still_active( + async fn agreement_on_chain( &self, agreement_id: &[u8; 16], - ) -> Result { - Ok(still_live(self.agreement_state(agreement_id).await?)) - } - - async fn agreement_ended_by_indexer( - &self, - agreement_id: &[u8; 16], - ) -> Result { - Ok(self.agreement_state(agreement_id).await? & STATE_BY_PROVIDER != 0) + ) -> Result { + Ok(on_chain(self.agreement_state(agreement_id).await?)) } async fn reconcile_provider( @@ -1335,6 +1358,22 @@ mod tests { assert!(!still_live(0), "revoked or never offered"); } + #[test] + fn tells_an_end_by_the_indexer_from_any_other() { + const BY_PAYER: u16 = 16; + let ended = STATE_REGISTERED | STATE_ACCEPTED | STATE_NOTICE_GIVEN; + assert_eq!( + on_chain(ended | STATE_BY_PROVIDER), + AgreementOnChain::EndedByIndexer + ); + assert_eq!(on_chain(ended | BY_PAYER), AgreementOnChain::NotLive); + assert_eq!(on_chain(0), AgreementOnChain::NotLive); + assert_eq!( + on_chain(STATE_REGISTERED | STATE_ACCEPTED), + AgreementOnChain::Live + ); + } + /// Answers a send with a fixed transaction hash, echoing the request id so alloy's /// transport accepts the response. Any other call is a mistake in the test rather than /// something to answer with a hash, so say so instead of returning nonsense. @@ -2403,9 +2442,10 @@ mod tests { client.note_block(95); let live = client - .agreement_still_active(&[0xab; 16]) + .agreement_on_chain(&[0xab; 16]) .await - .expect("read once the endpoint caught up"); + .expect("read once the endpoint caught up") + .is_live(); assert!(live); } @@ -2424,9 +2464,10 @@ mod tests { client.note_block(95); let live = client - .agreement_still_active(&[0xab; 16]) + .agreement_on_chain(&[0xab; 16]) .await - .expect("read"); + .expect("read") + .is_live(); assert!(!live, "read from the endpoint that has reached block 95"); assert_eq!(client.seen_block().number, 100); @@ -2438,7 +2479,7 @@ mod tests { let client = client_over(vec![lagging.uri().parse().expect("provider URL")]); client.note_block(95); - let read = client.agreement_still_active(&[0xab; 16]).await; + let read = client.agreement_on_chain(&[0xab; 16]).await; assert!(read.is_err(), "got {read:?}"); } @@ -2456,9 +2497,10 @@ mod tests { trust_block(&client, 100); let live = client - .agreement_still_active(&[0xab; 16]) + .agreement_on_chain(&[0xab; 16]) .await - .expect("read"); + .expect("read") + .is_live(); assert!(!live, "read from the endpoint at block 150"); assert_eq!(client.seen_block().number, 150); @@ -2476,9 +2518,10 @@ mod tests { let client = client_over(vec![endpoint.uri().parse().expect("provider URL")]); client - .agreement_still_active(&[0xab; 16]) + .agreement_on_chain(&[0xab; 16]) .await - .expect("read"); + .expect("read") + .is_live(); assert_eq!(client.seen_block().number, WEEK_OF_BLOCKS * 3); } @@ -2510,6 +2553,36 @@ mod tests { assert_eq!(SeenBlock::new().bounds(now), (0, u64::MAX)); } + #[tokio::test] + async fn says_whether_any_receipt_check_was_answered() { + // A receipt that was never seen only shows the chain didn't mine it when an endpoint + // answered; when every check failed, an outage may be hiding one that did. + let empty = MockServer::start().await; + Mock::given(method("POST")) + .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({ + "jsonrpc": "2.0", + "id": 0, + "result": null, + }))) + .mount(&empty) + .await; + let down = server_answering_500().await; + let wait = Duration::from_millis(100); + + for (server, answered) in [(empty, true), (down, false)] { + let client = client_over(vec![server.uri().parse().expect("provider URL")]); + let found = client + .wait_for_receipt(B256::repeat_byte(0x01), wait) + .await + .expect("waited"); + + assert!( + matches!(found, ReceiptWait::NotSeen { checked } if checked == answered), + "answered: {answered}" + ); + } + } + #[test] fn never_goes_back_even_before_it_is_confirmed() { // So a read just after dipper's cancel mined can't come from an endpoint behind it. @@ -2576,9 +2649,10 @@ mod tests { ]); let live = client - .agreement_still_active(&[0xab; 16]) + .agreement_on_chain(&[0xab; 16]) .await - .expect("read"); + .expect("read") + .is_live(); assert!(!live, "read from an endpoint that is right"); assert!(client.seen_block().confirmed); diff --git a/bin/dipper-service/src/network/service/cancel_retry.rs b/bin/dipper-service/src/network/service/cancel_retry.rs index 49a651a6..4f6dfc1f 100644 --- a/bin/dipper-service/src/network/service/cancel_retry.rs +++ b/bin/dipper-service/src/network/service/cancel_retry.rs @@ -6,7 +6,7 @@ use dipper_core::time::now_secs; use thegraph_core::alloy::primitives::B256; use crate::{ - cancel_dispatch::{LiveCancel, cancel_if_live, confirm_cancelled}, + cancel_dispatch::{LiveCancel, cancel_if_live, confirm_cancelled, log_unconfirmed}, chain_client::{ChainClient, ChainClientError}, config::IndexingAgreementConfig, registry::{AgreementRegistry, CancelKind, CancellingAgreement, IndexingAgreement}, @@ -43,9 +43,6 @@ pub async fn retry_cancelling_agreements( R: AgreementRegistry + Sync, T: ChainClient, { - let Some(chain_now) = chain_time(chain_client).await else { - return; - }; let cancelling = match registry .get_cancelling_agreements(BATCH_SIZE, MAX_CANCEL_ATTEMPTS, SETTLE_MINUTES) .await @@ -56,6 +53,12 @@ pub async fn retry_cancelling_agreements( return; } }; + if cancelling.is_empty() { + return; + } + let Some(chain_now) = chain_time(chain_client).await else { + return; + }; let started = std::time::Instant::now(); for (done, row) in cancelling.iter().enumerate() { if started.elapsed() >= SWEEP_BUDGET { @@ -93,29 +96,34 @@ async fn retry_cancel( T: ChainClient, { let agreement_id = row.agreement.id; - let (tx_hash, failure) = match cancel_if_live(chain_client, &row.agreement, config).await { - LiveCancel::ReadFailed(err) => { - tracing::warn!( - %agreement_id, - error = %err, - "Failed to read a cancelling agreement on-chain, will retry" - ); - // Unread, it may still be live, so it can't be confirmed ended. - return note_check(registry, row, None).await; - } - LiveCancel::NotLive => (None, None), - LiveCancel::Ended(tx_hash) => { - tracing::info!( - %agreement_id, - tx_hash = ?tx_hash, - "Cancelled an agreement still live on-chain" - ); - (tx_hash, None) - } - LiveCancel::CancelFailed(err) => (None, Some(err)), - }; + let (tx_hash, by_indexer, failure) = + match cancel_if_live(chain_client, &row.agreement, config).await { + LiveCancel::ReadFailed(err) => { + tracing::warn!( + %agreement_id, + error = %err, + "Failed to read a cancelling agreement on-chain, will retry" + ); + // Unread, it may still be live, so it can't be confirmed ended. + return note_check(registry, row, None).await; + } + LiveCancel::NotLive { by_indexer } => (None, by_indexer, None), + LiveCancel::Ended(tx_hash) => { + tracing::info!( + %agreement_id, + tx_hash = ?tx_hash, + "Cancelled an agreement still live on-chain" + ); + (tx_hash, false, None) + } + LiveCancel::CancelFailed(err) => (None, false, Some(err)), + LiveCancel::Unconfirmed { tx_hash, err } => { + log_unconfirmed(&row.agreement, tx_hash, &err); + return note_check(registry, row, None).await; + } + }; if failure.is_none() - && confirm_if_over(registry, chain_client, config, row, tx_hash, chain_now).await + && confirm_if_over(registry, config, row, tx_hash, by_indexer, chain_now).await { return; } @@ -125,18 +133,14 @@ async fn retry_cancel( /// Mark the agreement `CanceledByRequester` once it can't go live again: this sweep's cancel /// ended it, or nobody accepted its offer before the deadline to. One ended otherwise is left /// to the chain listener for a while; one the indexer ended then becomes `CanceledByIndexer`. -async fn confirm_if_over( +async fn confirm_if_over( registry: &R, - chain_client: &T, config: &IndexingAgreementConfig, row: &CancellingAgreement, tx_hash: Option, + by_indexer: bool, chain_now: u64, -) -> bool -where - R: AgreementRegistry + Sync, - T: ChainClient, -{ +) -> bool { let agreement = &row.agreement; let past_grace = agreement.updated_at < time::OffsetDateTime::now_utc() - LISTENER_GRACE; let can_confirm = if row.accepted_on_chain { @@ -147,46 +151,16 @@ where if !can_confirm { return false; } - if tx_hash.is_none() { - match ended_by_indexer(chain_client, agreement).await { - None => return false, - Some(true) => return past_grace && record_end_by_indexer(registry, agreement).await, - Some(false) => {} - } + if by_indexer { + tracing::info!( + agreement_id = %agreement.id, + "The indexer ended an agreement dipper was cancelling" + ); + return past_grace && record_end_by_indexer(registry, agreement).await; } confirm_cancelled(registry, agreement, tx_hash, config).await } -/// Whether the chain shows the indexer ended the agreement, or `None` when it can't be read, -/// so an end is never wrongly put down to dipper. -async fn ended_by_indexer( - chain_client: &T, - agreement: &IndexingAgreement, -) -> Option { - match chain_client - .agreement_ended_by_indexer(agreement.id.as_bytes()) - .await - { - Ok(by_indexer) => { - if by_indexer { - tracing::info!( - agreement_id = %agreement.id, - "The indexer ended an agreement dipper was cancelling" - ); - } - Some(by_indexer) - } - Err(err) => { - tracing::warn!( - agreement_id = %agreement.id, - error = %err, - "Failed to read who ended a cancelling agreement, will retry" - ); - None - } - } -} - /// Mark an agreement the indexer ended `CanceledByIndexer` when the chain listener hasn't in /// time, so it doesn't stay cancelling for good. The indexer is recorded as ending it first, /// so its announcement names them; the time recorded is when dipper noticed. @@ -284,7 +258,7 @@ fn log_failed_cancel( agreement_id = %agreement.id, attempts, error = %err, - "Cancel did not end the agreement, will retry" + "Cancel failed, never mined, or did not end the agreement; will retry" ); return; } @@ -300,12 +274,17 @@ fn log_failed_cancel( } /// How many of an agreement's cancel attempts a failure uses up. A cancel the contract -/// refused, before sending or once mined, or that mined without ending the agreement counts, -/// and one that can never be sent uses them all. An unreachable chain is retried freely. +/// refused, before sending or once mined, that mined without ending the agreement, or that +/// endpoints answered had no receipt counts, and one that can never be sent uses them all. An +/// unreachable chain, including one whose receipt checks all failed, is retried freely. fn failed_attempts(err: &ChainClientError) -> u32 { match err { ChainClientError::CancelNotConfirmed { .. } | ChainClientError::TxReverted { .. } + | ChainClientError::TxDropped { + receipt_checked: true, + .. + } | ChainClientError::ContractRevert { .. } => 1, ChainClientError::MissingTermsVersionHash { .. } => MAX_CANCEL_ATTEMPTS, _ => 0, @@ -327,6 +306,7 @@ mod tests { use super::*; use crate::{ cancel_dispatch::tests::agreement, + chain_client::AgreementOnChain, registry::{IndexingAgreementStatus, StubAgreementRegistry}, }; @@ -341,6 +321,7 @@ mod tests { audited_by: Mutex>, attempts: AtomicU32, checks: AtomicU32, + writes: Mutex>, } #[async_trait] @@ -358,6 +339,7 @@ mod tests { id: &IndexingAgreementId, ) -> crate::registry::Result<()> { self.marked_cancelled.lock().unwrap().push(*id); + self.writes.lock().unwrap().push("ended"); Ok(()) } async fn record_cancel_audit( @@ -368,6 +350,7 @@ mod tests { canceled_tx: Option<&str>, ) -> crate::registry::Result<()> { self.audited_by.lock().unwrap().push(canceled_by.to_owned()); + self.writes.lock().unwrap().push("cancel recorded"); self.audits .lock() .unwrap() @@ -404,13 +387,18 @@ mod tests { read_fails: bool, send_fails: bool, mined_cancel_reverts: bool, + never_mines: bool, + receipt_unreadable: bool, reverts_before_sending: bool, cancel_has_no_effect: bool, clock_fails: bool, + clock_reads: AtomicU32, now: AtomicU64, ended_by_indexer: bool, - who_read_fails: bool, + indexer_ends_it_first: bool, + read_back_fails: bool, cancels_sent: AtomicU32, + reads: AtomicU32, } #[async_trait] @@ -442,8 +430,14 @@ mod tests { tx_hash: B256::repeat_byte(0xee), }); } + if self.never_mines || self.receipt_unreadable { + return Err(ChainClientError::TxDropped { + tx_hash: B256::repeat_byte(0xdd), + receipt_checked: self.never_mines, + }); + } self.cancels_sent.fetch_add(1, Ordering::SeqCst); - if !self.cancel_has_no_effect { + if !self.cancel_has_no_effect || self.indexer_ends_it_first { self.live.store(false, Ordering::SeqCst); } Ok(Some(B256::repeat_byte(0xcd))) @@ -462,25 +456,24 @@ mod tests { ) -> Result, ChainClientError> { unimplemented!() } - async fn agreement_still_active( + async fn agreement_on_chain( &self, _agreement_id: &[u8; 16], - ) -> Result { - if self.read_fails { + ) -> Result { + let earlier_reads = self.reads.fetch_add(1, Ordering::SeqCst); + if self.read_fails || (self.read_back_fails && earlier_reads > 0) { return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); } - Ok(self.live.load(Ordering::SeqCst)) - } - async fn agreement_ended_by_indexer( - &self, - _agreement_id: &[u8; 16], - ) -> Result { - if self.who_read_fails { - return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); - } - Ok(self.ended_by_indexer) + Ok(if self.live.load(Ordering::SeqCst) { + AgreementOnChain::Live + } else if self.ended_by_indexer || self.indexer_ends_it_first { + AgreementOnChain::EndedByIndexer + } else { + AgreementOnChain::NotLive + }) } async fn latest_block_timestamp(&self) -> Result { + self.clock_reads.fetch_add(1, Ordering::SeqCst); if self.clock_fails { return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); } @@ -527,6 +520,16 @@ mod tests { assert_eq!(registry.checks.load(Ordering::SeqCst), 0); } + #[tokio::test] + async fn leaves_the_chain_alone_when_nothing_is_being_cancelled() { + let registry = MockRegistry::default(); + let chain = live_chain(); + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.clock_reads.load(Ordering::SeqCst), 0); + } + #[tokio::test] async fn cancels_a_live_accepted_agreement_and_records_the_cancel() { let registry = registry_with_one(true); @@ -540,6 +543,20 @@ mod tests { assert_eq!(*registry.audits.lock().unwrap(), vec![Some(tx)]); } + #[tokio::test] + async fn records_the_cancel_before_marking_the_agreement_ended() { + // The end is announced once the mark lands; recorded after, the announcement could + // go out without its transaction and never be sent again. + let registry = registry_with_one(true); + + retry(®istry, &live_chain(), 0).await; + + assert_eq!( + *registry.writes.lock().unwrap(), + vec!["cancel recorded", "ended"] + ); + } + #[tokio::test] async fn leaves_an_accepted_agreement_that_already_ended_to_the_listener() { // The indexer may have ended it, or an earlier cancel whose result went unread; @@ -609,19 +626,54 @@ mod tests { } #[tokio::test] - async fn does_not_confirm_an_end_when_who_ended_it_cannot_be_read() { + async fn reads_an_ended_agreement_once_to_learn_who_ended_it() { let mut registry = registry_with_one(true); registry.cancelling[0].agreement.updated_at = time::OffsetDateTime::now_utc() - LISTENER_GRACE - time::Duration::MINUTE; let chain = MockChain { - who_read_fails: true, + ended_by_indexer: true, ..MockChain::default() }; retry(®istry, &chain, 0).await; + assert_eq!(chain.reads.load(Ordering::SeqCst), 1); + assert_eq!(registry.marked_by_indexer.lock().unwrap().len(), 1); + } + + #[tokio::test] + async fn leaves_an_end_the_indexer_beat_dipper_to_as_theirs() { + // Dipper's cancel mined as a no-op after the indexer's; recording it as dipper's + // would announce the wrong canceller, and the listener's details would be ignored. + let registry = registry_with_one(true); + let chain = MockChain { + indexer_ends_it_first: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 1); + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + assert!(registry.audits.lock().unwrap().is_empty()); + assert_eq!(registry.attempts.load(Ordering::SeqCst), 0); + } + + #[tokio::test] + async fn neither_counts_nor_confirms_a_mined_cancel_it_could_not_read_back() { + // It may well have worked; the next check reads the agreement again. + let registry = registry_with_one(true); + let chain = MockChain { + read_back_fails: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 1); + assert_eq!(registry.attempts.load(Ordering::SeqCst), 0); + assert_eq!(registry.checks.load(Ordering::SeqCst), 1); assert!(registry.marked_cancelled.lock().unwrap().is_empty()); - assert!(registry.marked_by_indexer.lock().unwrap().is_empty()); } #[tokio::test] @@ -729,6 +781,34 @@ mod tests { assert_eq!(registry.attempts.load(Ordering::SeqCst), 1); } + #[tokio::test] + async fn counts_a_cancel_that_never_mines() { + // Otherwise one that keeps being dropped is sent every sweep for ever, never alerting. + let registry = registry_with_one(true); + let chain = MockChain { + never_mines: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(registry.attempts.load(Ordering::SeqCst), 1); + } + + #[tokio::test] + async fn does_not_count_a_cancel_whose_receipt_could_not_be_checked() { + // Every receipt check failing is an outage, which may hide a cancel that mined. + let registry = registry_with_one(true); + let chain = MockChain { + receipt_unreadable: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(registry.attempts.load(Ordering::SeqCst), 0); + } + #[tokio::test] async fn counts_a_cancel_the_contract_refuses_before_it_is_sent() { // Otherwise one that always reverts is retried, and alerted on, for ever. diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index 80111bad..e7e5c76f 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -1241,10 +1241,10 @@ async fn expired_but_live( agreement: &IndexingAgreement, ) -> Option { match chain_client - .agreement_still_active(agreement.id.as_bytes()) + .agreement_on_chain(agreement.id.as_bytes()) .await { - Ok(live) => Some(live), + Ok(on_chain) => Some(on_chain.is_live()), Err(err) => { tracing::warn!( old_agreement_id = %agreement.id, @@ -2635,19 +2635,16 @@ mod tests { Ok(None) } - async fn agreement_still_active( + async fn agreement_on_chain( &self, agreement_id: &[u8; 16], - ) -> Result { + ) -> Result + { // Cancel dispatch always reads back after a mined cancel; reporting // not-active here means "cancel confirmed", which these tests expect. - Ok(self.live_until_cancelled && !self.cancels.lock().unwrap().contains(agreement_id)) - } - async fn agreement_ended_by_indexer( - &self, - _agreement_id: &[u8; 16], - ) -> Result { - Ok(false) + Ok(crate::chain_client::AgreementOnChain::live_if( + self.live_until_cancelled && !self.cancels.lock().unwrap().contains(agreement_id), + )) } } diff --git a/bin/dipper-service/src/network/service/escrow_reconciler.rs b/bin/dipper-service/src/network/service/escrow_reconciler.rs index 5fba0f1a..7151042b 100644 --- a/bin/dipper-service/src/network/service/escrow_reconciler.rs +++ b/bin/dipper-service/src/network/service/escrow_reconciler.rs @@ -540,6 +540,7 @@ mod tests { use thegraph_core::alloy::primitives::B256; use super::*; + use crate::chain_client::AgreementOnChain; const NOW: u64 = 1_800_000_000; @@ -692,18 +693,12 @@ mod tests { } Ok(Some(B256::ZERO)) } - async fn agreement_still_active( + async fn agreement_on_chain( &self, _agreement_id: &[u8; 16], - ) -> Result { + ) -> Result { unimplemented!() } - async fn agreement_ended_by_indexer( - &self, - _agreement_id: &[u8; 16], - ) -> Result { - Ok(false) - } async fn reconcile_agreement( &self, _collector: Address, diff --git a/bin/dipper-service/src/network/service/liveness_checker.rs b/bin/dipper-service/src/network/service/liveness_checker.rs index 367f3a98..4c2733a4 100644 --- a/bin/dipper-service/src/network/service/liveness_checker.rs +++ b/bin/dipper-service/src/network/service/liveness_checker.rs @@ -43,6 +43,7 @@ use tokio::{sync::mpsc, time::MissedTickBehavior}; use url::Url; use crate::{ + cancel_dispatch::LiveCancel, chain_client::{ChainClient, ChainClientError}, config::LivenessCheckerConfig, network::provider::NetworkProviderService, @@ -508,21 +509,21 @@ async fn cancel_and_reassess( match crate::cancel_dispatch::cancel_agreement_on_chain(chain_client, agreement, agreement_conf) .await { - Ok(Some(tx_hash)) => { + LiveCancel::Ended(tx_hash) => { tracing::info!( agreement_id = %agreement.id, - tx_hash = %tx_hash, + tx_hash = ?tx_hash, "canceled stale agreement on-chain" ); - on_chain_cancel_tx = Some(tx_hash.to_string()); + on_chain_cancel_tx = tx_hash.map(|hash| hash.to_string()); } - Ok(None) => { + LiveCancel::NotLive { .. } => { tracing::info!( agreement_id = %agreement.id, "stale agreement already canceled on-chain; proceeding to mark abandoned" ); } - Err(err @ ChainClientError::MissingTermsVersionHash { .. }) => { + LiveCancel::CancelFailed(err @ ChainClientError::MissingTermsVersionHash { .. }) => { // Permanent per-agreement condition: the on-chain agreement is // still live, so do NOT mark abandoned (that would hide a // money-draining agreement). Surface for operator action. @@ -533,7 +534,7 @@ async fn cancel_and_reassess( ); return; } - Err(ChainClientError::ConfigError(_)) => { + LiveCancel::CancelFailed(ChainClientError::ConfigError(_)) => { // Chain client disabled: still proceed to mark and reassess so the // DB reflects the detected abandonment even without an on-chain tx. tracing::warn!( @@ -541,7 +542,11 @@ async fn cancel_and_reassess( "chain client not configured, skipping on-chain cancellation" ); } - Err(err) => { + LiveCancel::Unconfirmed { tx_hash, err } => { + crate::cancel_dispatch::log_unconfirmed(agreement, tx_hash, &err); + return; + } + LiveCancel::ReadFailed(err) | LiveCancel::CancelFailed(err) => { tracing::error!( agreement_id = %agreement.id, error = %err, @@ -834,7 +839,7 @@ mod tests { record_progress, tolerance_duration, }; use crate::{ - chain_client::{ChainClient, ChainClientError}, + chain_client::{AgreementOnChain, ChainClient, ChainClientError}, config::LivenessCheckerConfig, registry::{ AgreementFeeRate, IndexingAgreement, IndexingAgreementStatus, IndexingAgreementTerms, @@ -1194,19 +1199,13 @@ mod tests { unimplemented!() } - async fn agreement_still_active( + async fn agreement_on_chain( &self, _agreement_id: &[u8; 16], - ) -> Result { + ) -> Result { // Cancel dispatch reads back after a mined cancel; reporting // not-active means "cancel confirmed", which these tests expect. - Ok(false) - } - async fn agreement_ended_by_indexer( - &self, - _agreement_id: &[u8; 16], - ) -> Result { - Ok(false) + Ok(AgreementOnChain::NotLive) } } diff --git a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs index 700cb26a..cde9677a 100644 --- a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs +++ b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs @@ -62,7 +62,7 @@ mod tests { use super::*; use crate::{ cancel_dispatch::tests::agreement, - chain_client::ChainClientError, + chain_client::{AgreementOnChain, ChainClientError}, registry::{IndexingAgreement, StubAgreementRegistry}, }; @@ -123,17 +123,11 @@ mod tests { ) -> Result, ChainClientError> { unimplemented!() } - async fn agreement_still_active( + async fn agreement_on_chain( &self, _agreement_id: &[u8; 16], - ) -> Result { - Ok(self.live) - } - async fn agreement_ended_by_indexer( - &self, - _agreement_id: &[u8; 16], - ) -> Result { - unimplemented!() + ) -> Result { + Ok(AgreementOnChain::live_if(self.live)) } async fn latest_block_timestamp(&self) -> Result { unimplemented!() diff --git a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs index d2cf1d2a..1e486ff1 100644 --- a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs +++ b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs @@ -829,7 +829,7 @@ mod lifecycle_event_tests { use super::{Ctx, Message, handle}; use crate::{ - chain_client::{ChainClient, ChainClientError}, + chain_client::{AgreementOnChain, ChainClient, ChainClientError}, config::IndexingAgreementConfig, network::{ provider::NetworkProviderService, @@ -1007,17 +1007,13 @@ mod lifecycle_event_tests { unimplemented!() } - async fn agreement_still_active( + async fn agreement_on_chain( &self, agreement_id: &[u8; 16], - ) -> std::result::Result { - Ok(!self.nothing_on_chain && !self.cancelled.lock().unwrap().contains(agreement_id)) - } - async fn agreement_ended_by_indexer( - &self, - _agreement_id: &[u8; 16], - ) -> Result { - Ok(false) + ) -> std::result::Result { + Ok(AgreementOnChain::live_if( + !self.nothing_on_chain && !self.cancelled.lock().unwrap().contains(agreement_id), + )) } } @@ -2297,7 +2293,7 @@ mod deadline_clock_tests { use super::{JobError, resolve_deadline_clock}; use crate::{ - chain_client::{ChainClient, ChainClientError}, + chain_client::{AgreementOnChain, ChainClient, ChainClientError}, network::service::{ chain_events::Cursor, chain_listener::{ChainListenerState, ChainListenerStateRegistry}, @@ -2348,17 +2344,11 @@ mod deadline_clock_tests { unimplemented!() } - async fn agreement_still_active( - &self, - _agreement_id: &[u8; 16], - ) -> Result { - Ok(false) - } - async fn agreement_ended_by_indexer( + async fn agreement_on_chain( &self, _agreement_id: &[u8; 16], - ) -> Result { - Ok(false) + ) -> Result { + Ok(AgreementOnChain::NotLive) } } diff --git a/bin/dipper-service/src/worker/handlers/submit_offer.rs b/bin/dipper-service/src/worker/handlers/submit_offer.rs index 2a24a1a4..f77fa2ae 100644 --- a/bin/dipper-service/src/worker/handlers/submit_offer.rs +++ b/bin/dipper-service/src/worker/handlers/submit_offer.rs @@ -23,7 +23,7 @@ use thegraph_core::{DeploymentId, alloy::primitives::ChainId}; use url::Url; use crate::{ - cancel_dispatch::{LiveCancel, cancel_if_live}, + cancel_dispatch::{LiveCancel, cancel_if_live, log_unconfirmed}, chain_client::{ChainClient, ChainClientError, decode_revert_reason}, config::IndexingAgreementConfig, indexer_rpc_client::into_sol_rca, @@ -233,7 +233,7 @@ async fn withdraw_offer_if_stored( agreement: &IndexingAgreement, ) -> JobResult<()> { match cancel_if_live(&ctx.chain_client, agreement, &ctx.agreement_conf).await { - LiveCancel::NotLive => Ok(()), + LiveCancel::NotLive { .. } => Ok(()), LiveCancel::Ended(tx_hash) => { tracing::info!( agreement_id = %agreement.id, @@ -250,6 +250,10 @@ async fn withdraw_offer_if_stored( ); Err(JobError::Fatal(err.into())) } + LiveCancel::Unconfirmed { tx_hash, err } => { + log_unconfirmed(agreement, tx_hash, &err); + Err(retry_withdraw(agreement, err)) + } LiveCancel::ReadFailed(err) | LiveCancel::CancelFailed(err) => { Err(retry_withdraw(agreement, err)) } @@ -310,6 +314,7 @@ mod tests { use super::*; use crate::{ + chain_client::AgreementOnChain, indexer_rpc_client::compute_on_chain_id, registry::{ IndexingAgreement, IndexingAgreementTerms, IndexingAgreementTermsMetadata, @@ -412,17 +417,13 @@ mod tests { ) -> Result, ChainClientError> { unimplemented!() } - async fn agreement_still_active( - &self, - _agreement_id: &[u8; 16], - ) -> Result { - Ok(self.on_chain.load(Ordering::SeqCst)) - } - async fn agreement_ended_by_indexer( + async fn agreement_on_chain( &self, _agreement_id: &[u8; 16], - ) -> Result { - Ok(false) + ) -> Result { + Ok(AgreementOnChain::live_if( + self.on_chain.load(Ordering::SeqCst), + )) } async fn latest_block_timestamp(&self) -> Result { unimplemented!() From 72bda2ab031bd715da578a754cc411f80c0af198 Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Mon, 5 Oct 2026 12:10:52 +0100 Subject: [PATCH 15/26] fix: time and order the cancel retry by what it has seen (#728) * fix(cancel): give the listener its hour from when an end is first seen The chain listener gets an hour to record how an agreement ended before the cancel retry closes it out, but the hour ran from when it was marked, so one cancelling for hours closed at once. A new column notes when a check first finds it ended, and the hour runs from then. * fix(cancel): cancel a reopened agreement on the next sweep An agreement found live after dipper ended it goes back to cancelling, but then sat out the 2 minutes meant for a cancel already in flight, though none was sent, and kept paying for about 7. Only agreements never checked since being marked wait now. * fix(cancel): stop new cancels jumping ahead of ones that may be paying The cancel retry takes agreements checked longest ago first, with a head start for any that may be paying, but never-checked ones always came first, so a burst of new offers held back paying agreements. A never-checked one now counts as checked when it was marked. * docs(registry): describe when the cancel retry notes an end accurately Two comments overstated things: the end time is noted whenever a check finds the agreement no longer live, not only when dipper has no cancel to show, and an agreement can be moved back to cancelling after a failed chain read, not only after a successful one. --- .../src/network/service/cancel_retry.rs | 76 ++++++-- bin/dipper-service/src/registry.rs | 3 +- bin/dipper-service/src/registry/agreement.rs | 7 +- .../src/registry/agreement_stub.rs | 4 +- ...0261005000000_add_cancel_ended_seen_at.sql | 5 + dipper-pgregistry/src/postgres.rs | 43 +++-- .../tests/it_registry_postgres.rs | 162 +++++++++++++----- 7 files changed, 229 insertions(+), 71 deletions(-) create mode 100644 dipper-pgregistry/migrations/20261005000000_add_cancel_ended_seen_at.sql diff --git a/bin/dipper-service/src/network/service/cancel_retry.rs b/bin/dipper-service/src/network/service/cancel_retry.rs index 4f6dfc1f..e56a6c07 100644 --- a/bin/dipper-service/src/network/service/cancel_retry.rs +++ b/bin/dipper-service/src/network/service/cancel_retry.rs @@ -25,11 +25,12 @@ const BATCH_SIZE: i64 = 50; const SWEEP_BUDGET: std::time::Duration = std::time::Duration::from_secs(30); /// Minutes an agreement stays out of the retry after it is marked, so the cancel sent -/// when it was marked can be mined first instead of being sent again. +/// when it was marked can be mined first instead of being sent again. One moved back to +/// cancelling had none sent, so it doesn't wait. const SETTLE_MINUTES: i32 = 2; -/// How long the chain listener gets to record when, and in which transaction, an accepted -/// agreement ended, before the retry marks it ended without those details. +/// How long the chain listener gets, from when a check first finds an agreement ended, to +/// record when and in which transaction it ended, before the retry marks it without them. const LISTENER_GRACE: time::Duration = time::Duration::HOUR; /// Retry the cancel of agreements still `Cancelling`. The chain's own latest block time @@ -105,7 +106,7 @@ async fn retry_cancel( "Failed to read a cancelling agreement on-chain, will retry" ); // Unread, it may still be live, so it can't be confirmed ended. - return note_check(registry, row, None).await; + return note_check(registry, row, None, None).await; } LiveCancel::NotLive { by_indexer } => (None, by_indexer, None), LiveCancel::Ended(tx_hash) => { @@ -119,7 +120,7 @@ async fn retry_cancel( LiveCancel::CancelFailed(err) => (None, false, Some(err)), LiveCancel::Unconfirmed { tx_hash, err } => { log_unconfirmed(&row.agreement, tx_hash, &err); - return note_check(registry, row, None).await; + return note_check(registry, row, None, None).await; } }; if failure.is_none() @@ -127,7 +128,8 @@ async fn retry_cancel( { return; } - note_check(registry, row, failure.as_ref()).await; + // A cancel that failed found it live; otherwise it is over, or withdrawn until its deadline. + note_check(registry, row, failure.as_ref(), Some(failure.is_none())).await; } /// Mark the agreement `CanceledByRequester` once it can't go live again: this sweep's cancel @@ -142,7 +144,9 @@ async fn confirm_if_over( chain_now: u64, ) -> bool { let agreement = &row.agreement; - let past_grace = agreement.updated_at < time::OffsetDateTime::now_utc() - LISTENER_GRACE; + let past_grace = row + .ended_seen_at + .is_some_and(|seen| seen < time::OffsetDateTime::now_utc() - LISTENER_GRACE); let can_confirm = if row.accepted_on_chain { tx_hash.is_some() || past_grace } else { @@ -204,12 +208,13 @@ async fn record_end_by_indexer( } } -/// Record that the agreement was checked and is still cancelling, counting a cancel the -/// chain answered without ending it; past the limit, dipper gives up with an ERROR. +/// Record that the agreement was checked and is still cancelling, and whether it was found +/// ended, counting a cancel the chain answered without ending it; at the limit, an ERROR. async fn note_check( registry: &R, row: &CancellingAgreement, failure: Option<&ChainClientError>, + ended: Option, ) { let agreement = &row.agreement; let failed_attempts = failure.map_or(0, failed_attempts); @@ -219,7 +224,7 @@ async fn note_check( log_uncounted_failure(agreement, err); } match registry - .record_cancel_check(&agreement.id, failed_attempts) + .record_cancel_check(&agreement.id, failed_attempts, ended) .await { Ok(attempts) => { @@ -321,6 +326,7 @@ mod tests { audited_by: Mutex>, attempts: AtomicU32, checks: AtomicU32, + found_ended: Mutex>>, writes: Mutex>, } @@ -374,8 +380,10 @@ mod tests { &self, _id: &IndexingAgreementId, failed_attempts: u32, + ended: Option, ) -> crate::registry::Result { self.checks.fetch_add(1, Ordering::SeqCst); + self.found_ended.lock().unwrap().push(ended); Ok(self.attempts.fetch_add(failed_attempts, Ordering::SeqCst) + failed_attempts) } } @@ -488,6 +496,7 @@ mod tests { cancelling: vec![CancellingAgreement { agreement: cancelling, accepted_on_chain, + ended_seen_at: None, }], ..MockRegistry::default() } @@ -575,8 +584,8 @@ mod tests { async fn marks_an_ended_accepted_agreement_itself_once_the_listener_has_had_long_enough() { // In case the listener never reads its end, it would otherwise stay cancelling. let mut registry = registry_with_one(true); - registry.cancelling[0].agreement.updated_at = - time::OffsetDateTime::now_utc() - LISTENER_GRACE - time::Duration::MINUTE; + registry.cancelling[0].ended_seen_at = + Some(time::OffsetDateTime::now_utc() - LISTENER_GRACE - time::Duration::MINUTE); let chain = MockChain::default(); retry(®istry, &chain, 0).await; @@ -585,6 +594,41 @@ mod tests { assert!(registry.audits.lock().unwrap().is_empty()); } + #[tokio::test] + async fn gives_the_listener_its_hour_from_when_the_end_is_first_seen() { + // Cancelling for hours, as while the manager was paused, mustn't count towards it. + let mut registry = registry_with_one(true); + registry.cancelling[0].agreement.updated_at = + time::OffsetDateTime::now_utc() - LISTENER_GRACE * 3; + let chain = MockChain::default(); + + retry(®istry, &chain, 0).await; + + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + assert_eq!(*registry.found_ended.lock().unwrap(), vec![Some(true)]); + } + + #[tokio::test] + async fn notes_a_live_agreement_as_not_ended_and_an_unread_one_as_unknown() { + let registry = registry_with_one(true); + let chain = MockChain { + cancel_has_no_effect: true, + ..live_chain() + }; + retry(®istry, &chain, 0).await; + + let unread = MockChain { + read_fails: true, + ..MockChain::default() + }; + retry(®istry, &unread, 0).await; + + assert_eq!( + *registry.found_ended.lock().unwrap(), + vec![Some(false), None] + ); + } + #[tokio::test] async fn leaves_an_end_by_the_indexer_to_the_listener_for_a_while() { // The listener records when and in which transaction. An accepted agreement can lack @@ -608,8 +652,8 @@ mod tests { async fn marks_an_end_by_the_indexer_as_theirs_once_the_listener_has_had_long_enough() { for accepted_on_chain in [true, false] { let mut registry = registry_with_one(accepted_on_chain); - registry.cancelling[0].agreement.updated_at = - time::OffsetDateTime::now_utc() - LISTENER_GRACE - time::Duration::MINUTE; + registry.cancelling[0].ended_seen_at = + Some(time::OffsetDateTime::now_utc() - LISTENER_GRACE - time::Duration::MINUTE); let chain = MockChain { ended_by_indexer: true, ..MockChain::default() @@ -628,8 +672,8 @@ mod tests { #[tokio::test] async fn reads_an_ended_agreement_once_to_learn_who_ended_it() { let mut registry = registry_with_one(true); - registry.cancelling[0].agreement.updated_at = - time::OffsetDateTime::now_utc() - LISTENER_GRACE - time::Duration::MINUTE; + registry.cancelling[0].ended_seen_at = + Some(time::OffsetDateTime::now_utc() - LISTENER_GRACE - time::Duration::MINUTE); let chain = MockChain { ended_by_indexer: true, ..MockChain::default() diff --git a/bin/dipper-service/src/registry.rs b/bin/dipper-service/src/registry.rs index eb5c7185..89c96724 100644 --- a/bin/dipper-service/src/registry.rs +++ b/bin/dipper-service/src/registry.rs @@ -431,9 +431,10 @@ impl AgreementRegistry for RegistryProvider { &self, id: &IndexingAgreementId, failed_attempts: u32, + ended: Option, ) -> RegistryResult { self.inner - .record_cancel_check(id, failed_attempts) + .record_cancel_check(id, failed_attempts, ended) .await .map_err(Into::into) } diff --git a/bin/dipper-service/src/registry/agreement.rs b/bin/dipper-service/src/registry/agreement.rs index 24122084..3aeb452e 100644 --- a/bin/dipper-service/src/registry/agreement.rs +++ b/bin/dipper-service/src/registry/agreement.rs @@ -293,11 +293,13 @@ pub trait AgreementRegistry { ) -> RegistryResult>; /// Record a check of a `CANCELLING` agreement that left it cancelling, adding - /// `failed_attempts` to its failed cancels and returning the new count. + /// `failed_attempts` to its failed cancels and returning the new count. `ended` says whether + /// the check found it no longer live on-chain, or `None` when the chain couldn't tell. async fn record_cancel_check( &self, id: &IndexingAgreementId, failed_attempts: u32, + ended: Option, ) -> RegistryResult; /// Apply a reconciliation-driven state transition atomically. @@ -562,6 +564,8 @@ pub struct CancellingAgreement { pub agreement: IndexingAgreement, /// Whether dipper saw it accepted on-chain, so its end is announced. pub accepted_on_chain: bool, + /// When a check first found it no longer live on-chain, if one has. + pub ended_seen_at: Option, } impl TryFrom for CancellingAgreement { @@ -571,6 +575,7 @@ impl TryFrom for CancellingAgreement { Ok(Self { agreement: value.agreement.try_into()?, accepted_on_chain: value.accepted_on_chain, + ended_seen_at: value.ended_seen_at, }) } } diff --git a/bin/dipper-service/src/registry/agreement_stub.rs b/bin/dipper-service/src/registry/agreement_stub.rs index 9e516c4b..aa91b431 100644 --- a/bin/dipper-service/src/registry/agreement_stub.rs +++ b/bin/dipper-service/src/registry/agreement_stub.rs @@ -154,6 +154,7 @@ pub trait StubAgreementRegistry: Send + Sync { &self, _id: &IndexingAgreementId, failed_attempts: u32, + _ended: Option, ) -> Result { Ok(failed_attempts) } @@ -467,8 +468,9 @@ impl AgreementRegistry for T { &self, id: &IndexingAgreementId, failed_attempts: u32, + ended: Option, ) -> Result { - StubAgreementRegistry::record_cancel_check(self, id, failed_attempts).await + StubAgreementRegistry::record_cancel_check(self, id, failed_attempts, ended).await } async fn apply_reconciliation( diff --git a/dipper-pgregistry/migrations/20261005000000_add_cancel_ended_seen_at.sql b/dipper-pgregistry/migrations/20261005000000_add_cancel_ended_seen_at.sql new file mode 100644 index 00000000..fdd6df7a --- /dev/null +++ b/dipper-pgregistry/migrations/20261005000000_add_cancel_ended_seen_at.sql @@ -0,0 +1,5 @@ +-- ended_seen_at: when dipper's cancel retry first found a Cancelling agreement no longer live +-- on-chain. Where the retry has no cancel of its own to show for the end, the chain listener +-- gets an hour from then to record how it ended before the retry closes it out without that. +ALTER TABLE dipper_reg_indexing_agreements + ADD COLUMN ended_seen_at TIMESTAMPTZ; diff --git a/dipper-pgregistry/src/postgres.rs b/dipper-pgregistry/src/postgres.rs index 3e7d062b..1d7ebe1c 100644 --- a/dipper-pgregistry/src/postgres.rs +++ b/dipper-pgregistry/src/postgres.rs @@ -193,6 +193,8 @@ pub struct CancellingAgreement { pub agreement: IndexingAgreement, /// Whether dipper saw it accepted on-chain, so its end is announced. pub accepted_on_chain: bool, + /// When a check first found it no longer live on-chain, if one has. + pub ended_seen_at: Option, } impl sqlx::FromRow<'_, sqlx::postgres::PgRow> for CancellingAgreement { @@ -202,6 +204,7 @@ impl sqlx::FromRow<'_, sqlx::postgres::PgRow> for CancellingAgreement { Ok(Self { agreement: IndexingAgreement::from_row(row)?, accepted_on_chain: accepted_at.is_some(), + ended_seen_at: row.try_get("ended_seen_at")?, }) } } @@ -972,7 +975,9 @@ impl PgRegistry { } /// Move an agreement dipper had already ended, cancelled or rejected, back to `Cancelling` - /// once the chain shows it live after all, with its cancel attempts started afresh. + /// once the chain shows it live after all, with its cancel attempts started afresh. It counts + /// as checked, since no cancel is sent with it, so the retry takes it on its next sweep rather + /// than waiting for one to be mined. pub async fn reopen_indexing_agreement_cancel( &self, agreement_id: &IndexingAgreementId, @@ -983,7 +988,8 @@ impl PgRegistry { SET status = $1, cancel_attempts = 0, - cancel_checked_at = NULL, + cancel_checked_at = timezone('UTC', now()), + ended_seen_at = NULL, updated_at = timezone('UTC', now()) WHERE id = $2 AND status IN ($3, $4) "#, @@ -1000,10 +1006,12 @@ impl PgRegistry { Ok(()) } - /// `Cancelling` agreements marked over `min_age_minutes` ago, those checked longest ago - /// first; one whose cancel has failed `max_attempts` times only once an hour. One that may be paying + /// `Cancelling` agreements, those checked longest ago first, leaving out any never checked + /// that was marked in the last `min_age_minutes`, so the cancel sent with its mark can be + /// mined first; one whose cancel has failed `max_attempts` times only once an hour. One that may be paying /// an indexer (accepted, or past the offer deadline, which only an accepted one outlives) - /// counts as checked an hour earlier, so it goes first without holding the rest back. + /// counts as checked an hour earlier, so it goes first without holding the rest back. One + /// never checked counts as checked when it was marked, so a burst of new ones can't jump it. pub async fn get_cancelling_agreements( &self, batch_size: i64, @@ -1027,21 +1035,25 @@ impl PgRegistry { last_progress_at, rejection_reason, terms_version_hash, - accepted_at + accepted_at, + ended_seen_at FROM dipper_reg_indexing_agreements WHERE status = $1 AND ( cancel_attempts < $2 OR cancel_checked_at < timezone('UTC', now()) - INTERVAL '1 hour' ) - AND updated_at < timezone('UTC', now()) - make_interval(mins => $4) + AND ( + cancel_checked_at IS NOT NULL + OR updated_at < timezone('UTC', now()) - make_interval(mins => $4) + ) ORDER BY - cancel_checked_at - CASE + COALESCE(cancel_checked_at, updated_at) - CASE WHEN accepted_at IS NOT NULL OR CAST(terms->>'deadline' AS bigint) < EXTRACT(EPOCH FROM now()) THEN INTERVAL '1 hour' ELSE INTERVAL '0 seconds' - END ASC NULLS FIRST, + END ASC, updated_at ASC LIMIT $3 "#, @@ -1056,18 +1068,26 @@ impl PgRegistry { } /// Record a check of a `Cancelling` agreement that left it cancelling, adding - /// `failed_attempts` to its failed cancels and returning the new count. + /// `failed_attempts` to its failed cancels and returning the new count. `ended` says whether + /// the check found it no longer live on-chain, or `None` when the chain couldn't tell; the + /// first time it is found ended is kept until it is found live again. pub async fn record_cancel_check( &self, agreement_id: &IndexingAgreementId, failed_attempts: u32, + ended: Option, ) -> Result { let record: Option<(i32,)> = sqlx::query_as( r#" UPDATE dipper_reg_indexing_agreements SET cancel_attempts = LEAST(cancel_attempts::BIGINT + $3, 2147483647)::INTEGER, - cancel_checked_at = timezone('UTC', now()) + cancel_checked_at = timezone('UTC', now()), + ended_seen_at = CASE + WHEN $4::BOOLEAN IS NULL THEN ended_seen_at + WHEN $4 THEN COALESCE(ended_seen_at, timezone('UTC', now())) + ELSE NULL + END WHERE id = $1 AND status = $2 RETURNING cancel_attempts "#, @@ -1075,6 +1095,7 @@ impl PgRegistry { .bind(agreement_id) .bind(IndexingAgreementStatus::Cancelling) .bind(i64::from(failed_attempts)) + .bind(ended) .fetch_optional(&self.pool) .await?; let (attempts,) = record.ok_or(Error::NoRecordsUpdated)?; diff --git a/dipper-pgregistry/tests/it_registry_postgres.rs b/dipper-pgregistry/tests/it_registry_postgres.rs index 215e946c..e9de6789 100644 --- a/dipper-pgregistry/tests/it_registry_postgres.rs +++ b/dipper-pgregistry/tests/it_registry_postgres.rs @@ -3362,16 +3362,6 @@ async fn cancelling_agreements_are_listed_until_their_cancel_fails_too_often() { .await .expect("Failed to run fixture"); let created = fixture_agreement(0xaa); - // An offer still open to acceptance, so only the accepted agreement can be paying. - sqlx::query( - "UPDATE dipper_reg_indexing_agreements \ - SET terms = jsonb_set(terms::jsonb, '{deadline}', to_jsonb(4102444800::bigint)) \ - WHERE id = $1", - ) - .bind(created) - .execute(&db) - .await - .expect("Failed to update deadline"); let registry = PgRegistry::new(db.clone()); let accepted = IndexingAgreementId::from_bytes([0xaa, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2]); @@ -3431,8 +3421,10 @@ async fn cancelling_agreements_are_listed_until_their_cancel_fails_too_often() { ] ); - assert_eq!(registry.record_cancel_check(&created, 0).await.unwrap(), 0); - assert_eq!(registry.record_cancel_check(&accepted, 0).await.unwrap(), 0); + for attempts in [1, 2] { + let counted = registry.record_cancel_check(&created, 1, None).await; + assert_eq!(counted.unwrap(), attempts); + } let listed = registry .get_cancelling_agreements(100, 2, 0) .await @@ -3440,8 +3432,8 @@ async fn cancelling_agreements_are_listed_until_their_cancel_fails_too_often() { let ids: Vec<_> = listed.iter().map(|row| row.agreement.id).collect(); assert_eq!( ids, - vec![accepted, created], - "one accepted on-chain may be paying its indexer, so it goes first" + vec![accepted], + "one that failed too often waits an hour between checks" ); sqlx::query( @@ -3456,47 +3448,131 @@ async fn cancelling_agreements_are_listed_until_their_cancel_fails_too_often() { .get_cancelling_agreements(100, 2, 0) .await .expect("cancelling query"); - let ids: Vec<_> = listed.iter().map(|row| row.agreement.id).collect(); - assert_eq!( - ids, - vec![created, accepted], - "an offer unchecked for over an hour isn't held back for ever" + assert!( + listed.iter().any(|row| row.agreement.id == created), + "and is tried again after it" ); - assert_eq!(registry.record_cancel_check(&created, 1).await.unwrap(), 1); - assert_eq!(registry.record_cancel_check(&created, 1).await.unwrap(), 2); - let listed = registry - .get_cancelling_agreements(100, 2, 0) + let not_cancelling = registry + .record_cancel_check(&fixture_agreement(0xbb), 1, None) + .await; + assert!(matches!(not_cancelling, Err(Error::NoRecordsUpdated))); +} + +#[tokio::test] +async fn cancelling_agreements_that_may_be_paying_go_first() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let offer = fixture_agreement(0xaa); + // An offer still open to acceptance, so only the accepted agreement can be paying. + sqlx::query( + "UPDATE dipper_reg_indexing_agreements \ + SET terms = jsonb_set(terms::jsonb, '{deadline}', to_jsonb(4102444800::bigint)) \ + WHERE id = $1", + ) + .bind(offer) + .execute(&db) + .await + .expect("Failed to update deadline"); + let registry = PgRegistry::new(db.clone()); + let accepted = + IndexingAgreementId::from_bytes([0xaa, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2]); + registry + .record_accepted_audit(&accepted, 1_700_000_000, "0xacc") .await - .expect("cancelling query"); - let ids: Vec<_> = listed.iter().map(|row| row.agreement.id).collect(); + .expect("accept record"); + for id in [accepted, offer] { + registry + .mark_indexing_agreement_as_cancelling(&id) + .await + .expect("mark cancelling"); + } + let order = async || -> Vec { + let listed = registry + .get_cancelling_agreements(100, 2, 0) + .await + .expect("cancelling query"); + listed.iter().map(|row| row.agreement.id).collect() + }; + + registry + .record_cancel_check(&accepted, 0, None) + .await + .unwrap(); assert_eq!( - ids, - vec![accepted], - "one that failed too often waits an hour between checks" + order().await, + vec![accepted, offer], + "one accepted on-chain may be paying its indexer, so it goes ahead of an offer newly \ + marked, which counts as checked when it was marked" ); + registry.record_cancel_check(&offer, 0, None).await.unwrap(); + assert_eq!(order().await, vec![accepted, offer]); + sqlx::query( "UPDATE dipper_reg_indexing_agreements \ SET cancel_checked_at = cancel_checked_at - INTERVAL '2 hours' WHERE id = $1", ) - .bind(created) + .bind(offer) .execute(&db) .await .expect("Failed to age the check"); - let listed = registry - .get_cancelling_agreements(100, 2, 0) + assert_eq!( + order().await, + vec![offer, accepted], + "an offer unchecked for over an hour isn't held back for ever" + ); +} + +#[tokio::test] +async fn a_cancelling_agreement_keeps_when_it_was_first_found_ended() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let registry = PgRegistry::new(db); + let id = fixture_agreement(0xaa); + registry + .mark_indexing_agreement_as_cancelling(&id) .await - .expect("cancelling query"); - assert!( - listed.iter().any(|row| row.agreement.id == created), - "and is tried again after it" + .expect("mark cancelling"); + let ended_seen_at = async || { + registry + .get_cancelling_agreements(100, 10, 0) + .await + .expect("cancelling query")[0] + .ended_seen_at + }; + + registry + .record_cancel_check(&id, 0, Some(true)) + .await + .unwrap(); + let first = ended_seen_at().await.expect("found ended"); + registry + .record_cancel_check(&id, 0, Some(true)) + .await + .unwrap(); + registry.record_cancel_check(&id, 0, None).await.unwrap(); + assert_eq!( + ended_seen_at().await, + Some(first), + "kept from the first time" ); - let not_cancelling = registry - .record_cancel_check(&fixture_agreement(0xbb), 1) - .await; - assert!(matches!(not_cancelling, Err(Error::NoRecordsUpdated))); + registry + .record_cancel_check(&id, 0, Some(false)) + .await + .unwrap(); + assert_eq!(ended_seen_at().await, None, "found live again"); } #[tokio::test] @@ -3516,7 +3592,10 @@ async fn an_ended_agreement_found_live_on_chain_goes_back_to_cancelling() { .mark_indexing_agreement_as_cancelling(&ended) .await .expect("mark cancelling"); - assert_eq!(registry.record_cancel_check(&ended, 2).await.unwrap(), 2); + assert_eq!( + registry.record_cancel_check(&ended, 2, None).await.unwrap(), + 2 + ); registry .mark_indexing_agreement_as_canceled_by_requester(&ended) .await @@ -3527,8 +3606,9 @@ async fn an_ended_agreement_found_live_on_chain_goes_back_to_cancelling() { .await .expect("an ended agreement can be reopened"); + // Reopening sends no cancel, so there is none to wait on being mined. let listed = registry - .get_cancelling_agreements(100, 1, 0) + .get_cancelling_agreements(100, 1, 5) .await .expect("cancelling query"); let ids: Vec<_> = listed.iter().map(|row| row.agreement.id).collect(); From 9fda601188c0bbaffb64ccb96d93ef8ccc0eef65 Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Mon, 5 Oct 2026 12:10:52 +0100 Subject: [PATCH 16/26] fix: end stale agreements through the cancelling status (#729) * fix(liveness): end stale agreements through the cancelling status The check for indexers that stopped serving cancelled first and marked afterwards, missing the read before sending, the retry limit and the alert. It now marks the agreement cancelling, noted as abandoned so it still ends that way, and the code that only it used is gone. * docs(cancel): describe how an abandoned agreement now ends Several comments still said a cancelled agreement always ends cancelled by dipper and that the liveness check cancels directly. They now say an agreement dipper ends because its indexer stopped serving it passes through cancelling and ends abandoned by the indexer. * fix(cancel): say in the stuck-cancel alert why dipper is cancelling The alert for a cancel that keeps failing didn't say whether dipper was ending the agreement because its indexer stopped serving it, which an operator needs to decide what to do. It now carries that. * fix(liveness): leave a stale agreement that can't be cancelled alone A stale agreement with no stored terms hash was marked cancelling and replaced, but its cancel can never be sent, so both indexers would be paid. It is left active with an ERROR for an operator, as before. --- bin/dipper-service/src/cancel_dispatch.rs | 71 +++- .../src/network/service/cancel_retry.rs | 23 +- .../src/network/service/chain_listener.rs | 29 +- .../src/network/service/liveness_checker.rs | 399 ++++++------------ bin/dipper-service/src/registry.rs | 21 +- bin/dipper-service/src/registry/agreement.rs | 34 +- .../src/registry/agreement_stub.rs | 22 +- .../handlers/reassess_indexing_request.rs | 1 + .../send_indexing_agreement_proposal.rs | 7 - .../20261005000001_add_abandoned_flag.sql | 5 + dipper-pgregistry/src/indexing_agreement.rs | 8 +- dipper-pgregistry/src/postgres.rs | 92 ++-- .../tests/it_registry_postgres.rs | 77 +++- 13 files changed, 365 insertions(+), 424 deletions(-) create mode 100644 dipper-pgregistry/migrations/20261005000001_add_abandoned_flag.sql diff --git a/bin/dipper-service/src/cancel_dispatch.rs b/bin/dipper-service/src/cancel_dispatch.rs index 049dc105..17c0eed2 100644 --- a/bin/dipper-service/src/cancel_dispatch.rs +++ b/bin/dipper-service/src/cancel_dispatch.rs @@ -1,5 +1,5 @@ -//! On-chain cancel dispatch. Every cancel goes through -//! [`cancel_agreement_on_chain`] so the manager-routed path lives in one place. +//! On-chain cancel dispatch. Every cancel starts with [`start_cancel`] and goes out through +//! `cancel_agreement_on_chain`, so the manager-routed path lives in one place. use dipper_core::time::now_secs; use thegraph_core::alloy::primitives::B256; @@ -23,17 +23,12 @@ const SCOPE_BOTH: u16 = SCOPE_ACTIVE | SCOPE_PENDING; /// Cancel an agreement on-chain through the RecurringAgreementManager. Passes /// both scope bits so the collector cancels whichever scope the agreement is in, /// and treats a missing or short stored hash as `MissingTermsVersionHash`. -pub async fn cancel_agreement_on_chain( +async fn cancel_agreement_on_chain( chain_client: &T, agreement: &IndexingAgreement, config: &IndexingAgreementConfig, ) -> LiveCancel { - let Some(version_hash) = agreement - .terms_version_hash - .as_deref() - .filter(|h| h.len() == 32) - .map(B256::from_slice) - else { + let Some(version_hash) = cancel_hash(agreement) else { return LiveCancel::CancelFailed(ChainClientError::MissingTermsVersionHash { agreement_id: agreement.id.to_string(), }); @@ -68,10 +63,39 @@ pub async fn cancel_agreement_on_chain( } } +/// The stored terms hash an on-chain cancel needs, or `None` when the agreement has no 32-byte +/// one, so no cancel can ever be sent for it. +pub fn cancel_hash(agreement: &IndexingAgreement) -> Option { + agreement + .terms_version_hash + .as_deref() + .filter(|h| h.len() == 32) + .map(B256::from_slice) +} + +/// Why dipper is ending an agreement, which decides the status it ends in once the chain +/// confirms dipper's cancel. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum CancelReason { + /// Dipper no longer wants it: it ends `CanceledByRequester`. + NotWanted, + /// Its indexer stopped serving it: it ends `AbandonedByIndexer`. + Abandoned, +} + +impl CancelReason { + fn ended_status(self) -> &'static str { + match self { + Self::NotWanted => "CANCELED_BY_REQUESTER", + Self::Abandoned => "ABANDONED_BY_INDEXER", + } + } +} + /// What [`start_cancel`] left an agreement as. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum CancelStarted { - /// It was accepted and its cancel landed: now `CanceledByRequester`. + /// It was accepted and its cancel landed: now ended, as its [`CancelReason`] says. Ended, /// Still `Cancelling`; the chain listener finishes it once it can't go live. Cancelling, @@ -84,15 +108,25 @@ pub async fn start_cancel( registry: &R, chain_client: &T, agreement: &IndexingAgreement, + reason: CancelReason, config: &IndexingAgreementConfig, ) -> RegistryResult where R: AgreementRegistry + Sync, T: ChainClient, { - registry - .mark_indexing_agreement_as_cancelling(&agreement.id) - .await?; + match reason { + CancelReason::NotWanted => { + registry + .mark_indexing_agreement_as_cancelling(&agreement.id) + .await? + } + CancelReason::Abandoned => { + registry + .mark_indexing_agreement_as_abandoning(&agreement.id) + .await? + } + } let tx_hash = match cancel_if_live(chain_client, agreement, config).await { LiveCancel::Ended(tx_hash) => tx_hash, LiveCancel::NotLive { .. } => return Ok(CancelStarted::Cancelling), @@ -119,7 +153,7 @@ where return Ok(CancelStarted::Cancelling); } Ok( - if confirm_cancelled(registry, agreement, tx_hash, config).await { + if confirm_cancelled(registry, agreement, reason, tx_hash, config).await { CancelStarted::Ended } else { CancelStarted::Cancelling @@ -127,12 +161,13 @@ where ) } -/// Mark an agreement the chain shows dipper ended `CanceledByRequester`, recording the cancel -/// when its transaction is known, so the `terminated` sweep announces it. False, logged, when -/// the mark fails; it stays `Cancelling` for the cancel retry. +/// Mark an agreement the chain shows dipper ended as ended, recording the cancel when its +/// transaction is known, so the `terminated` sweep announces it. False, logged, when the mark +/// fails; it stays `Cancelling` for the cancel retry. pub async fn confirm_cancelled( registry: &R, agreement: &IndexingAgreement, + reason: CancelReason, tx_hash: Option, config: &IndexingAgreementConfig, ) -> bool { @@ -155,7 +190,7 @@ pub async fn confirm_cancelled( agreement_id = %agreement.id, indexing_request_id = %agreement.indexing_request_id, old_status = "CANCELLING", - new_status = "CANCELED_BY_REQUESTER", + new_status = reason.ended_status(), reason = "cancel_confirmed_on_chain", "agreement state transition" ); diff --git a/bin/dipper-service/src/network/service/cancel_retry.rs b/bin/dipper-service/src/network/service/cancel_retry.rs index e56a6c07..ec9e0e63 100644 --- a/bin/dipper-service/src/network/service/cancel_retry.rs +++ b/bin/dipper-service/src/network/service/cancel_retry.rs @@ -1,12 +1,15 @@ //! Finishes the cancels dipper starts. An agreement dipper wants ended is marked //! `Cancelling` before its on-chain cancel goes out; this sweep re-sends the cancel while -//! the chain shows it live, and marks it `CanceledByRequester` once it can no longer be. +//! the chain shows it live, and marks it ended once it can no longer be: `CanceledByRequester`, +//! or `AbandonedByIndexer` for one dipper ended because its indexer stopped serving it. use dipper_core::time::now_secs; use thegraph_core::alloy::primitives::B256; use crate::{ - cancel_dispatch::{LiveCancel, cancel_if_live, confirm_cancelled, log_unconfirmed}, + cancel_dispatch::{ + CancelReason, LiveCancel, cancel_if_live, confirm_cancelled, log_unconfirmed, + }, chain_client::{ChainClient, ChainClientError}, config::IndexingAgreementConfig, registry::{AgreementRegistry, CancelKind, CancellingAgreement, IndexingAgreement}, @@ -132,7 +135,7 @@ async fn retry_cancel( note_check(registry, row, failure.as_ref(), Some(failure.is_none())).await; } -/// Mark the agreement `CanceledByRequester` once it can't go live again: this sweep's cancel +/// Mark the agreement ended by dipper once it can't go live again: this sweep's cancel /// ended it, or nobody accepted its offer before the deadline to. One ended otherwise is left /// to the chain listener for a while; one the indexer ended then becomes `CanceledByIndexer`. async fn confirm_if_over( @@ -162,7 +165,12 @@ async fn confirm_if_over( ); return past_grace && record_end_by_indexer(registry, agreement).await; } - confirm_cancelled(registry, agreement, tx_hash, config).await + let reason = if row.abandoned { + CancelReason::Abandoned + } else { + CancelReason::NotWanted + }; + confirm_cancelled(registry, agreement, reason, tx_hash, config).await } /// Mark an agreement the indexer ended `CanceledByIndexer` when the chain listener hasn't in @@ -229,7 +237,7 @@ async fn note_check( { Ok(attempts) => { if let Some(err) = failure.filter(|_| failed_attempts > 0) { - log_failed_cancel(agreement, attempts, failed_attempts, err); + log_failed_cancel(row, attempts, failed_attempts, err); } } Err(err) => tracing::warn!( @@ -251,11 +259,12 @@ fn log_uncounted_failure(agreement: &IndexingAgreement, err: &ChainClientError) /// One ERROR as an agreement reaches the limit, for an operator to look into; a WARN for /// every other failed cancel. fn log_failed_cancel( - agreement: &IndexingAgreement, + row: &CancellingAgreement, attempts: u32, failed: u32, err: &ChainClientError, ) { + let agreement = &row.agreement; let reached_limit = attempts >= MAX_CANCEL_ATTEMPTS && attempts.saturating_sub(failed) < MAX_CANCEL_ATTEMPTS; if !reached_limit { @@ -272,6 +281,7 @@ fn log_failed_cancel( agreement_id = %agreement.id, indexer_id = %agreement.indexer.id, indexing_request_id = %agreement.indexing_request_id, + abandoned = row.abandoned, attempts, error = %err, "Cancelling an agreement keeps failing; it may still be live. Dipper now retries it hourly" @@ -497,6 +507,7 @@ mod tests { agreement: cancelling, accepted_on_chain, ended_seen_at: None, + abandoned: false, }], ..MockRegistry::default() } diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index e7e5c76f..4da1555c 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -1086,7 +1086,7 @@ where match prep.item.cancel { Some(CancelKind::ByRequester) => tracing::info!( agreement_id = %prep.agreement.id, - "Agreement marked as CanceledByRequester (on-chain confirmation)" + "Agreement marked as ended by dipper (on-chain confirmation)" ), Some(CancelKind::ByIndexer) => tracing::info!( agreement_id = %prep.agreement.id, @@ -1224,8 +1224,14 @@ where None => return Ok(false), } } - let started = - crate::cancel_dispatch::start_cancel(registry, chain_client, &old_agreement, config).await; + let started = crate::cancel_dispatch::start_cancel( + registry, + chain_client, + &old_agreement, + crate::cancel_dispatch::CancelReason::NotWanted, + config, + ) + .await; Ok(note_replaced_cancel( new_agreement_id, &old_agreement, @@ -1314,8 +1320,14 @@ async fn sweep_orphan_canceled_agreements( }; for agreement in orphans { - let started = - crate::cancel_dispatch::start_cancel(registry, chain_client, &agreement, config).await; + let started = crate::cancel_dispatch::start_cancel( + registry, + chain_client, + &agreement, + crate::cancel_dispatch::CancelReason::NotWanted, + config, + ) + .await; log_orphan_cancel(&agreement, started); } } @@ -2439,13 +2451,6 @@ mod tests { Ok((per_indexer, global)) } - async fn mark_indexing_agreement_as_abandoned( - &self, - _id: &IndexingAgreementId, - ) -> RegistryResult { - Err(crate::registry::Error::NoRecordsUpdated) - } - async fn get_agreement_fee_rates(&self) -> RegistryResult> { Ok(vec![]) } diff --git a/bin/dipper-service/src/network/service/liveness_checker.rs b/bin/dipper-service/src/network/service/liveness_checker.rs index 4c2733a4..e2dfc5d7 100644 --- a/bin/dipper-service/src/network/service/liveness_checker.rs +++ b/bin/dipper-service/src/network/service/liveness_checker.rs @@ -43,8 +43,8 @@ use tokio::{sync::mpsc, time::MissedTickBehavior}; use url::Url; use crate::{ - cancel_dispatch::LiveCancel, - chain_client::{ChainClient, ChainClientError}, + cancel_dispatch::CancelReason, + chain_client::ChainClient, config::LivenessCheckerConfig, network::provider::NetworkProviderService, registry::{ @@ -481,10 +481,9 @@ async fn record_progress( } } -/// Cancel a stale agreement on-chain and queue reassessment. -/// -/// If the on-chain cancel fails, the DB is not updated and reassessment is not -/// queued, leaving the agreement in `AcceptedOnChain` for the next cycle to retry. +/// End a stale agreement and queue a reassessment to replace it. It is marked `Cancelling` +/// before any cancel is sent, so the cancel retry finishes one that fails here, and it ends +/// `AbandonedByIndexer` once the chain confirms dipper's cancel. #[allow(clippy::too_many_arguments)] #[expect( clippy::cognitive_complexity, @@ -504,102 +503,47 @@ async fn cancel_and_reassess( W: WorkerQueue + Send + Sync, C: ChainClient + Send + Sync, { - // 1. Cancel on-chain (mode-aware dispatch) - let mut on_chain_cancel_tx: Option = None; - match crate::cancel_dispatch::cancel_agreement_on_chain(chain_client, agreement, agreement_conf) - .await - { - LiveCancel::Ended(tx_hash) => { - tracing::info!( - agreement_id = %agreement.id, - tx_hash = ?tx_hash, - "canceled stale agreement on-chain" - ); - on_chain_cancel_tx = tx_hash.map(|hash| hash.to_string()); - } - LiveCancel::NotLive { .. } => { - tracing::info!( - agreement_id = %agreement.id, - "stale agreement already canceled on-chain; proceeding to mark abandoned" - ); - } - LiveCancel::CancelFailed(err @ ChainClientError::MissingTermsVersionHash { .. }) => { - // Permanent per-agreement condition: the on-chain agreement is - // still live, so do NOT mark abandoned (that would hide a - // money-draining agreement). Surface for operator action. - tracing::error!( - agreement_id = %agreement.id, - error = %err, - "cannot cancel stale agreement: missing terms_version_hash; leaving active for operator action" - ); - return; - } - LiveCancel::CancelFailed(ChainClientError::ConfigError(_)) => { - // Chain client disabled: still proceed to mark and reassess so the - // DB reflects the detected abandonment even without an on-chain tx. - tracing::warn!( - agreement_id = %agreement.id, - "chain client not configured, skipping on-chain cancellation" - ); - } - LiveCancel::Unconfirmed { tx_hash, err } => { - crate::cancel_dispatch::log_unconfirmed(agreement, tx_hash, &err); - return; - } - LiveCancel::ReadFailed(err) | LiveCancel::CancelFailed(err) => { - tracing::error!( - agreement_id = %agreement.id, - error = %err, - "failed to cancel stale agreement on-chain, will retry next cycle" - ); - return; - } + // Without a cancel that can ever be sent, replacing it would pay 2 indexers until an + // operator ends it, so it stays as it is, active, for one to deal with. + if crate::cancel_dispatch::cancel_hash(agreement).is_none() { + tracing::error!( + agreement_id = %agreement.id, + "cannot cancel stale agreement: missing terms_version_hash; leaving active for operator action" + ); + return; } - // 2. Mark as abandoned in DB - let abandoned = match tokio::time::timeout( - db_timeout, - registry.mark_indexing_agreement_as_abandoned(&agreement.id), + // 1. Start the cancel + match crate::cancel_dispatch::start_cancel( + registry, + chain_client, + agreement, + CancelReason::Abandoned, + agreement_conf, ) .await { - Ok(Ok(a)) => a, - Ok(Err(err)) => { - tracing::error!( + Ok(started) => tracing::info!( + agreement_id = %agreement.id, + ?started, + reason = "indexer_stale", + "Cancelling stale agreement" + ), + Err(crate::registry::Error::NoRecordsUpdated) => { + tracing::debug!( agreement_id = %agreement.id, - error = %err, - "failed to mark agreement as abandoned" + "Stale agreement already ended or being cancelled" ); return; } - Err(_) => { + Err(err) => { tracing::error!( agreement_id = %agreement.id, - "timeout marking agreement as abandoned" + error = %err, + "failed to mark stale agreement cancelling, will retry next cycle" ); return; } - }; - - // The accepted agreement was canceled on-chain because the indexer went - // stale. Record the cancel audit so the chain_listener's `terminated` sweep - // announces it durably: this row is marked `AbandonedByIndexer` (terminal) - // and was accepted on-chain, so it is sweep-eligible. - let manager = agreement_conf.recurring_agreement_manager().to_string(); - if let Err(err) = registry - .record_cancel_audit( - &agreement.id, - dipper_core::time::now_secs(), - &manager, - on_chain_cancel_tx.as_deref(), - ) - .await - { - tracing::warn!( - agreement_id = %agreement.id, - error = %err, - "failed to record cancel audit; terminated event may emit with fallback fields" - ); } // Clean up pending cancellations: if this abandoned agreement was a @@ -615,10 +559,10 @@ async fn cancel_and_reassess( ); } - // 3. Fetch the indexing request for num_candidates + // 2. Fetch the indexing request for num_candidates let request = match tokio::time::timeout( db_timeout, - registry.get_indexing_request_by_id(&abandoned.indexing_request_id), + registry.get_indexing_request_by_id(&agreement.indexing_request_id), ) .await { @@ -626,7 +570,7 @@ async fn cancel_and_reassess( Ok(Ok(None)) => { tracing::warn!( agreement_id = %agreement.id, - indexing_request_id = %abandoned.indexing_request_id, + indexing_request_id = %agreement.indexing_request_id, "indexing request not found for abandoned agreement" ); return; @@ -648,13 +592,13 @@ async fn cancel_and_reassess( } }; - // 4. Queue reassessment + // 3. Queue reassessment let push_result = tokio::time::timeout( queue_timeout, worker_queue.reassess_indexing_request( - abandoned.indexing_request_id, - abandoned.terms.metadata.subgraph_deployment_id, - abandoned.terms.metadata.chain_id, + agreement.indexing_request_id, + agreement.terms.metadata.subgraph_deployment_id, + agreement.terms.metadata.chain_id, request.num_candidates, // Background: abandonment remediation yields to interactive work. JobPriority::Background, @@ -666,7 +610,7 @@ async fn cancel_and_reassess( Ok(Ok(_job_id)) => { tracing::info!( agreement_id = %agreement.id, - indexing_request_id = %abandoned.indexing_request_id, + indexing_request_id = %agreement.indexing_request_id, "queued reassessment for abandoned agreement" ); } @@ -920,7 +864,10 @@ mod tests { #[derive(Clone, Default)] struct MockCalls { progress_updates: Arc>>, - abandoned: Arc>>, + /// Ids marked cancelling as abandoned, before any cancel is sent. + abandoning: Arc>>, + /// Ids marked ended once the chain confirmed dipper's cancel. + ended: Arc>>, reassessments: Arc>>, chain_cancels: Arc>>, /// Ids passed to `record_cancel_audit` -- the signal the handler drives @@ -930,30 +877,19 @@ mod tests { struct MockRegistry { calls: MockCalls, - mark_abandoned_result: Arc>>>, + already_ending: bool, get_request_result: Arc>>>>, } impl MockRegistry { fn new(calls: MockCalls, agreement: IndexingAgreement) -> Self { - let abandoned_agreement = { - let mut a = agreement.clone(); - a.status = IndexingAgreementStatus::AbandonedByIndexer; - a - }; let request = make_request(agreement.indexing_request_id, 2); Self { calls, - mark_abandoned_result: Arc::new(Mutex::new(Some(Ok(abandoned_agreement)))), + already_ending: false, get_request_result: Arc::new(Mutex::new(Some(Ok(Some(request))))), } } - - fn with_chain_error(calls: MockCalls, agreement: IndexingAgreement) -> Self { - let mut mock = Self::new(calls, agreement); - mock.mark_abandoned_result = Arc::new(Mutex::new(None)); - mock - } } #[async_trait] @@ -978,16 +914,23 @@ mod tests { .push((*id, block_height)); Ok(()) } - async fn mark_indexing_agreement_as_abandoned( + async fn mark_indexing_agreement_as_abandoning( &self, id: &IndexingAgreementId, - ) -> RegistryResult { - self.calls.abandoned.lock().unwrap().push(*id); - self.mark_abandoned_result - .lock() - .unwrap() - .take() - .expect("mark_abandoned called more than once") + ) -> RegistryResult<()> { + if self.already_ending { + return Err(crate::registry::Error::NoRecordsUpdated); + } + self.calls.abandoning.lock().unwrap().push(*id); + Ok(()) + } + + async fn mark_indexing_agreement_as_canceled_by_requester( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()> { + self.calls.ended.lock().unwrap().push(*id); + Ok(()) } async fn record_cancel_audit( @@ -1116,9 +1059,11 @@ mod tests { } } + /// An agreement live on-chain until a cancel that succeeds ends it. struct MockChainClient { calls: MockCalls, result: Result, + live: std::sync::atomic::AtomicBool, } impl MockChainClient { @@ -1126,13 +1071,7 @@ mod tests { Self { calls, result: Ok(B256::ZERO), - } - } - - fn config_error(calls: MockCalls) -> Self { - Self { - calls, - result: Err(ChainClientError::ConfigError("disabled".into())), + live: true.into(), } } @@ -1140,6 +1079,7 @@ mod tests { Self { calls, result: Err(ChainClientError::RpcError(anyhow::anyhow!("network error"))), + live: true.into(), } } } @@ -1172,12 +1112,9 @@ mod tests { // existing cancel-path assertions hold. self.calls.chain_cancels.lock().unwrap().push(*agreement_id); match &self.result { - Ok(hash) => Ok(Some(*hash)), - Err(ChainClientError::ConfigError(s)) => { - Err(ChainClientError::ConfigError(s.clone())) - } - Err(ChainClientError::RpcError(e)) => { - Err(ChainClientError::RpcError(anyhow::anyhow!("{e}"))) + Ok(hash) => { + self.live.store(false, std::sync::atomic::Ordering::SeqCst); + Ok(Some(*hash)) } Err(e) => Err(ChainClientError::RpcError(anyhow::anyhow!("{e}"))), } @@ -1203,9 +1140,9 @@ mod tests { &self, _agreement_id: &[u8; 16], ) -> Result { - // Cancel dispatch reads back after a mined cancel; reporting - // not-active means "cancel confirmed", which these tests expect. - Ok(AgreementOnChain::NotLive) + Ok(AgreementOnChain::live_if( + self.live.load(std::sync::atomic::Ordering::SeqCst), + )) } } @@ -1450,183 +1387,115 @@ mod tests { // ---- cancel_and_reassess behavior tests ---- - #[tokio::test] - async fn test_cancel_and_reassess_success() { - // Arrange + fn stale_agreement() -> IndexingAgreement { let dep: DeploymentId = "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" .parse() .unwrap(); - let indexer_id = IndexerId::from(Address::ZERO); let url: Url = "http://indexer.example.com/".parse().unwrap(); - let agreement = make_agreement( - indexer_id, + make_agreement( + IndexerId::from(Address::ZERO), url, dep, Some(100), Some(OffsetDateTime::now_utc()), - ); - let req_id = agreement.indexing_request_id; - let agr_id = agreement.id; + ) + } - let calls = MockCalls::default(); - let registry = MockRegistry::new(calls.clone(), agreement.clone()); + async fn end_stale( + agreement: &IndexingAgreement, + registry: &MockRegistry, + chain: &MockChainClient, + ) { let queue = MockWorkerQueue { - calls: calls.clone(), + calls: registry.calls.clone(), }; - let chain = MockChainClient::success(calls.clone()); - - // Act cancel_and_reassess( - &agreement, - ®istry, + agreement, + registry, &queue, - &chain, + chain, &test_agreement_conf(), DB_TIMEOUT, QUEUE_TIMEOUT, ) .await; - - // Assert - assert_eq!( - calls.chain_cancels.lock().unwrap().as_slice(), - &[agreement.id.into_bytes()] - ); - assert_eq!(calls.abandoned.lock().unwrap().as_slice(), &[agr_id]); - assert_eq!(calls.reassessments.lock().unwrap().as_slice(), &[req_id]); } #[tokio::test] - async fn cancel_and_reassess_records_cancel_audit() { - // The stale-agreement cancel no longer emits `terminated` directly: it - // records the cancel audit and the chain_listener sweep announces it. - let dep: DeploymentId = "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" - .parse() - .unwrap(); - let indexer_id = IndexerId::from(Address::ZERO); - let url: Url = "http://indexer.example.com/".parse().unwrap(); - let agreement = make_agreement( - indexer_id, - url, - dep, - Some(100), - Some(OffsetDateTime::now_utc()), - ); - let agr_id = agreement.id; - + async fn ends_a_stale_agreement_through_cancelling_and_reassesses() { + let agreement = stale_agreement(); let calls = MockCalls::default(); let registry = MockRegistry::new(calls.clone(), agreement.clone()); - let queue = MockWorkerQueue { - calls: calls.clone(), - }; let chain = MockChainClient::success(calls.clone()); - cancel_and_reassess( - &agreement, - ®istry, - &queue, - &chain, - &test_agreement_conf(), - DB_TIMEOUT, - QUEUE_TIMEOUT, - ) - .await; + end_stale(&agreement, ®istry, &chain).await; + assert_eq!(calls.abandoning.lock().unwrap().as_slice(), &[agreement.id]); + assert_eq!( + calls.chain_cancels.lock().unwrap().as_slice(), + &[agreement.id.into_bytes()] + ); + assert_eq!(calls.ended.lock().unwrap().as_slice(), &[agreement.id]); assert_eq!( calls.cancel_audits.lock().unwrap().as_slice(), - &[agr_id], - "exactly one cancel audit recorded for the stale agreement" + &[agreement.id], + "its cancel is recorded, so the terminated sweep announces it" + ); + assert_eq!( + calls.reassessments.lock().unwrap().as_slice(), + &[agreement.indexing_request_id] ); } #[tokio::test] - async fn test_cancel_and_reassess_config_error_proceeds() { - // Arrange: chain client disabled (ConfigError) → still mark abandoned and reassess - let dep: DeploymentId = "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" - .parse() - .unwrap(); - let indexer_id = IndexerId::from(Address::ZERO); - let url: Url = "http://indexer.example.com/".parse().unwrap(); - let agreement = make_agreement( - indexer_id, - url, - dep, - Some(100), - Some(OffsetDateTime::now_utc()), - ); - let req_id = agreement.indexing_request_id; - let agr_id = agreement.id; - + async fn leaves_a_failed_cancel_to_the_retry_and_still_reassesses() { + // Marked cancelling, the agreement is out of the checker's sight, and its indexer out + // of selection until the cancel lands, so it is replaced now and the retry finishes it. + let agreement = stale_agreement(); let calls = MockCalls::default(); let registry = MockRegistry::new(calls.clone(), agreement.clone()); - let queue = MockWorkerQueue { - calls: calls.clone(), - }; - let chain = MockChainClient::config_error(calls.clone()); + let chain = MockChainClient::rpc_error(calls.clone()); - // Act - cancel_and_reassess( - &agreement, - ®istry, - &queue, - &chain, - &test_agreement_conf(), - DB_TIMEOUT, - QUEUE_TIMEOUT, - ) - .await; + end_stale(&agreement, ®istry, &chain).await; - // Assert: no on-chain cancel (ConfigError is treated as disabled, not a real error) - // but DB mark and reassessment still happen + assert_eq!(calls.abandoning.lock().unwrap().as_slice(), &[agreement.id]); + assert!(calls.ended.lock().unwrap().is_empty()); + assert!(calls.cancel_audits.lock().unwrap().is_empty()); assert_eq!( - calls.chain_cancels.lock().unwrap().as_slice(), - &[agreement.id.into_bytes()] + calls.reassessments.lock().unwrap().as_slice(), + &[agreement.indexing_request_id] ); - assert_eq!(calls.abandoned.lock().unwrap().as_slice(), &[agr_id]); - assert_eq!(calls.reassessments.lock().unwrap().as_slice(), &[req_id]); } #[tokio::test] - async fn test_cancel_and_reassess_chain_error_skips() { - // Arrange: transient RPC error → do nothing (retry next cycle) - let dep: DeploymentId = "QmTXzATwNfgGVukV1fX2T6xw9f6LAYRVWpsdXyRWzUR2H9" - .parse() - .unwrap(); - let indexer_id = IndexerId::from(Address::ZERO); - let url: Url = "http://indexer.example.com/".parse().unwrap(); - let agreement = make_agreement( - indexer_id, - url, - dep, - Some(100), - Some(OffsetDateTime::now_utc()), - ); + async fn leaves_a_stale_agreement_it_can_never_cancel_for_an_operator() { + // Replacing it would pay both indexers, since its cancel can never be sent. + let mut agreement = stale_agreement(); + agreement.terms_version_hash = None; + let calls = MockCalls::default(); + let registry = MockRegistry::new(calls.clone(), agreement.clone()); + let chain = MockChainClient::success(calls.clone()); + + end_stale(&agreement, ®istry, &chain).await; + + assert!(calls.abandoning.lock().unwrap().is_empty()); + assert!(calls.chain_cancels.lock().unwrap().is_empty()); + assert!(calls.reassessments.lock().unwrap().is_empty()); + } + #[tokio::test] + async fn leaves_an_agreement_already_ending_alone() { + let agreement = stale_agreement(); let calls = MockCalls::default(); - let registry = MockRegistry::with_chain_error(calls.clone(), agreement.clone()); - let queue = MockWorkerQueue { - calls: calls.clone(), + let registry = MockRegistry { + already_ending: true, + ..MockRegistry::new(calls.clone(), agreement.clone()) }; - let chain = MockChainClient::rpc_error(calls.clone()); + let chain = MockChainClient::success(calls.clone()); - // Act - cancel_and_reassess( - &agreement, - ®istry, - &queue, - &chain, - &test_agreement_conf(), - DB_TIMEOUT, - QUEUE_TIMEOUT, - ) - .await; + end_stale(&agreement, ®istry, &chain).await; - // Assert: chain cancel attempted, but DB and queue untouched - assert_eq!( - calls.chain_cancels.lock().unwrap().as_slice(), - &[agreement.id.into_bytes()] - ); - assert!(calls.abandoned.lock().unwrap().is_empty()); + assert!(calls.chain_cancels.lock().unwrap().is_empty()); assert!(calls.reassessments.lock().unwrap().is_empty()); } diff --git a/bin/dipper-service/src/registry.rs b/bin/dipper-service/src/registry.rs index 89c96724..239ff563 100644 --- a/bin/dipper-service/src/registry.rs +++ b/bin/dipper-service/src/registry.rs @@ -401,6 +401,16 @@ impl AgreementRegistry for RegistryProvider { .map_err(Into::into) } + async fn mark_indexing_agreement_as_abandoning( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()> { + self.inner + .mark_indexing_agreement_as_abandoning(id) + .await + .map_err(Into::into) + } + async fn reopen_indexing_agreement_cancel( &self, id: &IndexingAgreementId, @@ -716,17 +726,6 @@ impl AgreementRegistry for RegistryProvider { .map_err(Into::into) } - async fn mark_indexing_agreement_as_abandoned( - &self, - id: &IndexingAgreementId, - ) -> RegistryResult { - let raw = self.inner.mark_indexing_agreement_as_abandoned(id).await?; - // The conversion only fails for Unknown status; since we just wrote - // AbandonedByIndexer, this cannot fail in practice. - IndexingAgreement::try_from(raw) - .map_err(|_| dipper_pgregistry::Error::NoRecordsUpdated.into()) - } - async fn get_agreement_fee_rates(&self) -> RegistryResult> { self.inner .get_agreement_fee_rates() diff --git a/bin/dipper-service/src/registry/agreement.rs b/bin/dipper-service/src/registry/agreement.rs index 3aeb452e..54ac69f0 100644 --- a/bin/dipper-service/src/registry/agreement.rs +++ b/bin/dipper-service/src/registry/agreement.rs @@ -260,7 +260,8 @@ pub trait AgreementRegistry { /// /// If there is no indexing agreement with the given ID, or if the agreement is not in the /// `CREATED`, `ACCEPTED_ON_CHAIN`, `REJECTED` or `CANCELLING` state, this method returns a - /// [`NoRecordUpdated`](Error::NoRecordsUpdated) error. + /// [`NoRecordUpdated`](Error::NoRecordsUpdated) error. One dipper was cancelling because its + /// indexer stopped serving it becomes `ABANDONED_BY_INDEXER` instead. async fn mark_indexing_agreement_as_canceled_by_requester( &self, id: &IndexingAgreementId, @@ -273,6 +274,14 @@ pub trait AgreementRegistry { id: &IndexingAgreementId, ) -> RegistryResult<()>; + /// Mark an `ACCEPTED_ON_CHAIN` agreement whose indexer stopped serving it `CANCELLING`, + /// before its cancel is sent, so it ends `ABANDONED_BY_INDEXER` once the chain confirms it; + /// [`NoRecordUpdated`](Error::NoRecordsUpdated) otherwise. + async fn mark_indexing_agreement_as_abandoning( + &self, + id: &IndexingAgreementId, + ) -> RegistryResult<()>; + /// Move a `CANCELED_BY_REQUESTER` or `REJECTED` agreement the chain shows live back to /// `CANCELLING`, its cancel attempts reset; [`NoRecordUpdated`](Error::NoRecordsUpdated) /// otherwise. @@ -527,17 +536,6 @@ pub trait AgreementRegistry { .map(|m| !m.is_empty()) } - /// Mark an indexing agreement as `ABANDONED_BY_INDEXER`. - /// - /// Transitions `AcceptedOnChain → AbandonedByIndexer`. Returns the full agreement - /// for use in the subsequent reassessment call. - /// Returns [`NoRecordsUpdated`](Error::NoRecordsUpdated) if the agreement doesn't - /// exist or isn't in `AcceptedOnChain` status. - async fn mark_indexing_agreement_as_abandoned( - &self, - id: &IndexingAgreementId, - ) -> RegistryResult; - /// Get per-agreement rate fields from active agreements. /// /// Returns base rate, entity rate, and deployment ID for each active @@ -566,6 +564,9 @@ pub struct CancellingAgreement { pub accepted_on_chain: bool, /// When a check first found it no longer live on-chain, if one has. pub ended_seen_at: Option, + /// Whether it is being cancelled because its indexer stopped serving it, so it ends + /// `ABANDONED_BY_INDEXER`. + pub abandoned: bool, } impl TryFrom for CancellingAgreement { @@ -576,6 +577,7 @@ impl TryFrom for CancellingAgreement { agreement: value.agreement.try_into()?, accepted_on_chain: value.accepted_on_chain, ended_seen_at: value.ended_seen_at, + abandoned: value.abandoned, }) } } @@ -742,15 +744,15 @@ pub enum Status { /// The liveness checker detected no indexing progress within the tolerance window. /// - /// Dipper canceled the agreement via `cancelIndexingAgreementByPayer` and will - /// trigger reassignment to find a replacement indexer. + /// Dipper cancelled the agreement on-chain, passing through `Cancelling` until the chain + /// confirmed it, and triggered reassignment to find a replacement indexer. /// /// This is a terminal state. AbandonedByIndexer, /// Dipper decided to end the agreement and is cancelling it on-chain, where it may - /// still be live. It becomes `CanceledByRequester`, announced as ended, only once - /// the chain confirms the end. + /// still be live. It becomes `CanceledByRequester`, or `AbandonedByIndexer` when its + /// indexer stopped serving it, announced as ended, only once the chain confirms the end. Cancelling, } diff --git a/bin/dipper-service/src/registry/agreement_stub.rs b/bin/dipper-service/src/registry/agreement_stub.rs index aa91b431..cbb62dd6 100644 --- a/bin/dipper-service/src/registry/agreement_stub.rs +++ b/bin/dipper-service/src/registry/agreement_stub.rs @@ -137,6 +137,10 @@ pub trait StubAgreementRegistry: Send + Sync { unimplemented!("mark_indexing_agreement_as_cancelling") } + async fn mark_indexing_agreement_as_abandoning(&self, _id: &IndexingAgreementId) -> Result<()> { + unimplemented!("mark_indexing_agreement_as_abandoning") + } + async fn reopen_indexing_agreement_cancel(&self, _id: &IndexingAgreementId) -> Result<()> { unimplemented!("reopen_indexing_agreement_cancel") } @@ -241,13 +245,6 @@ pub trait StubAgreementRegistry: Send + Sync { .map(|m| !m.is_empty()) } - async fn mark_indexing_agreement_as_abandoned( - &self, - _id: &IndexingAgreementId, - ) -> Result { - unimplemented!("mark_indexing_agreement_as_abandoned") - } - async fn get_agreement_fee_rates(&self) -> Result> { unimplemented!("get_agreement_fee_rates") } @@ -445,6 +442,10 @@ impl AgreementRegistry for T { StubAgreementRegistry::mark_indexing_agreement_as_cancelling(self, id).await } + async fn mark_indexing_agreement_as_abandoning(&self, id: &IndexingAgreementId) -> Result<()> { + StubAgreementRegistry::mark_indexing_agreement_as_abandoning(self, id).await + } + async fn reopen_indexing_agreement_cancel(&self, id: &IndexingAgreementId) -> Result<()> { StubAgreementRegistry::reopen_indexing_agreement_cancel(self, id).await } @@ -546,13 +547,6 @@ impl AgreementRegistry for T { StubAgreementRegistry::exists_active_agreements(self).await } - async fn mark_indexing_agreement_as_abandoned( - &self, - id: &IndexingAgreementId, - ) -> Result { - StubAgreementRegistry::mark_indexing_agreement_as_abandoned(self, id).await - } - async fn get_agreement_fee_rates(&self) -> Result> { StubAgreementRegistry::get_agreement_fee_rates(self).await } diff --git a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs index 1e486ff1..9fcc916b 100644 --- a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs +++ b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs @@ -1901,6 +1901,7 @@ where &ctx.registry, &ctx.chain_client, agreement, + crate::cancel_dispatch::CancelReason::NotWanted, &ctx.agreement_conf, ) .await diff --git a/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs b/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs index 1022faff..e3e6fbc4 100644 --- a/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs +++ b/bin/dipper-service/src/worker/handlers/send_indexing_agreement_proposal.rs @@ -607,13 +607,6 @@ mod tests { Ok((std::collections::HashMap::new(), 0)) } - async fn mark_indexing_agreement_as_abandoned( - &self, - _id: &IndexingAgreementId, - ) -> crate::registry::Result { - Err(crate::registry::Error::NoRecordsUpdated) - } - async fn get_agreement_fee_rates(&self) -> crate::registry::Result> { Ok(vec![]) } diff --git a/dipper-pgregistry/migrations/20261005000001_add_abandoned_flag.sql b/dipper-pgregistry/migrations/20261005000001_add_abandoned_flag.sql new file mode 100644 index 00000000..87fd09c3 --- /dev/null +++ b/dipper-pgregistry/migrations/20261005000001_add_abandoned_flag.sql @@ -0,0 +1,5 @@ +-- abandoned: dipper is cancelling the agreement because its indexer stopped serving it, so once +-- the chain confirms dipper's cancel it ends AbandonedByIndexer (8) rather than +-- CanceledByRequester (3). +ALTER TABLE dipper_reg_indexing_agreements + ADD COLUMN abandoned BOOLEAN NOT NULL DEFAULT false; diff --git a/dipper-pgregistry/src/indexing_agreement.rs b/dipper-pgregistry/src/indexing_agreement.rs index 4a3b24fe..241f8419 100644 --- a/dipper-pgregistry/src/indexing_agreement.rs +++ b/dipper-pgregistry/src/indexing_agreement.rs @@ -200,15 +200,15 @@ pub enum Status { /// The liveness checker detected no indexing progress within the tolerance window. /// - /// Dipper canceled the agreement via `cancelIndexingAgreementByPayer` and will - /// trigger reassignment to find a replacement indexer. + /// Dipper cancelled the agreement on-chain, passing through `Cancelling` until the chain + /// confirmed it, and triggered reassignment to find a replacement indexer. /// /// This is a terminal state. AbandonedByIndexer = 8, /// Dipper decided to end the agreement and is cancelling it on-chain, where it may - /// still be live. It becomes `CanceledByRequester`, announced as ended, only once - /// the chain confirms the end. + /// still be live. It becomes `CanceledByRequester`, or `AbandonedByIndexer` when its + /// indexer stopped serving it, announced as ended, only once the chain confirms the end. Cancelling = 9, /// A fallback for unknown status values. diff --git a/dipper-pgregistry/src/postgres.rs b/dipper-pgregistry/src/postgres.rs index 1d7ebe1c..2a553614 100644 --- a/dipper-pgregistry/src/postgres.rs +++ b/dipper-pgregistry/src/postgres.rs @@ -195,6 +195,8 @@ pub struct CancellingAgreement { pub accepted_on_chain: bool, /// When a check first found it no longer live on-chain, if one has. pub ended_seen_at: Option, + /// Whether it is being cancelled because its indexer stopped serving it. + pub abandoned: bool, } impl sqlx::FromRow<'_, sqlx::postgres::PgRow> for CancellingAgreement { @@ -205,6 +207,7 @@ impl sqlx::FromRow<'_, sqlx::postgres::PgRow> for CancellingAgreement { agreement: IndexingAgreement::from_row(row)?, accepted_on_chain: accepted_at.is_some(), ended_seen_at: row.try_get("ended_seen_at")?, + abandoned: row.try_get("abandoned")?, }) } } @@ -943,6 +946,7 @@ impl PgRegistry { Ok(()) } + /// One being cancelled because its indexer stopped serving it ends `AbandonedByIndexer`. pub async fn mark_indexing_agreement_as_canceled_by_requester( &self, agreement_id: &IndexingAgreementId, @@ -974,6 +978,33 @@ impl PgRegistry { .await } + /// Start ending an accepted agreement whose indexer stopped serving it: `Cancelling`, and + /// noted as abandoned, so the chain confirming dipper's cancel ends it `AbandonedByIndexer`. + pub async fn mark_indexing_agreement_as_abandoning( + &self, + agreement_id: &IndexingAgreementId, + ) -> Result<(), Error> { + let updated = sqlx::query( + r#" + UPDATE dipper_reg_indexing_agreements + SET + status = $1, + abandoned = true, + updated_at = timezone('UTC', now()) + WHERE id = $2 AND status = $3 + "#, + ) + .bind(IndexingAgreementStatus::Cancelling) + .bind(agreement_id) + .bind(IndexingAgreementStatus::AcceptedOnChain) + .execute(&self.pool) + .await?; + if updated.rows_affected() == 0 { + return Err(Error::NoRecordsUpdated); + } + Ok(()) + } + /// Move an agreement dipper had already ended, cancelled or rejected, back to `Cancelling` /// once the chain shows it live after all, with its cancel attempts started afresh. It counts /// as checked, since no cancel is sent with it, so the retry takes it on its next sweep rather @@ -1036,7 +1067,8 @@ impl PgRegistry { rejection_reason, terms_version_hash, accepted_at, - ended_seen_at + ended_seen_at, + abandoned FROM dipper_reg_indexing_agreements WHERE status = $1 AND ( @@ -1933,50 +1965,6 @@ impl PgRegistry { Ok(exists) } - /// Mark an agreement as `AbandonedByIndexer`. - /// - /// Transitions `AcceptedOnChain → AbandonedByIndexer`. Returns the full - /// agreement for use in the subsequent reassessment call. - /// - /// Returns [`NoRecordsUpdated`](Error::NoRecordsUpdated) if the agreement - /// doesn't exist or isn't in `AcceptedOnChain` status. - pub async fn mark_indexing_agreement_as_abandoned( - &self, - agreement_id: &IndexingAgreementId, - ) -> Result { - let record: Option = sqlx::query_as( - r#" - UPDATE dipper_reg_indexing_agreements - SET - status = $1, - updated_at = timezone('UTC', now()) - WHERE id = $2 AND status = $3 - RETURNING - id, - nonce_uuid, - created_at, - updated_at, - status, - indexing_request_id, - deployment_id, - indexer_id, - indexer_url, - terms, - last_block_height, - last_progress_at, - rejection_reason, - terms_version_hash - "#, - ) - .bind(IndexingAgreementStatus::AbandonedByIndexer) - .bind(agreement_id) - .bind(IndexingAgreementStatus::AcceptedOnChain) - .fetch_optional(&self.pool) - .await?; - - record.ok_or(Error::NoRecordsUpdated) - } - // ========================================================================= // Indexer denylist operations // ========================================================================= @@ -2295,7 +2283,8 @@ impl PgRegistry { /// Batched form of `update_status_from`: transitions all rows whose `id` /// is in `agreement_ids` and whose current status is in `allowed_from` to -/// `new_status`, in one statement. Returns the ids of the rows that +/// `new_status`, in one statement; a row noted abandoned that `new_status` would make +/// `CanceledByRequester` becomes `AbandonedByIndexer` instead. Returns the ids of the rows that /// actually flipped (matched the CAS guard) so callers can build per-id /// outcome maps. Empty input is a fast-path no-op. async fn batch_update_status_from( @@ -2308,20 +2297,25 @@ async fn batch_update_status_from( return Ok(Vec::new()); } let placeholders = (0..allowed_from.len()) - .map(|i| format!("${}", i + 3)) + .map(|i| format!("${}", i + 5)) .collect::>() .join(", "); + // An agreement dipper ended because its indexer stopped serving it ends as abandoned. let sql = format!( r#" UPDATE dipper_reg_indexing_agreements - SET status = $1, updated_at = timezone('UTC', now()) + SET + status = CASE WHEN abandoned AND $1 = $3 THEN $4 ELSE $1 END, + updated_at = timezone('UTC', now()) WHERE id = ANY($2) AND status IN ({placeholders}) RETURNING id "# ); let mut query = sqlx::query_as::<_, (IndexingAgreementId,)>(&sql) .bind(new_status) - .bind(agreement_ids); + .bind(agreement_ids) + .bind(IndexingAgreementStatus::CanceledByRequester) + .bind(IndexingAgreementStatus::AbandonedByIndexer); for status in allowed_from { query = query.bind(*status); } diff --git a/dipper-pgregistry/tests/it_registry_postgres.rs b/dipper-pgregistry/tests/it_registry_postgres.rs index e9de6789..c0b5de0f 100644 --- a/dipper-pgregistry/tests/it_registry_postgres.rs +++ b/dipper-pgregistry/tests/it_registry_postgres.rs @@ -1768,43 +1768,76 @@ async fn test_count_active_agreements_by_deployment() { ); } -#[tokio::test] -async fn test_mark_as_abandoned_transitions_status() { - //* Given +/// Start abandoning fixture 0002's accepted agreement, then end it the way `end` does. +#[expect( + clippy::expect_used, + reason = "a test helper fails its test on any error" +)] +async fn abandon_then_end(end: F) -> (Result<(), Error>, IndexingAgreementStatus) +where + F: AsyncFnOnce(&PgRegistry, &IndexingAgreementId), +{ let (db, _temp_db) = temp_registry_db().await; run_fixture(&db, include_str!("fixtures/0002_indexing_agreements.sql")) .await .expect("Failed to run fixture"); let registry = PgRegistry::new(db); - - // AcceptedOnChain agreement from fixture 0002 let agreement_id = IndexingAgreementId::from_bytes([0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 3]); - - //* When - let abandoned = registry - .mark_indexing_agreement_as_abandoned(&agreement_id) + registry + .mark_indexing_agreement_as_abandoning(&agreement_id) + .await + .expect("an accepted agreement can be abandoned"); + let listed = registry + .get_cancelling_agreements(100, 10, 0) .await - .expect("Failed to mark agreement as abandoned"); + .expect("cancelling query"); + assert!(listed[0].abandoned); - //* Then - assert_eq!( - abandoned.status, - IndexingAgreementStatus::AbandonedByIndexer, - "Status should be AbandonedByIndexer" - ); + end(®istry, &agreement_id).await; - // Second call must fail — agreement is no longer AcceptedOnChain - let err = registry - .mark_indexing_agreement_as_abandoned(&agreement_id) + let again = registry + .mark_indexing_agreement_as_abandoning(&agreement_id) + .await; + let ended = registry + .get_indexing_agreement_by_id(&agreement_id) .await - .expect_err("Expected error on second mark_as_abandoned call"); + .expect("agreement query") + .expect("agreement"); + (again, ended.status) +} + +#[tokio::test] +async fn an_abandoned_agreement_dipper_cancels_ends_abandoned() { + let (again, status) = abandon_then_end(async |registry, id| { + registry + .mark_indexing_agreement_as_canceled_by_requester(id) + .await + .expect("dipper's cancel confirmed"); + }) + .await; + + assert_eq!(status, IndexingAgreementStatus::AbandonedByIndexer); assert!( - matches!(err, Error::NoRecordsUpdated), - "Expected NoRecordsUpdated, got: {err:?}" + matches!(again, Err(Error::NoRecordsUpdated)), + "got {again:?}" ); } +#[tokio::test] +async fn an_abandoned_agreement_the_listener_sees_cancelled_ends_abandoned() { + let (_, status) = abandon_then_end(async |registry, id| { + let outcome = registry + .apply_reconciliation(id, false, Some(CancelKind::ByRequester)) + .await + .expect("reconciliation"); + assert!(outcome.did_cancel); + }) + .await; + + assert_eq!(status, IndexingAgreementStatus::AbandonedByIndexer); +} + // ============================================================================= // Rejection reason storage tests // ============================================================================= From 768f42652be1f06054163d3bf9570421af61668b Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Mon, 5 Oct 2026 12:10:53 +0100 Subject: [PATCH 17/26] fix: record reopened accepts and quiet expected races (#730) * fix(listener): record the accept of an agreement it reopens When the listener found an agreement dipper had cancelled live after all and moved it back to cancelling, it then checked the old status and skipped recording the accept, so the cancel retry treated a paying agreement as an unaccepted offer. The accept is now recorded. * fix(cancel): log an agreement that ended while being marked as expected An agreement that ended, or started cancelling, between being listed and being marked was logged as a failure, with an ERROR and a failure count in reassessment. That race is expected and needs nothing more, so it is now logged at debug level and not counted. * test(cancel): remove test helpers that nothing reads or writes A list of queued cancels that was never filled made its assertion always pass, and a way to mark an agreement already cancelled on-chain was never read, so tests using it passed for another reason. Both are gone, along with a comment naming code that no longer exists. * docs(listener): say plainly what reopening an agreement returns The comment ended with "True if it was" straight after "unless the chain shows it already ended", which read as the opposite of what it returns. --- .../src/network/service/chain_listener.rs | 49 ++++---- .../handlers/reassess_indexing_request.rs | 116 +++++++++++------- 2 files changed, 100 insertions(+), 65 deletions(-) diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index 4da1555c..64947665 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -881,8 +881,9 @@ where } } - reopen_if_cancelled_but_accepted(snapshot, &agreement, registry, chain_client).await?; - record_accept_of_cancelling(snapshot, &agreement, registry).await; + let reopened = + reopen_if_cancelled_but_accepted(snapshot, &agreement, registry, chain_client).await?; + record_accept_of_cancelling(snapshot, &agreement, reopened, registry).await; // Both transitions are applied atomically downstream so the // Accept-then-Cancel-in-one-snapshot path can't leak an intermediate @@ -998,13 +999,14 @@ fn created_after_events_started(agreement: &IndexingAgreement) -> bool { /// Safety net for an agreement dipper cancelled whose offer the indexer accepted /// anyway, such as one that landed after dipper's cancel. Nothing else would end /// it: reconciliation ignores an accept on a cancelled row. It goes back to -/// `Cancelling` for the cancel retry, unless the chain shows it already ended. +/// `Cancelling` for the cancel retry, unless the chain shows it already ended. Returns whether it +/// moved it back. async fn reopen_if_cancelled_but_accepted( snapshot: &AgreementStateSnapshot, agreement: &IndexingAgreement, registry: &R, chain_client: &T, -) -> anyhow::Result<()> +) -> anyhow::Result where R: AgreementRegistry + Sync, T: ChainClient, @@ -1013,20 +1015,24 @@ where && snapshot.state.reached_accepted() && !snapshot.state.is_canceled() { - crate::cancel_dispatch::reopen_if_live(registry, chain_client, agreement).await?; + return Ok( + crate::cancel_dispatch::reopen_if_live(registry, chain_client, agreement).await?, + ); } - Ok(()) + Ok(false) } -/// An agreement dipper is cancelling stays `Cancelling` when the chain shows it accepted, -/// so its accept is recorded here; its end is then announced, along with the accept, -/// once the cancel lands. A withdrawn offer reads as cancelled with no accept time. +/// An agreement dipper is cancelling, or has just moved back to cancelling, stays `Cancelling` +/// when the chain shows it accepted, so its accept is recorded here; its end is then announced, +/// along with the accept, once the cancel lands. A withdrawn offer reads as cancelled with no +/// accept time. async fn record_accept_of_cancelling( snapshot: &AgreementStateSnapshot, agreement: &IndexingAgreement, + reopened: bool, registry: &R, ) { - if agreement.status != IndexingAgreementStatus::Cancelling + if (agreement.status != IndexingAgreementStatus::Cancelling && !reopened) || !snapshot.state.reached_accepted() || snapshot.accepted_at == 0 || !created_after_events_started(agreement) @@ -1343,6 +1349,10 @@ fn log_orphan_cancel( reason = "request_canceled", "Cancelling orphan agreement" ), + Err(crate::registry::Error::NoRecordsUpdated) => tracing::debug!( + agreement_id = %agreement.id, + "Orphan agreement already ended or being cancelled" + ), Err(err) => tracing::warn!( error = %err, agreement_id = %agreement.id, @@ -2534,14 +2544,12 @@ mod tests { } } - /// Minimal `ChainClient` mock for chain_listener tests. Records every - /// on-chain cancel attempt. Tests can mark specific agreements as - /// already-canceled-on-chain (cancel returns `Ok(None)`); unmarked - /// agreements get a successful `Ok(Some(zero))`. + /// Minimal `ChainClient` mock for chain_listener tests. Records every on-chain cancel + /// attempt. Every agreement reads as not live, so nothing is sent, unless + /// `live_until_cancelled` is set. #[derive(Clone, Default)] struct MockChainClient { cancels: Arc>>, - already_canceled: Arc>>, /// When set, each cancel records whether its agreement was already marked. registry: Option, marked_at_cancel: Arc>>, @@ -2562,10 +2570,6 @@ mod tests { fn was_on_chain_cancel_attempted(&self, id: &IndexingAgreementId) -> bool { self.cancels.lock().unwrap().contains(id.as_bytes()) } - - fn mark_already_canceled_on_chain(&self, id: &IndexingAgreementId) { - self.already_canceled.lock().unwrap().push(*id.as_bytes()); - } } #[async_trait::async_trait] @@ -3112,6 +3116,10 @@ mod tests { assert!(result.is_ok()); assert!(registry.was_reopened(&agreement_id)); assert!(!registry.was_marked_accepted_on_chain(&agreement_id)); + assert!( + registry.audit_writes().contains(&("accept", agreement_id)), + "its accept is recorded, so the cancel retry treats it as paying" + ); } #[tokio::test] @@ -3708,7 +3716,6 @@ mod tests { registry.add_agreement(new_id, IndexingAgreementStatus::AcceptedOnChain); registry.add_agreement(old_id, IndexingAgreementStatus::AcceptedOnChain); registry.add_pending_cancellation(new_id, old_id); - chain_client.mark_already_canceled_on_chain(&old_id); let result = execute_pending_cancellations( &new_id, @@ -3740,7 +3747,6 @@ mod tests { registry.add_agreement(new_id, IndexingAgreementStatus::AcceptedOnChain); registry.add_agreement(old_id, IndexingAgreementStatus::AcceptedOnChain); registry.add_pending_cancellation(new_id, old_id); - chain_client.mark_already_canceled_on_chain(&old_id); sweep_executable_pending_cancellations( ®istry, @@ -4519,7 +4525,6 @@ mod tests { registry.add_agreement(agreement_id, IndexingAgreementStatus::AcceptedOnChain); registry.set_agreement_request_id(agreement_id, request_id); registry.mark_request_canceled(request_id); - chain_client.mark_already_canceled_on_chain(&agreement_id); sweep_orphan_canceled_agreements(®istry, &chain_client, test_agreement_conf().as_ref()) .await; diff --git a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs index 9fcc916b..e07f5ba5 100644 --- a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs +++ b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs @@ -638,9 +638,13 @@ where let mut directly_cancelled = 0u32; let mut cancel_failures = 0u32; for old_agreement in old_iter { - let Some(new_status) = cancel_unpaired(&ctx, old_agreement).await else { - cancel_failures += 1; - continue; + let new_status = match cancel_unpaired(&ctx, old_agreement).await { + Unpaired::Moved(new_status) => new_status, + Unpaired::AlreadyEnding => continue, + Unpaired::Failed => { + cancel_failures += 1; + continue; + } }; tracing::info!( agreement_id = %old_agreement.id, @@ -897,12 +901,11 @@ mod lifecycle_event_tests { // ---- Mock: worker queue -------------------------------------------------- - /// Records every `send_indexing_agreement_proposal` call's indexer URL and every - /// queued on-chain cancel. Clone shares the buffers for inspection after `handle`. + /// Records every `send_indexing_agreement_proposal` call's indexer URL. Clone shares the + /// buffer for inspection after `handle`. #[derive(Default, Clone)] struct MockQueue { proposals: Arc>>, - cancels_queued: Arc>>, } #[async_trait] @@ -1707,9 +1710,8 @@ mod lifecycle_event_tests { async fn never_accepted_unpaired_cancel_does_not_emit_terminated() { // Both old agreements were never accepted on-chain (Created). One add // pairs with the first old agreement; the second, unpaired old agreement - // reaches the cancel loop but, being never-accepted - // (`!was_accepted`), must NOT emit `terminated`. The add still - // emits `proposed`. Net: exactly one event, a `proposed`. + // reaches the cancel loop but, never accepted, must NOT emit `terminated`. + // The add still emits `proposed`. Net: exactly one event, a `proposed`. let new_idx = indexer_id(0x44); let old_paired = indexer_id(0x55); let old_unpaired = indexer_id(0x56); @@ -1836,14 +1838,12 @@ mod lifecycle_event_tests { ctx.chain_client.fail_cancel = true; let cancelling = ctx.registry.marked_cancelling.clone(); let cancelled = ctx.registry.marked_cancelled.clone(); - let queue = ctx.queue.clone(); let result = handle(ctx, &test_message(0)).await; assert!(result.is_ok(), "got {result:?}"); assert_eq!(*cancelling.lock().unwrap(), vec![leaving.id]); assert!(cancelled.lock().unwrap().is_empty()); - assert!(queue.cancels_queued.lock().unwrap().is_empty()); } } @@ -1876,15 +1876,45 @@ mod lifecycle_event_tests { assert_eq!(*chain_client.cancelled.lock().unwrap(), vec![leaving_id]); } + + #[test] + fn an_agreement_that_ended_since_it_was_listed_is_not_a_failure() { + let agreement = crate::cancel_dispatch::tests::agreement( + crate::registry::IndexingAgreementStatus::AcceptedOnChain, + None, + ); + let backend_down = crate::registry::Error::BackendError(dipper_pgregistry::Error::DbError( + sqlx::Error::PoolTimedOut, + )); + + assert_eq!( + super::unmarked(&agreement, &crate::registry::Error::NoRecordsUpdated), + super::Unpaired::AlreadyEnding + ); + assert_eq!( + super::unmarked(&agreement, &backend_down), + super::Unpaired::Failed + ); + } } -/// Take an old agreement out of the target group, returning its new status, or `None` -/// (logged) when it couldn't be marked. One that may be live on-chain is cancelled there -/// too, which the chain listener retries until it ends. +/// What became of an old agreement taken out of the target group. +#[derive(Debug, PartialEq, Eq)] +enum Unpaired { + /// Moved to the status named. + Moved(&'static str), + /// Already ended or being cancelled, as a race with another path can leave it. + AlreadyEnding, + /// Couldn't be marked; logged. + Failed, +} + +/// Take an old agreement out of the target group. One that may be live on-chain is cancelled +/// there too, which the chain listener retries until it ends. async fn cancel_unpaired( ctx: &Ctx, agreement: &crate::registry::IndexingAgreement, -) -> Option<&'static str> +) -> Unpaired where R: AgreementRegistry + Sync, T: ChainClient, @@ -1893,9 +1923,14 @@ where || (agreement.status == crate::registry::IndexingAgreementStatus::Created && agreement.terms_version_hash.is_some()); if !may_be_live { - return mark_unpaired_cancelled(&ctx.registry, agreement) + return match ctx + .registry + .mark_indexing_agreement_as_canceled_by_requester(&agreement.id) .await - .then_some("CANCELED_BY_REQUESTER"); + { + Ok(()) => Unpaired::Moved("CANCELED_BY_REQUESTER"), + Err(err) => unmarked(agreement, &err), + }; } match crate::cancel_dispatch::start_cancel( &ctx.registry, @@ -1906,38 +1941,33 @@ where ) .await { - Ok(crate::cancel_dispatch::CancelStarted::Ended) => Some("CANCELED_BY_REQUESTER"), - Ok(crate::cancel_dispatch::CancelStarted::Cancelling) => Some("CANCELLING"), - Err(err) => { - tracing::error!( - error = %err, - agreement_id = %agreement.id, - "Failed to mark unpaired old agreement as cancelling in local DB" - ); - None + Ok(crate::cancel_dispatch::CancelStarted::Ended) => { + Unpaired::Moved("CANCELED_BY_REQUESTER") } + Ok(crate::cancel_dispatch::CancelStarted::Cancelling) => Unpaired::Moved("CANCELLING"), + Err(err) => unmarked(agreement, &err), } } -/// Mark an unpaired old agreement CanceledByRequester; false, logged, if that fails. -async fn mark_unpaired_cancelled( - registry: &R, +/// Why an unpaired old agreement couldn't be marked, logged at the level it deserves: one that +/// ended, or started cancelling, since it was listed is expected and needs nothing more. +fn unmarked( agreement: &crate::registry::IndexingAgreement, -) -> bool { - match registry - .mark_indexing_agreement_as_canceled_by_requester(&agreement.id) - .await - { - Ok(()) => true, - Err(err) => { - tracing::error!( - error = %err, - agreement_id = %agreement.id, - "Failed to mark unpaired old agreement as canceled in local DB" - ); - false - } + err: &crate::registry::Error, +) -> Unpaired { + if matches!(err, crate::registry::Error::NoRecordsUpdated) { + tracing::debug!( + agreement_id = %agreement.id, + "Unpaired old agreement already ended or being cancelled" + ); + return Unpaired::AlreadyEnding; } + tracing::error!( + error = %err, + agreement_id = %agreement.id, + "Failed to mark unpaired old agreement as ended in local DB" + ); + Unpaired::Failed } /// Olds reserved from cancellation: one per add-cancel pairing lost to a From 07b2772e68a9d131fe42942787e2b88a2e84dfce Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Tue, 6 Oct 2026 16:12:58 +0100 Subject: [PATCH 18/26] fix: keep a reopened agreement's end record accurate (#732) * fix(cancel): clear the old end when an agreement is found live again An agreement reopened because the chain shows it live kept its earlier end on record, such as a withdrawn offer, so its later real end was announced with the old transaction and time, or never if one had gone out. That record is now cleared when the chain showed it live. * fix(cancel): count an agreement the listener already ended as ended If the chain listener marked an agreement ended between dipper sending its cancel and confirming it, the confirm found nothing to update, warned that it failed, and reported the agreement as still cancelling. It now counts it as ended and logs that at debug level. * fix(registry): replace an end recorded before the agreement's accept A reopen after a failed chain read keeps the end on record, and later cancels only filled blank fields, so a withdrawn offer's end could still be announced for an agreement accepted after it. An end recorded before the accept can't be the real one, so a later end now replaces it. --- bin/dipper-service/src/cancel_dispatch.rs | 94 +++++++++++++++---- .../src/network/service/cancel_retry.rs | 18 ++++ .../src/network/service/chain_listener.rs | 1 + bin/dipper-service/src/registry.rs | 3 +- bin/dipper-service/src/registry/agreement.rs | 4 +- .../src/registry/agreement_stub.rs | 14 ++- .../cancel_rejected_agreement_on_chain.rs | 1 + dipper-pgregistry/src/postgres.rs | 26 ++++- .../tests/it_registry_postgres.rs | 71 +++++++++++++- 9 files changed, 203 insertions(+), 29 deletions(-) diff --git a/bin/dipper-service/src/cancel_dispatch.rs b/bin/dipper-service/src/cancel_dispatch.rs index 17c0eed2..54e50521 100644 --- a/bin/dipper-service/src/cancel_dispatch.rs +++ b/bin/dipper-service/src/cancel_dispatch.rs @@ -163,7 +163,8 @@ where /// Mark an agreement the chain shows dipper ended as ended, recording the cancel when its /// transaction is known, so the `terminated` sweep announces it. False, logged, when the mark -/// fails; it stays `Cancelling` for the cancel retry. +/// fails; it stays `Cancelling` for the cancel retry. One the chain listener already marked +/// ended counts as ended. pub async fn confirm_cancelled( registry: &R, agreement: &IndexingAgreement, @@ -175,16 +176,26 @@ pub async fn confirm_cancelled( if tx_hash.is_some() { record_cancel(registry, agreement, tx_hash, config).await; } - if let Err(err) = registry + match registry .mark_indexing_agreement_as_canceled_by_requester(&agreement.id) .await { - tracing::warn!( - agreement_id = %agreement.id, - error = %err, - "Failed to mark an ended agreement cancelled; the cancel retry tries again" - ); - return false; + Ok(()) => {} + Err(crate::registry::Error::NoRecordsUpdated) => { + tracing::debug!( + agreement_id = %agreement.id, + "Agreement already marked ended, as the chain listener can do first" + ); + return true; + } + Err(err) => { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to mark an ended agreement cancelled; the cancel retry tries again" + ); + return false; + } } tracing::info!( agreement_id = %agreement.id, @@ -232,20 +243,23 @@ where R: AgreementRegistry + Sync, T: ChainClient, { - match chain_client + let seen_live = match chain_client .agreement_on_chain(agreement.id.as_bytes()) .await { - Ok(AgreementOnChain::Live) => {} + Ok(AgreementOnChain::Live) => true, Ok(_) => return Ok(false), - Err(err) => tracing::warn!( - agreement_id = %agreement.id, - error = %err, - "Failed to read an ended agreement reported live; the cancel retry checks it" - ), - } + Err(err) => { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to read an ended agreement reported live; the cancel retry checks it" + ); + false + } + }; match registry - .reopen_indexing_agreement_cancel(&agreement.id) + .reopen_indexing_agreement_cancel(&agreement.id, seen_live) .await { Ok(()) => {} @@ -632,4 +646,50 @@ pub(crate) mod tests { } )); } + + /// Records what each reopen was told about the chain. + #[derive(Default)] + struct ReopenRegistry { + seen_live: Mutex>, + } + + #[async_trait] + impl crate::registry::StubAgreementRegistry for ReopenRegistry { + async fn reopen_indexing_agreement_cancel( + &self, + _id: &IndexingAgreementId, + seen_live: bool, + ) -> crate::registry::Result<()> { + self.seen_live.lock().unwrap().push(seen_live); + Ok(()) + } + } + + #[tokio::test] + async fn a_reopen_clears_the_end_on_record_only_when_the_chain_shows_it_live() { + // An unread chain reopens it all the same, but it may have ended, so its record stays. + let ag = agreement( + IndexingAgreementStatus::CanceledByRequester, + Some(vec![7u8; 32]), + ); + let live = RecordingChainClient { + still_active_after_cancel: true, + ..Default::default() + }; + let unread = RecordingChainClient { + read_back_fails: true, + ..Default::default() + }; + + for (client, seen_live) in [(live, true), (unread, false)] { + let registry = ReopenRegistry::default(); + + let reopened = super::reopen_if_live(®istry, &client, &ag) + .await + .expect("reopen"); + + assert!(reopened); + assert_eq!(*registry.seen_live.lock().unwrap(), vec![seen_live]); + } + } } diff --git a/bin/dipper-service/src/network/service/cancel_retry.rs b/bin/dipper-service/src/network/service/cancel_retry.rs index ec9e0e63..63cf5d50 100644 --- a/bin/dipper-service/src/network/service/cancel_retry.rs +++ b/bin/dipper-service/src/network/service/cancel_retry.rs @@ -337,6 +337,8 @@ mod tests { attempts: AtomicU32, checks: AtomicU32, found_ended: Mutex>>, + /// The chain listener marks it ended before the retry's own mark lands. + listener_ended_it: bool, writes: Mutex>, } @@ -354,6 +356,9 @@ mod tests { &self, id: &IndexingAgreementId, ) -> crate::registry::Result<()> { + if self.listener_ended_it { + return Err(crate::registry::Error::NoRecordsUpdated); + } self.marked_cancelled.lock().unwrap().push(*id); self.writes.lock().unwrap().push("ended"); Ok(()) @@ -577,6 +582,19 @@ mod tests { ); } + #[tokio::test] + async fn counts_an_agreement_the_listener_marked_ended_first_as_ended() { + // Not a failure: there is nothing left to retry. + let registry = MockRegistry { + listener_ended_it: true, + ..registry_with_one(true) + }; + + retry(®istry, &live_chain(), 0).await; + + assert_eq!(registry.checks.load(Ordering::SeqCst), 0); + } + #[tokio::test] async fn leaves_an_accepted_agreement_that_already_ended_to_the_listener() { // The indexer may have ended it, or an earlier cancel whose result went unread; diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index 64947665..eacc5ebb 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -2217,6 +2217,7 @@ mod tests { async fn reopen_indexing_agreement_cancel( &self, id: &IndexingAgreementId, + _seen_live: bool, ) -> RegistryResult<()> { self.state.lock().unwrap().reopened.push(*id); Ok(()) diff --git a/bin/dipper-service/src/registry.rs b/bin/dipper-service/src/registry.rs index 239ff563..e49d8c4c 100644 --- a/bin/dipper-service/src/registry.rs +++ b/bin/dipper-service/src/registry.rs @@ -414,9 +414,10 @@ impl AgreementRegistry for RegistryProvider { async fn reopen_indexing_agreement_cancel( &self, id: &IndexingAgreementId, + seen_live: bool, ) -> RegistryResult<()> { self.inner - .reopen_indexing_agreement_cancel(id) + .reopen_indexing_agreement_cancel(id, seen_live) .await .map_err(Into::into) } diff --git a/bin/dipper-service/src/registry/agreement.rs b/bin/dipper-service/src/registry/agreement.rs index 54ac69f0..c7cfeaab 100644 --- a/bin/dipper-service/src/registry/agreement.rs +++ b/bin/dipper-service/src/registry/agreement.rs @@ -284,10 +284,12 @@ pub trait AgreementRegistry { /// Move a `CANCELED_BY_REQUESTER` or `REJECTED` agreement the chain shows live back to /// `CANCELLING`, its cancel attempts reset; [`NoRecordUpdated`](Error::NoRecordsUpdated) - /// otherwise. + /// otherwise. `seen_live` when the chain was read and showed it live, which clears the end + /// on record and any announcement of it. async fn reopen_indexing_agreement_cancel( &self, id: &IndexingAgreementId, + seen_live: bool, ) -> RegistryResult<()>; /// `CANCELLING` agreements marked over `min_age_minutes` ago, those checked longest ago diff --git a/bin/dipper-service/src/registry/agreement_stub.rs b/bin/dipper-service/src/registry/agreement_stub.rs index cbb62dd6..d9092a6d 100644 --- a/bin/dipper-service/src/registry/agreement_stub.rs +++ b/bin/dipper-service/src/registry/agreement_stub.rs @@ -141,7 +141,11 @@ pub trait StubAgreementRegistry: Send + Sync { unimplemented!("mark_indexing_agreement_as_abandoning") } - async fn reopen_indexing_agreement_cancel(&self, _id: &IndexingAgreementId) -> Result<()> { + async fn reopen_indexing_agreement_cancel( + &self, + _id: &IndexingAgreementId, + _seen_live: bool, + ) -> Result<()> { unimplemented!("reopen_indexing_agreement_cancel") } @@ -446,8 +450,12 @@ impl AgreementRegistry for T { StubAgreementRegistry::mark_indexing_agreement_as_abandoning(self, id).await } - async fn reopen_indexing_agreement_cancel(&self, id: &IndexingAgreementId) -> Result<()> { - StubAgreementRegistry::reopen_indexing_agreement_cancel(self, id).await + async fn reopen_indexing_agreement_cancel( + &self, + id: &IndexingAgreementId, + seen_live: bool, + ) -> Result<()> { + StubAgreementRegistry::reopen_indexing_agreement_cancel(self, id, seen_live).await } async fn get_cancelling_agreements( diff --git a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs index cde9677a..c6d5735c 100644 --- a/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs +++ b/bin/dipper-service/src/worker/handlers/cancel_rejected_agreement_on_chain.rs @@ -82,6 +82,7 @@ mod tests { async fn reopen_indexing_agreement_cancel( &self, id: &IndexingAgreementId, + _seen_live: bool, ) -> crate::registry::Result<()> { self.reopened.lock().unwrap().push(*id); Ok(()) diff --git a/dipper-pgregistry/src/postgres.rs b/dipper-pgregistry/src/postgres.rs index 2a553614..23faca70 100644 --- a/dipper-pgregistry/src/postgres.rs +++ b/dipper-pgregistry/src/postgres.rs @@ -1008,10 +1008,13 @@ impl PgRegistry { /// Move an agreement dipper had already ended, cancelled or rejected, back to `Cancelling` /// once the chain shows it live after all, with its cancel attempts started afresh. It counts /// as checked, since no cancel is sent with it, so the retry takes it on its next sweep rather - /// than waiting for one to be mined. + /// than waiting for one to be mined. When the chain was read and showed it live (`seen_live`), + /// the end on record, and any announcement of it, no longer stands, so both are cleared for + /// the end still to come; an unread chain leaves them, as the agreement may have ended. pub async fn reopen_indexing_agreement_cancel( &self, agreement_id: &IndexingAgreementId, + seen_live: bool, ) -> Result<(), Error> { let updated = sqlx::query( r#" @@ -1021,6 +1024,11 @@ impl PgRegistry { cancel_attempts = 0, cancel_checked_at = timezone('UTC', now()), ended_seen_at = NULL, + canceled_at = CASE WHEN $5::BOOLEAN THEN NULL ELSE canceled_at END, + canceled_by = CASE WHEN $5 THEN NULL ELSE canceled_by END, + canceled_tx = CASE WHEN $5 THEN NULL ELSE canceled_tx END, + terminated_event_emitted_at = + CASE WHEN $5 THEN NULL ELSE terminated_event_emitted_at END, updated_at = timezone('UTC', now()) WHERE id = $2 AND status IN ($3, $4) "#, @@ -1029,6 +1037,7 @@ impl PgRegistry { .bind(agreement_id) .bind(IndexingAgreementStatus::CanceledByRequester) .bind(IndexingAgreementStatus::Rejected) + .bind(seen_live) .execute(&self.pool) .await?; if updated.rows_affected() == 0 { @@ -1544,6 +1553,8 @@ impl PgRegistry { /// emission sweep can populate the `terminated` event's tx/by/at fields. /// `COALESCE` keeps any value already observed on-chain. Best-effort /// enrichment: the event still emits (with fallbacks) if never recorded. + /// An end recorded before the agreement's accept, such as its offer's withdrawal before the + /// offer landed after all, can't be its end, so a later one replaces it and is announced. #[expect( clippy::cast_possible_wrap, reason = "predates this lint; fix when next touched" @@ -1558,9 +1569,16 @@ impl PgRegistry { sqlx::query( r#" UPDATE dipper_reg_indexing_agreements - SET canceled_at = COALESCE(canceled_at, $2), - canceled_by = COALESCE(canceled_by, $3), - canceled_tx = COALESCE(canceled_tx, $4) + SET canceled_at = CASE WHEN canceled_at < accepted_at AND $2 >= accepted_at + THEN $2 ELSE COALESCE(canceled_at, $2) END, + canceled_by = CASE WHEN canceled_at < accepted_at AND $2 >= accepted_at + THEN $3 ELSE COALESCE(canceled_by, $3) END, + canceled_tx = CASE WHEN canceled_at < accepted_at AND $2 >= accepted_at + THEN $4 ELSE COALESCE(canceled_tx, $4) END, + terminated_event_emitted_at = CASE + WHEN canceled_at < accepted_at AND $2 >= accepted_at THEN NULL + ELSE terminated_event_emitted_at + END WHERE id = $1 "#, ) diff --git a/dipper-pgregistry/tests/it_registry_postgres.rs b/dipper-pgregistry/tests/it_registry_postgres.rs index c0b5de0f..a246b2a3 100644 --- a/dipper-pgregistry/tests/it_registry_postgres.rs +++ b/dipper-pgregistry/tests/it_registry_postgres.rs @@ -3617,7 +3617,7 @@ async fn an_ended_agreement_found_live_on_chain_goes_back_to_cancelling() { ) .await .expect("Failed to run fixture"); - let registry = PgRegistry::new(db); + let registry = PgRegistry::new(db.clone()); let ended = fixture_agreement(0xaa); let accepted = IndexingAgreementId::from_bytes([0xaa, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2]); @@ -3625,6 +3625,11 @@ async fn an_ended_agreement_found_live_on_chain_goes_back_to_cancelling() { .mark_indexing_agreement_as_cancelling(&ended) .await .expect("mark cancelling"); + // The withdrawal of its offer, recorded as its end before the offer landed after all. + registry + .record_cancel_audit(&ended, 1_700_000_000, "0xpayer", Some("0xwithdrawal")) + .await + .expect("cancel record"); assert_eq!( registry.record_cancel_check(&ended, 2, None).await.unwrap(), 2 @@ -3635,9 +3640,19 @@ async fn an_ended_agreement_found_live_on_chain_goes_back_to_cancelling() { .expect("mark ended"); registry - .reopen_indexing_agreement_cancel(&ended) + .reopen_indexing_agreement_cancel(&ended, true) .await .expect("an ended agreement can be reopened"); + let (canceled_tx,): (Option,) = + sqlx::query_as("SELECT canceled_tx FROM dipper_reg_indexing_agreements WHERE id = $1") + .bind(ended) + .fetch_one(&db) + .await + .expect("cancel record query"); + assert_eq!( + canceled_tx, None, + "the end on record no longer stands once the chain shows it live" + ); // Reopening sends no cancel, so there is none to wait on being mined. let listed = registry @@ -3646,13 +3661,63 @@ async fn an_ended_agreement_found_live_on_chain_goes_back_to_cancelling() { .expect("cancelling query"); let ids: Vec<_> = listed.iter().map(|row| row.agreement.id).collect(); assert_eq!(ids, vec![ended], "its cancel attempts start afresh"); - let still_wanted = registry.reopen_indexing_agreement_cancel(&accepted).await; + let still_wanted = registry + .reopen_indexing_agreement_cancel(&accepted, true) + .await; assert!( matches!(still_wanted, Err(Error::NoRecordsUpdated)), "got {still_wanted:?}" ); } +#[tokio::test] +async fn an_end_recorded_before_the_accept_gives_way_to_the_real_one() { + // An offer withdrawn, then accepted when it landed after all: the withdrawal can't be the + // end of an agreement accepted later, however the agreement came back to be cancelled. + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let registry = PgRegistry::new(db.clone()); + let id = fixture_agreement(0xaa); + let canceled_tx = async || -> Option { + let (tx,): (Option,) = + sqlx::query_as("SELECT canceled_tx FROM dipper_reg_indexing_agreements WHERE id = $1") + .bind(id) + .fetch_one(&db) + .await + .expect("cancel record query"); + tx + }; + registry + .record_cancel_audit(&id, 1_000, "0xpayer", Some("0xwithdrawal")) + .await + .expect("cancel record"); + registry + .record_accepted_audit(&id, 2_000, "0xaccept") + .await + .expect("accept record"); + + registry + .record_cancel_audit(&id, 3_000, "0xpayer", Some("0xcancel")) + .await + .expect("cancel record"); + assert_eq!(canceled_tx().await.as_deref(), Some("0xcancel")); + + registry + .record_cancel_audit(&id, 4_000, "0xpayer", Some("0xlater")) + .await + .expect("cancel record"); + assert_eq!( + canceled_tx().await.as_deref(), + Some("0xcancel"), + "a real end, after the accept, is kept" + ); +} + #[tokio::test] async fn a_cancelling_agreement_stays_live_and_unannounced_until_it_ends() { let (db, _temp_db) = temp_registry_db().await; From 9a222620ab58e78f6d8e52177fd5555b95caf05e Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Tue, 6 Oct 2026 16:32:53 +0100 Subject: [PATCH 19/26] fix: skip an indexer that abandoned a deployment for a month (#733) * fix(selection): skip an indexer that abandoned a deployment for a month When an indexer stops serving an agreement, dipper ends it as abandoned and asks for a replacement, but nothing kept that indexer out, so it could be picked straight back. It is now left out of that deployment for the standard 30-day lookback, like one that cancelled. * docs(registry): rewrap the declined-indexer comments to the line limit Adding the abandoned status left 2 doc comment lines well past 100 characters. --- bin/dipper-service/src/config.rs | 5 +- bin/dipper-service/src/registry/agreement.rs | 7 +-- dipper-pgregistry/src/postgres.rs | 10 ++-- .../tests/it_registry_postgres.rs | 46 +++++++++++++++++++ 4 files changed, 59 insertions(+), 9 deletions(-) diff --git a/bin/dipper-service/src/config.rs b/bin/dipper-service/src/config.rs index 73ee033a..4a85f18d 100644 --- a/bin/dipper-service/src/config.rs +++ b/bin/dipper-service/src/config.rs @@ -817,8 +817,9 @@ pub struct DipsAgreementConfig { pub max_grt_per_billion_entities_per_30_days: f64, /// Number of days to look back for declined indexers (standard exclusion). Covers - /// CanceledByIndexer, expiries whose offer reached the chain, and structurally - /// persistent rejections (UNSUPPORTED_NETWORK, MANIFEST_TOO_LARGE). Default: 30 days. + /// CanceledByIndexer, AbandonedByIndexer, expiries whose offer reached the chain, and + /// structurally persistent rejections (UNSUPPORTED_NETWORK, MANIFEST_TOO_LARGE). Default: 30 + /// days. #[serde(default = "default_declined_indexer_lookback_days")] pub declined_indexer_lookback_days: i32, diff --git a/bin/dipper-service/src/registry/agreement.rs b/bin/dipper-service/src/registry/agreement.rs index c7cfeaab..06e0ed96 100644 --- a/bin/dipper-service/src/registry/agreement.rs +++ b/bin/dipper-service/src/registry/agreement.rs @@ -172,9 +172,10 @@ pub trait AgreementRegistry { indexer_ids: &[IndexerId], ) -> RegistryResult>>; - /// Get declined `CanceledByIndexer`/`Expired`/`Rejected` indexers grouped by - /// deployment. Each rejection reason gets its own exclusion window, as does an - /// expiry that never had an offer transaction; see the query for the details. + /// Get declined `CanceledByIndexer`/`Expired`/`Rejected` indexers, and those whose agreement + /// dipper ended `AbandonedByIndexer`, grouped by deployment. Each rejection reason gets its + /// own exclusion window, as does an expiry that never had an offer transaction; see the query + /// for the details. async fn get_declined_indexers_by_deployment( &self, default_lookback_days: i32, diff --git a/dipper-pgregistry/src/postgres.rs b/dipper-pgregistry/src/postgres.rs index 23faca70..6af8d05a 100644 --- a/dipper-pgregistry/src/postgres.rs +++ b/dipper-pgregistry/src/postgres.rs @@ -738,9 +738,10 @@ impl PgRegistry { .collect()) } - /// Get declined `CanceledByIndexer`/`Expired`/`Rejected` indexers grouped by - /// deployment (deployment id -> indexer ids). Each rejection reason gets its own - /// exclusion window, as does an expiry that never had an offer transaction. + /// Get declined `CanceledByIndexer`/`Expired`/`Rejected` indexers, and those whose agreement + /// dipper ended `AbandonedByIndexer`, grouped by deployment (deployment id -> indexer ids). + /// Each rejection reason gets its own exclusion window, as does an expiry that never had an + /// offer transaction. pub async fn get_declined_indexers_by_deployment( &self, default_lookback_days: i32, @@ -761,7 +762,7 @@ impl PgRegistry { deployment_id, array_agg(DISTINCT indexer_id) as indexer_ids FROM dipper_reg_indexing_agreements - WHERE status IN ($1, $2, $3) + WHERE status IN ($1, $2, $3, $20) AND ( -- PRICE_TOO_LOW: shorter lookback (until next IISA refresh) (rejection_reason = $6 @@ -816,6 +817,7 @@ impl PgRegistry { .bind(uncertain_lookback_days) // $17 .bind(SENDER_NOT_TRUSTED) // $18 .bind(UNSPECIFIED) // $19 + .bind(IndexingAgreementStatus::AbandonedByIndexer) // $20 .fetch_all(&self.pool) .await?; diff --git a/dipper-pgregistry/tests/it_registry_postgres.rs b/dipper-pgregistry/tests/it_registry_postgres.rs index a246b2a3..2f4c194f 100644 --- a/dipper-pgregistry/tests/it_registry_postgres.rs +++ b/dipper-pgregistry/tests/it_registry_postgres.rs @@ -677,6 +677,52 @@ async fn get_declined_indexers_by_deployment_returns_rejected() { assert!(declined_5e.contains(&indexer_c)); } +#[tokio::test] +async fn get_declined_indexers_by_deployment_includes_an_indexer_that_abandoned_it() { + // Otherwise the reassessment that replaces it could pick the same indexer straight back. + let (db, _temp_db) = temp_registry_db().await; + run_fixture(&db, include_str!("fixtures/0002_indexing_agreements.sql")) + .await + .expect("Failed to run fixture"); + let registry = PgRegistry::new(db); + // Fixture 0002's accepted agreement. + let agreement_id = + IndexingAgreementId::from_bytes([0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 3]); + let deployment: DeploymentId = "QmUzRg2HHMpbgf6Q4VHKNDbtBEJnyp5JWCh2gUX9AV6jXv" + .parse() + .unwrap(); + let indexer = indexer_id!("d609e9fdd6ce53e5a26278c50486dd6791d4d705"); + registry + .mark_indexing_agreement_as_abandoning(&agreement_id) + .await + .expect("abandon"); + registry + .mark_indexing_agreement_as_canceled_by_requester(&agreement_id) + .await + .expect("ended abandoned"); + + let within = registry + .get_declined_indexers_by_deployment(30, 1, 5, 1) + .await + .expect("Failed to get declined indexers"); + let past = registry + .get_declined_indexers_by_deployment(0, 1, 5, 1) + .await + .expect("Failed to get declined indexers"); + + assert!( + within + .get(&deployment) + .is_some_and(|ids| ids.contains(&indexer)) + ); + assert!( + !past + .get(&deployment) + .is_some_and(|ids| ids.contains(&indexer)), + "only for the standard lookback" + ); +} + #[tokio::test] async fn get_declined_indexers_by_deployment_empty_when_no_declines() { //* Given From 2cb3bd54dd647df70bc2e2fdcd95e2604d5d0749 Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Thu, 8 Oct 2026 13:23:27 +0100 Subject: [PATCH 20/26] refactor: announce agreements however old they are (#734) * refactor(listener): drop the date that left old agreements unannounced Dipper skipped announcing the accept and end of agreements created before its first release with lifecycle events, but that only spared consumers late events for old testnet agreements. Every agreement is now announced when the chain shows it accepted or ended. * fix(listener): let an end recorded before the accept give way When the listener recorded an agreement's accept and end from the chain, it recorded the end before the accept, so an older end from before the accept, such as its offer's withdrawal, was kept. The end is now recorded again after the accept, so the real one replaces it. * docs(listener): stop saying old accepted agreements are never announced A comment on the accepted-event sweep said agreements from before events existed are never announced, which stopped being true once the date that skipped them was removed. --- .../src/network/service/chain_listener.rs | 91 +++++-------------- 1 file changed, 24 insertions(+), 67 deletions(-) diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index eacc5ebb..15c00d57 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -946,21 +946,18 @@ where } } -/// When v0.1.10, the first release that announces lifecycle events, came out (2026-08-11 -/// UTC). Agreements dipper created before then are never announced, even when a replay of -/// the chain reads them again. -const LIFECYCLE_EVENTS_START: i64 = 1_786_406_400; - /// Record the accept and cancel of an agreement dipper had already marked /// cancelled that went live on-chain first, so its accepted and terminated /// events go out. Cancel first: the terminated sweep waits only for the accept. -/// Existing values win, so an agreement dipper already recorded is unchanged. +/// Existing values win, so an agreement dipper already recorded is unchanged, +/// except an end recorded before the accept, such as its offer's withdrawal: the +/// cancel is recorded again once the accept is, so that end gives way to this one. async fn record_accept_and_cancel_from_chain( snapshot: &AgreementStateSnapshot, agreement: &IndexingAgreement, registry: &R, ) { - if snapshot.accepted_at == 0 || !created_after_events_started(agreement) { + if snapshot.accepted_at == 0 { return; } let canceled_by = snapshot.canceled_by.to_string(); @@ -980,6 +977,19 @@ async fn record_accept_and_cancel_from_chain( } Err(err) => Err(err), }; + let recorded = match recorded { + Ok(()) => { + registry + .record_cancel_audit( + &agreement.id, + snapshot.canceled_at, + &canceled_by, + Some(&snapshot.canceled_tx), + ) + .await + } + Err(err) => Err(err), + }; if let Err(err) = recorded { tracing::warn!( agreement_id = %agreement.id, @@ -990,12 +1000,6 @@ async fn record_accept_and_cancel_from_chain( } } -/// Whether dipper created the agreement once it announced lifecycle events. Its own clock -/// at creation, unlike any read of the chain, doesn't depend on how far the listener lags. -fn created_after_events_started(agreement: &IndexingAgreement) -> bool { - agreement.created_at.unix_timestamp() >= LIFECYCLE_EVENTS_START -} - /// Safety net for an agreement dipper cancelled whose offer the indexer accepted /// anyway, such as one that landed after dipper's cancel. Nothing else would end /// it: reconciliation ignores an accept on a cancelled row. It goes back to @@ -1035,7 +1039,6 @@ async fn record_accept_of_cancelling( if (agreement.status != IndexingAgreementStatus::Cancelling && !reopened) || !snapshot.state.reached_accepted() || snapshot.accepted_at == 0 - || !created_after_events_started(agreement) { return; } @@ -1592,7 +1595,7 @@ where /// confirmed send. /// /// Eligibility (`get_agreements_pending_accepted_emission`) requires -/// `accepted_at IS NOT NULL` (so pre-feature rows are never backfilled) but is +/// `accepted_at IS NOT NULL` (only rows whose accept was recorded) but is /// NOT gated on current status: an agreement accepted and then cancelled in a /// single snapshot is already terminal yet must still emit its `accepted` (which /// is why this sweep runs before the terminated sweep). @@ -1991,12 +1994,6 @@ mod tests { } } - fn set_agreement_created_at(&self, agreement_id: IndexingAgreementId, unix: i64) { - if let Some(a) = self.state.lock().unwrap().agreements.get_mut(&agreement_id) { - a.created_at = OffsetDateTime::from_unix_timestamp(unix).unwrap(); - } - } - fn set_agreement_request_id( &self, agreement_id: IndexingAgreementId, @@ -2705,27 +2702,6 @@ mod tests { assert!(!registry.was_reopened(&agreement_id)); } - #[tokio::test] - async fn reconcile_records_no_accept_of_an_agreement_from_before_events_existed() { - let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); - let agreement_id = IndexingAgreementId::from_bytes(rand::random()); - registry.add_agreement(agreement_id, IndexingAgreementStatus::Cancelling); - registry.set_agreement_created_at(agreement_id, LIFECYCLE_EVENTS_START - 1); - - let snapshot = make_snapshot(agreement_id, AgreementState::Accepted, Address::ZERO); - reconcile_agreement( - &snapshot, - ®istry, - &chain_client, - test_agreement_conf().as_ref(), - ) - .await - .expect("reconcile ok"); - - assert!(registry.audit_writes().is_empty()); - } - #[tokio::test] async fn reconcile_records_no_accept_for_a_withdrawn_offer() { // The subgraph reports a withdrawn offer as cancelled by the payer with no accept @@ -3170,7 +3146,7 @@ mod tests { // Dipper had marked the agreement cancelled, but it was accepted on-chain // before being ended there. Recording both lets the accepted and terminated // events go out; the cancel goes first because the terminated sweep only - // waits for the accept. + // waits for the accept, and again after it, to replace an end from before it. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); @@ -3191,7 +3167,11 @@ mod tests { assert!(result.is_ok()); assert_eq!( registry.audit_writes(), - vec![("cancel", agreement_id), ("accept", agreement_id)] + vec![ + ("cancel", agreement_id), + ("accept", agreement_id), + ("cancel", agreement_id) + ] ); } @@ -3218,29 +3198,6 @@ mod tests { assert!(registry.audit_writes().is_empty()); } - #[tokio::test] - async fn test_reconcile_does_not_announce_an_agreement_accepted_before_events_existed() { - // Agreements created before lifecycle events existed have no recorded accept - // and are never announced, even when a replay of the chain reads them again. - let registry = MockRegistry::new(); - let chain_client = MockChainClient::default(); - let agreement_id = IndexingAgreementId::from_bytes(rand::random()); - registry.add_agreement(agreement_id, IndexingAgreementStatus::CanceledByRequester); - registry.set_agreement_created_at(agreement_id, LIFECYCLE_EVENTS_START - 1); - - let snapshot = make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); - let result = reconcile_agreement( - &snapshot, - ®istry, - &chain_client, - test_agreement_conf().as_ref(), - ) - .await; - - assert!(result.is_ok()); - assert!(registry.audit_writes().is_empty()); - } - #[tokio::test] async fn test_reconcile_records_nothing_for_cancelled_agreement_never_accepted() { // A withdrawn offer was never accepted, so there is nothing to announce. From 0f7f89c94ec29cb544940be3db79df6de8b5c045 Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Thu, 8 Oct 2026 20:18:38 +0100 Subject: [PATCH 21/26] feat: send dipper's alerts to Slack (#735) * fix(logging): write dipper's logs without colour codes Dipper wrote terminal colour codes into its logs even when they went to a log store, which split fields like the event tag so a search for them found nothing. Logs are now plain text. * feat(alerts): post the log lines an operator must act on to Slack Dipper's tagged warnings and errors, such as a cancel that keeps failing, were only in its logs. With a Slack webhook in the config, each listed tag is now posted, at most once per tag every 15 minutes with a count of the rest, without ever holding up the code that logged it. * fix(alerts): escape the characters Slack reads as markup Slack treats &, < and > in a message as links, mentions or entities, so an error carrying an HTML page, such as a proxy's 502, came through mangled. They are now escaped. * fix(alerts): describe a throttle window shorter than a minute correctly The count of held-back alerts named the window in whole minutes rounded down, so 30 seconds read as "the last 0 minutes", and a window under a minute was only checked once a minute. Both now follow the window as configured. * fix(alerts): count alerts dropped in a burst into Slack's figure Alerts dropped because the queue was full only showed up as a warning in dipper's own logs, so the "N more" figure in Slack was too low during a burst. They are now counted by event into that figure. --- bin/dipper-service/src/alerts.rs | 466 +++++++++++++++++++++++++++++++ bin/dipper-service/src/config.rs | 52 ++++ bin/dipper-service/src/main.rs | 25 +- k8s/configmap-example.yaml | 5 + 4 files changed, 540 insertions(+), 8 deletions(-) create mode 100644 bin/dipper-service/src/alerts.rs diff --git a/bin/dipper-service/src/alerts.rs b/bin/dipper-service/src/alerts.rs new file mode 100644 index 00000000..d2460b08 --- /dev/null +++ b/bin/dipper-service/src/alerts.rs @@ -0,0 +1,466 @@ +//! Sends the log lines an operator has to act on to Slack. A line is picked out by its `event` +//! tag, so the code that logs it needs no change. Posting happens in the background, at most once +//! per event in each throttle window, and never holds up or fails the code that logged it. + +use std::{ + collections::{HashMap, HashSet}, + sync::{Arc, Mutex, PoisonError}, + time::Duration, +}; + +use tokio::{sync::mpsc, time::Instant}; +use tracing::{ + Event, Level, Subscriber, + field::{Field, Visit}, +}; +use tracing_subscriber::{Layer, layer::Context}; +use url::Url; + +use crate::config::AlertsConfig; + +/// Alerts that can wait to be posted. More are dropped, as a burst that size is mostly held back +/// by the throttle anyway, and counted into the held-back figure for their event. +const QUEUE: usize = 64; + +/// How long a single post to Slack may take. +const POST_TIMEOUT: Duration = Duration::from_secs(10); + +/// How often to check for events whose window has ended with alerts held back, to post a count. +const HELD_BACK_CHECK: Duration = Duration::from_secs(60); + +/// A tagged log line to alert on. +#[derive(Debug, Clone, PartialEq, Eq)] +struct Alert { + event: String, + level: Level, + message: String, + fields: String, +} + +/// Picks the tagged log lines out of dipper's logging and queues them for posting. +pub struct AlertLayer { + events: HashSet, + queue: mpsc::Sender, + dropped: Dropped, +} + +/// Alerts dropped because the queue was full, counted by event. +type Dropped = Arc>>; + +/// The alert layer for dipper's logging, with its poster running in the background, or `None` +/// when no Slack webhook is configured. +pub fn layer(config: &AlertsConfig) -> Option { + let url = config.slack_webhook_url.as_ref()?.as_ref().clone(); + let (queue, alerts) = mpsc::channel(QUEUE); + let dropped = Dropped::default(); + tokio::spawn(post_alerts( + alerts, + url, + config.throttle, + Arc::clone(&dropped), + )); + Some(AlertLayer { + events: config.events.iter().cloned().collect(), + queue, + dropped, + }) +} + +impl Layer for AlertLayer { + fn on_event(&self, event: &Event<'_>, _ctx: Context<'_, S>) { + let mut line = LogLine::default(); + event.record(&mut line); + let Some(tag) = line.event.filter(|tag| self.events.contains(tag)) else { + return; + }; + let alert = Alert { + event: tag, + level: *event.metadata().level(), + message: line.message, + fields: line.fields, + }; + // Waiting here would hold up the code that logged it, so a full queue drops the alert; + // the poster counts it into the next message for its event. + if let Err(full) = self.queue.try_send(alert) { + *self + .dropped + .lock() + .unwrap_or_else(PoisonError::into_inner) + .entry(full.into_inner().event) + .or_default() += 1; + } + } +} + +/// A log line's `event` tag, message and other fields. +#[derive(Default)] +struct LogLine { + event: Option, + message: String, + fields: String, +} + +impl LogLine { + fn record(&mut self, field: &Field, value: String) { + match field.name() { + "event" => self.event = Some(value), + "message" => self.message = value, + name => { + if !self.fields.is_empty() { + self.fields.push(' '); + } + self.fields.push_str(&format!("{name}={value}")); + } + } + } +} + +impl Visit for LogLine { + fn record_str(&mut self, field: &Field, value: &str) { + self.record(field, value.to_owned()); + } + + fn record_debug(&mut self, field: &Field, value: &dyn std::fmt::Debug) { + self.record(field, format!("{value:?}")); + } +} + +/// Post each queued alert to Slack, subject to the throttle, until the queue closes. +async fn post_alerts( + mut alerts: mpsc::Receiver, + url: Url, + window: Duration, + dropped: Dropped, +) { + let http = match reqwest::Client::builder().timeout(POST_TIMEOUT).build() { + Ok(http) => http, + Err(err) => { + tracing::error!(error = %err, "Failed to build the HTTP client for Slack alerts; none will be sent"); + return; + } + }; + let mut throttle = Throttle::new(window); + // A window shorter than the usual check is checked as often as it ends. + let mut check = tokio::time::interval(HELD_BACK_CHECK.min(window).max(Duration::from_secs(1))); + loop { + let mut texts = Vec::new(); + tokio::select! { + alert = alerts.recv() => { + let Some(alert) = alert else { + return; + }; + if let Some(held_back) = throttle.admit(&alert.event, Instant::now()) { + texts.push(alert_text(&alert, held_back)); + } + } + _ = check.tick() => { + for (event, held_back) in throttle.due(Instant::now()) { + texts.push(held_back_text(&event, held_back, window)); + } + } + } + let lost = std::mem::take(&mut *dropped.lock().unwrap_or_else(PoisonError::into_inner)); + for (event, count) in lost { + throttle.hold_back(&event, count, Instant::now()); + } + for text in texts { + post(&http, &url, &text).await; + } + } +} + +/// At most 1 message per event in each window; alerts in between are counted and reported in +/// the next message for that event. +struct Throttle { + window: Duration, + events: HashMap, +} + +struct Window { + started: Instant, + held_back: u64, +} + +impl Throttle { + fn new(window: Duration) -> Self { + Self { + window, + events: HashMap::new(), + } + } + + /// Whether to post this alert now and, if so, how many were held back before it. + fn admit(&mut self, event: &str, now: Instant) -> Option { + if let Some(window) = self.events.get_mut(event) + && now.duration_since(window.started) < self.window + { + window.held_back += 1; + return None; + } + let held_back = self + .events + .insert( + event.to_owned(), + Window { + started: now, + held_back: 0, + }, + ) + .map_or(0, |window| window.held_back); + Some(held_back) + } + + /// Count alerts that never reached the throttle into its held-back figure for their event; + /// one with no window yet is reported at the next check. + fn hold_back(&mut self, event: &str, count: u64, now: Instant) { + let window = self.window; + self.events + .entry(event.to_owned()) + .or_insert_with(|| Window { + started: now.checked_sub(window).unwrap_or(now), + held_back: 0, + }) + .held_back += count; + } + + /// Events whose window has ended with alerts held back, each with how many; the message + /// reporting them starts that event's next window. + fn due(&mut self, now: Instant) -> Vec<(String, u64)> { + let mut due = Vec::new(); + for (event, window) in &mut self.events { + if window.held_back > 0 && now.duration_since(window.started) >= self.window { + due.push((event.clone(), window.held_back)); + *window = Window { + started: now, + held_back: 0, + }; + } + } + due + } +} + +fn alert_text(alert: &Alert, held_back: u64) -> String { + let mut text = format!( + "dipper {} `{}`: {}", + alert.level, + slack_escape(&alert.event), + slack_escape(&alert.message) + ); + if !alert.fields.is_empty() { + text.push_str(&format!("\n{}", slack_escape(&alert.fields))); + } + if held_back > 0 { + text.push_str(&format!("\n{held_back} more since the last alert")); + } + text +} + +/// Escape the 3 characters Slack reads as markup, so text like an HTML error page shows as written +/// instead of turning into links or mentions. +fn slack_escape(text: &str) -> String { + text.replace('&', "&") + .replace('<', "<") + .replace('>', ">") +} + +fn held_back_text(event: &str, held_back: u64, window: Duration) -> String { + format!( + "dipper `{event}`: {held_back} more in the last {}", + describe(window) + ) +} + +/// A throttle window in words: whole minutes where it divides into them, seconds otherwise. +fn describe(window: Duration) -> String { + let seconds = window.as_secs(); + match (seconds / 60, seconds % 60) { + (1, 0) => "minute".to_owned(), + (minutes, 0) if minutes > 0 => format!("{minutes} minutes"), + _ if seconds == 1 => "second".to_owned(), + _ => format!("{seconds} seconds"), + } +} + +/// Post a message to the Slack webhook, logging a failure without the webhook's URL, which is a +/// secret. +async fn post(http: &reqwest::Client, url: &Url, text: &str) { + let sent = http + .post(url.clone()) + .json(&serde_json::json!({ "text": text })) + .send() + .await + .and_then(reqwest::Response::error_for_status); + if let Err(err) = sent { + tracing::warn!(error = %err.without_url(), "Failed to post an alert to Slack"); + } +} + +#[cfg(test)] +mod tests { + use tracing_subscriber::layer::SubscriberExt; + use wiremock::{ + Mock, MockServer, ResponseTemplate, + matchers::{body_json, method}, + }; + + use super::*; + + fn test_layer(events: &[&str]) -> (AlertLayer, mpsc::Receiver) { + let (queue, alerts) = mpsc::channel(QUEUE); + let layer = AlertLayer { + events: events.iter().map(|event| (*event).to_owned()).collect(), + queue, + dropped: Dropped::default(), + }; + (layer, alerts) + } + + #[test] + fn picks_out_only_the_listed_events_whatever_their_level() { + let (layer, mut alerts) = test_layer(&["agreement_cancel_stuck", "nonce_gap_fill_failed"]); + let logging = tracing::Dispatch::new(tracing_subscriber::registry().with(layer)); + + tracing::dispatcher::with_default(&logging, || { + tracing::error!( + event = "agreement_cancel_stuck", + agreement_id = %"0xab", + attempts = 10u32, + "Cancelling an agreement keeps failing" + ); + tracing::warn!( + event = "nonce_gap_fill_failed", + nonce = 7u64, + "Gap fill failed" + ); + }); + tracing::dispatcher::with_default(&logging, || { + tracing::error!(event = "something_else", "Not listed"); + tracing::error!("Not tagged"); + }); + + assert_eq!( + alerts.try_recv().expect("first alert"), + Alert { + event: "agreement_cancel_stuck".to_owned(), + level: Level::ERROR, + message: "Cancelling an agreement keeps failing".to_owned(), + fields: "agreement_id=0xab attempts=10".to_owned(), + } + ); + assert_eq!( + alerts.try_recv().expect("second alert").event, + "nonce_gap_fill_failed" + ); + assert!(alerts.try_recv().is_err(), "nothing else"); + } + + #[test] + fn counts_alerts_dropped_when_the_queue_is_full() { + let (queue, _alerts) = mpsc::channel(1); + let layer = AlertLayer { + events: HashSet::from(["rpc_blocks_refused".to_owned()]), + queue, + dropped: Dropped::default(), + }; + let dropped = Arc::clone(&layer.dropped); + let subscriber = tracing_subscriber::registry().with(layer); + + tracing::subscriber::with_default(subscriber, || { + for _ in 0..3 { + tracing::error!(event = "rpc_blocks_refused", "Refused"); + } + }); + + assert_eq!( + *dropped.lock().unwrap(), + HashMap::from([("rpc_blocks_refused".to_owned(), 2)]) + ); + } + + #[test] + fn sends_at_most_1_message_per_event_in_each_window() { + let window = Duration::from_secs(900); + let mut throttle = Throttle::new(window); + let start = Instant::now(); + + assert_eq!(throttle.admit("a", start), Some(0)); + assert_eq!(throttle.admit("a", start + Duration::from_secs(60)), None); + assert_eq!(throttle.admit("a", start + Duration::from_secs(120)), None); + assert_eq!(throttle.admit("b", start), Some(0), "each event on its own"); + + assert!(throttle.due(start + Duration::from_secs(600)).is_empty()); + let after = start + window; + assert_eq!(throttle.due(after), vec![("a".to_owned(), 2)]); + assert_eq!( + throttle.admit("a", after + Duration::from_secs(60)), + None, + "the count starts the next window" + ); + assert_eq!(throttle.admit("a", after + window), Some(1)); + } + + #[test] + fn counts_dropped_alerts_into_the_held_back_figure() { + let window = Duration::from_secs(900); + let mut throttle = Throttle::new(window); + let start = Instant::now(); + assert_eq!(throttle.admit("a", start), Some(0)); + + throttle.hold_back("a", 5, start + Duration::from_secs(60)); + throttle.hold_back("b", 2, start + Duration::from_secs(60)); + + let mut due = throttle.due(start + window); + due.sort(); + assert_eq!(due, vec![("a".to_owned(), 5), ("b".to_owned(), 2)]); + } + + #[test] + fn escapes_what_slack_reads_as_markup() { + let alert = Alert { + event: "nonce_gap_fill_failed".to_owned(), + level: Level::WARN, + message: "Gap fill failed".to_owned(), + fields: "error=502 & more".to_owned(), + }; + + assert!(alert_text(&alert, 0).ends_with("error=<html>502 & more</html>")); + } + + #[test] + fn says_what_happened_and_how_many_more() { + let alert = Alert { + event: "agreement_cancel_stuck".to_owned(), + level: Level::ERROR, + message: "Cancelling an agreement keeps failing".to_owned(), + fields: "agreement_id=0xab".to_owned(), + }; + + assert_eq!( + alert_text(&alert, 3), + "dipper ERROR `agreement_cancel_stuck`: Cancelling an agreement keeps failing\n\ + agreement_id=0xab\n3 more since the last alert" + ); + assert_eq!( + held_back_text("rpc_blocks_refused", 5, Duration::from_secs(900)), + "dipper `rpc_blocks_refused`: 5 more in the last 15 minutes" + ); + assert_eq!(describe(Duration::from_secs(60)), "minute"); + assert_eq!(describe(Duration::from_secs(90)), "90 seconds"); + assert_eq!(describe(Duration::from_secs(30)), "30 seconds"); + } + + #[tokio::test] + async fn posts_the_message_as_slack_text() { + let slack = MockServer::start().await; + Mock::given(method("POST")) + .and(body_json(serde_json::json!({ "text": "hello" }))) + .respond_with(ResponseTemplate::new(200)) + .expect(1) + .mount(&slack) + .await; + let url: Url = slack.uri().parse().expect("URL"); + + post(&reqwest::Client::new(), &url, "hello").await; + } +} diff --git a/bin/dipper-service/src/config.rs b/bin/dipper-service/src/config.rs index 4a85f18d..8bc4619c 100644 --- a/bin/dipper-service/src/config.rs +++ b/bin/dipper-service/src/config.rs @@ -97,6 +97,42 @@ pub struct Config { /// the server off. #[serde(default)] pub health: HealthConfig, + /// Where to send alerts for the log lines an operator has to act on. Without a webhook, + /// none are sent. + #[serde(default)] + pub alerts: AlertsConfig, +} + +/// Alerts for the log lines an operator has to act on, picked out by their `event` tag and +/// posted to Slack, at most once per event in each throttle window. +#[serde_as] +#[derive(Debug, serde::Deserialize)] +#[serde(default, deny_unknown_fields)] +pub struct AlertsConfig { + /// The Slack incoming-webhook URL to post alerts to. Without one, none are sent. + pub slack_webhook_url: Option>, + /// The `event` tags of the log lines to alert on. + pub events: Vec, + /// Least time, in seconds, between 2 messages for the same event; alerts in between are + /// counted into the next one. + #[serde_as(as = "serde_with::DurationSeconds")] + pub throttle: Duration, +} + +impl Default for AlertsConfig { + fn default() -> Self { + Self { + slack_webhook_url: None, + events: [ + "agreement_cancel_stuck", + "rpc_blocks_refused", + "nonce_gap_fill_failed", + ] + .map(str::to_owned) + .to_vec(), + throttle: Duration::from_secs(15 * 60), + } + } } /// Configuration for the HTTP health endpoint used by orchestrator liveness probes. Omitting the @@ -2082,6 +2118,22 @@ mod tests { ); } + #[test] + fn alerts_config_takes_a_webhook_and_keeps_the_default_events() { + let alerts = serde_json::from_str::( + r#"{"slack_webhook_url": "https://hooks.slack.com/services/T/B/x", "throttle": 60}"#, + ) + .expect("alerts config"); + + assert!(alerts.slack_webhook_url.is_some()); + assert_eq!(alerts.throttle, Duration::from_secs(60)); + assert_eq!(alerts.events, AlertsConfig::default().events); + assert!( + !format!("{alerts:?}").contains("hooks.slack.com"), + "the webhook is a secret, kept out of logs" + ); + } + /// A misspelled key would otherwise be dropped in silence, leaving an operator who meant to /// change the endpoint with the defaults and no sign that the setting never took effect. #[test] diff --git a/bin/dipper-service/src/main.rs b/bin/dipper-service/src/main.rs index e6a243f1..4b52b7c2 100644 --- a/bin/dipper-service/src/main.rs +++ b/bin/dipper-service/src/main.rs @@ -8,7 +8,10 @@ use dipper_producer::events::{ use futures_lite::StreamExt; use thegraph_core::alloy::signers::local::PrivateKeySigner; use tokio::task::JoinSet; -use tracing_subscriber::EnvFilter; +use tracing_subscriber::{ + EnvFilter, Layer as _, filter::LevelFilter, layer::SubscriberExt as _, + util::SubscriberInitExt as _, +}; use self::{ config::DEFAULT_MAX_CANDIDATES, registry::RegistryProvider, signing::eip712::Eip712Signer, @@ -17,6 +20,7 @@ use self::{ use crate::config::EventStreamingConfig; mod admin_rpc_server; +mod alerts; mod cancel_dispatch; mod chain_client; mod config; @@ -61,19 +65,24 @@ const STOP_STEP_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(5) reason = "predates this lint; fix when next touched" )] pub async fn main() -> anyhow::Result<()> { - // Set up logging - tracing_subscriber::fmt() - .with_env_filter(EnvFilter::from_default_env()) - .init(); - - // Load the configuration - tracing::debug!("loading configuration"); + // Load the configuration first, since logging needs it to know where to send alerts let conf_path = env::args() .nth(1) .expect("Missing argument for config path") .parse::() .expect("Invalid path"); let conf = config::load_from_file(&conf_path).expect("Failed to load config"); + + // Set up logging. Plain text, with no colour codes, so log stores can search it. Alerts see + // warnings and errors whatever the log level is set to. + tracing_subscriber::registry() + .with( + tracing_subscriber::fmt::layer() + .with_ansi(false) + .with_filter(EnvFilter::from_default_env()), + ) + .with(alerts::layer(&conf.alerts).map(|layer| layer.with_filter(LevelFilter::WARN))) + .init(); tracing::debug!(conf=?conf, "configuration loaded"); // Reject a config the protocol-managed path can't run with before building diff --git a/k8s/configmap-example.yaml b/k8s/configmap-example.yaml index 2d369960..88222ea5 100644 --- a/k8s/configmap-example.yaml +++ b/k8s/configmap-example.yaml @@ -118,5 +118,10 @@ data: "enabled": true, "listen_addr": "0.0.0.0:8546", "threshold": 840 + }, + "alerts": { + "slack_webhook_url": "REPLACE_ME_OR_REMOVE_FOR_NO_ALERTS", + "events": ["agreement_cancel_stuck", "rpc_blocks_refused", "nonce_gap_fill_failed"], + "throttle": 900 } } From 43039c1a19d2cc9fc5b3aaa046230642a6c87a43 Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Fri, 9 Oct 2026 19:03:42 +0100 Subject: [PATCH 22/26] fix: hide RPC keys in logs and keep late offer hashes (#736) * fix(rpc): hide endpoint API keys in cross-check failure logs The latest-block cross-check logged each endpoint's raw error, which names its URL, and hosted endpoints carry their API key in the URL. It now logs the same redacted description other calls use. * fix(registry): keep the offer hash of an agreement being cancelled An agreement can be marked for cancelling while its offer is mining, and the offer can still land. Its hash is now stored then, without moving the time that paces its cancel retry, and an ended agreement reports that nothing was stored instead of skipping silently. * perf(alerts): read only the event tag of lines no alert is set up for Every warning passes through the Slack alert layer, which formatted all of a line's fields before checking its tag. It now reads the tag first and formats the rest only for listed events. * docs(cancel): drop comments saying ConfigError switches off cancels 2 comments said the cancel path reads ConfigError as the chain client being switched off, which no code does any more. --- bin/dipper-service/src/alerts.rs | 54 ++++++++++++++++--- bin/dipper-service/src/cancel_dispatch.rs | 5 +- bin/dipper-service/src/chain_client/client.rs | 5 +- .../src/chain_client/rpc_provider.rs | 36 +++++++++++-- bin/dipper-service/src/registry/agreement.rs | 7 ++- .../src/worker/handlers/submit_offer.rs | 13 +++-- dipper-pgregistry/src/postgres.rs | 25 ++++----- .../tests/it_registry_postgres.rs | 50 +++++++++++++++++ 8 files changed, 156 insertions(+), 39 deletions(-) diff --git a/bin/dipper-service/src/alerts.rs b/bin/dipper-service/src/alerts.rs index d2460b08..8147ceef 100644 --- a/bin/dipper-service/src/alerts.rs +++ b/bin/dipper-service/src/alerts.rs @@ -68,11 +68,13 @@ pub fn layer(config: &AlertsConfig) -> Option { impl Layer for AlertLayer { fn on_event(&self, event: &Event<'_>, _ctx: Context<'_, S>) { - let mut line = LogLine::default(); - event.record(&mut line); - let Some(tag) = line.event.filter(|tag| self.events.contains(tag)) else { + let mut tag = EventTag::default(); + event.record(&mut tag); + let Some(tag) = tag.0.filter(|tag| self.events.contains(tag)) else { return; }; + let mut line = LogLine::default(); + event.record(&mut line); let alert = Alert { event: tag, level: *event.metadata().level(), @@ -92,10 +94,27 @@ impl Layer for AlertLayer { } } -/// A log line's `event` tag, message and other fields. +/// Only a log line's `event` tag, read first so a line no alert is set up for costs no more. +#[derive(Default)] +struct EventTag(Option); + +impl Visit for EventTag { + fn record_str(&mut self, field: &Field, value: &str) { + if field.name() == "event" { + self.0 = Some(value.to_owned()); + } + } + + fn record_debug(&mut self, field: &Field, value: &dyn std::fmt::Debug) { + if field.name() == "event" { + self.0 = Some(format!("{value:?}")); + } + } +} + +/// A log line's message and its fields other than the `event` tag. #[derive(Default)] struct LogLine { - event: Option, message: String, fields: String, } @@ -103,7 +122,7 @@ struct LogLine { impl LogLine { fn record(&mut self, field: &Field, value: String) { match field.name() { - "event" => self.event = Some(value), + "event" => {} "message" => self.message = value, name => { if !self.fields.is_empty() { @@ -355,6 +374,29 @@ mod tests { assert!(alerts.try_recv().is_err(), "nothing else"); } + /// Every warning in dipper passes through this layer, so 1 it won't post shouldn't cost + /// formatting its fields. + #[test] + fn leaves_the_fields_of_lines_it_wont_post_unformatted() { + struct Counted<'a>(&'a std::sync::atomic::AtomicUsize); + impl std::fmt::Debug for Counted<'_> { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.0.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + f.write_str("counted") + } + } + let formatted = std::sync::atomic::AtomicUsize::new(0); + let (layer, _alerts) = test_layer(&["agreement_cancel_stuck"]); + let subscriber = tracing_subscriber::registry().with(layer); + + tracing::subscriber::with_default(subscriber, || { + tracing::warn!(event = "something_else", value = ?Counted(&formatted), "Not listed"); + tracing::warn!(value = ?Counted(&formatted), "Not tagged"); + }); + + assert_eq!(formatted.load(std::sync::atomic::Ordering::Relaxed), 0); + } + #[test] fn counts_alerts_dropped_when_the_queue_is_full() { let (queue, _alerts) = mpsc::channel(1); diff --git a/bin/dipper-service/src/cancel_dispatch.rs b/bin/dipper-service/src/cancel_dispatch.rs index 54e50521..bd3674db 100644 --- a/bin/dipper-service/src/cancel_dispatch.rs +++ b/bin/dipper-service/src/cancel_dispatch.rs @@ -530,9 +530,8 @@ pub(crate) mod tests { #[tokio::test] async fn manager_cancel_missing_hash_is_distinct_error_and_sends_nothing() { - // eh-1: a missing hash must be the distinct MissingTermsVersionHash, not - // a ConfigError the liveness checker reads as "chain client disabled" - // and would silently abandon while the agreement stays live on-chain. + // A missing hash must be the distinct MissingTermsVersionHash: no cancel can be sent + // without it, so the cancel retry spends every attempt at once and raises the alert. let client = RecordingChainClient::default(); let ag = agreement(IndexingAgreementStatus::AcceptedOnChain, None); diff --git a/bin/dipper-service/src/chain_client/client.rs b/bin/dipper-service/src/chain_client/client.rs index 7ca8a9c5..69122dea 100644 --- a/bin/dipper-service/src/chain_client/client.rs +++ b/bin/dipper-service/src/chain_client/client.rs @@ -734,9 +734,8 @@ impl AlloyChainClient { /// already does. Signing happens once, up front, so every endpoint is offered the same /// bytes under one hash and the hash is known before anyone is asked to accept them. async fn send_transaction(&self, tx: &TransactionRequest) -> Result { - // Nothing fills a field in on this path any more, and a request that names no chain is - // signed for chain 1 rather than refused, so check before the signature exists. Not a - // `ConfigError`: the cancel path reads that as the chain client being switched off. + // Nothing fills a field in on this path, and a request that names no chain is signed + // for chain 1 rather than refused, so check before the signature exists. if tx.chain_id() != Some(self.inner.chain_id) { return Err(ChainClientError::SubmitFailed(anyhow::anyhow!( "refusing to sign for chain {:?} while configured for chain {}", diff --git a/bin/dipper-service/src/chain_client/rpc_provider.rs b/bin/dipper-service/src/chain_client/rpc_provider.rs index 836c6b98..d6e8bf56 100644 --- a/bin/dipper-service/src/chain_client/rpc_provider.rs +++ b/bin/dipper-service/src/chain_client/rpc_provider.rs @@ -53,6 +53,16 @@ fn describe_failure(url: &Url, error: &TransportError) -> String { .replace(url.as_str().trim_end_matches('/'), &name) } +/// 1 endpoint's latest block, asked once. A failure is described without the URL, since +/// hosted endpoints carry their API key in it. +async fn latest_block(http: reqwest::Client, url: &Url) -> Result { + let provider = ProviderBuilder::new().connect_reqwest(http, url.clone()); + provider + .get_block_number() + .await + .map_err(|err| describe_failure(url, &err)) +} + /// Error text that indicates a transient failure worth retrying, used only for faults /// that arrive as prose rather than as a status code or JSON-RPC error object. const RETRYABLE_ERROR_PATTERNS: &[&str] = &[ @@ -171,17 +181,16 @@ impl RpcProviderPool { pub async fn latest_blocks(&self) -> Vec { let mut asks = tokio::task::JoinSet::new(); for url in &self.providers { - let provider = ProviderBuilder::new().connect_reqwest(self.http.clone(), url.clone()); - let endpoint = endpoint_name(url); - asks.spawn(async move { (endpoint, provider.get_block_number().await) }); + let (http, url) = (self.http.clone(), url.clone()); + asks.spawn(async move { (endpoint_name(&url), latest_block(http, &url).await) }); } let mut heads = Vec::with_capacity(self.providers.len()); while let Some(answer) = asks.join_next().await { match answer { Ok((_, Ok(head))) => heads.push(head), - Ok((endpoint, Err(err))) => tracing::debug!( + Ok((endpoint, Err(reason))) => tracing::debug!( provider = %endpoint, - error = %err, + error = %reason, "RPC endpoint didn't give its latest block for a cross-check" ), Err(err) => tracing::warn!(error = %err, "Latest-block cross-check task failed"), @@ -592,6 +601,23 @@ mod tests { ); } + /// The latest-block cross-check logs each endpoint's failure, so it must hide the key too. + #[tokio::test] + async fn a_cross_check_failure_hides_the_api_key() { + let keyed: Url = "http://127.0.0.1:1/v2/super-secret-key" + .parse() + .expect("keyed endpoint URL"); + + let reason = latest_block(reqwest::Client::new(), &keyed) + .await + .expect_err("nothing is listening, so the ask fails"); + + assert!( + !reason.contains("super-secret-key"), + "the API key must not appear in the failure: {reason}" + ); + } + /// An endpoint that answers can only describe its own refusal, so it never repeats the /// URL. One that never answers is described by the HTTP client instead, which says which /// URL it was reaching for, and that is where the key sits. Nothing listens on port 1. diff --git a/bin/dipper-service/src/registry/agreement.rs b/bin/dipper-service/src/registry/agreement.rs index 06e0ed96..95da2b73 100644 --- a/bin/dipper-service/src/registry/agreement.rs +++ b/bin/dipper-service/src/registry/agreement.rs @@ -247,10 +247,9 @@ pub trait AgreementRegistry { id: &IndexingAgreementId, ) -> RegistryResult<()>; - /// Record the on-chain tx hash of the most recent `offer()` submission - /// for this agreement. Observability-only; does not transition status. - /// Called once per submit (including resubmits after a dropped tx) so - /// the DB reflects the live hash rather than an evicted one. + /// Record the hash of the latest `offer()` transaction, unless the agreement has ended, so + /// a resubmit replaces a dropped one. [`NoRecordUpdated`](Error::NoRecordsUpdated) when no + /// row took it. async fn update_offer_tx_hash( &self, id: &IndexingAgreementId, diff --git a/bin/dipper-service/src/worker/handlers/submit_offer.rs b/bin/dipper-service/src/worker/handlers/submit_offer.rs index f77fa2ae..f5ee5088 100644 --- a/bin/dipper-service/src/worker/handlers/submit_offer.rs +++ b/bin/dipper-service/src/worker/handlers/submit_offer.rs @@ -121,19 +121,24 @@ where tx_hash = %tx_hash, "Offer submitted on-chain successfully" ); - // Observability only: record which tx hash actually mined. // Any failure here is non-fatal to the overall flow. - if let Err(err) = ctx + match ctx .registry .update_offer_tx_hash(agreement_id, tx_hash.as_ref()) .await { - tracing::warn!( + Ok(()) => {} + Err(crate::registry::Error::NoRecordsUpdated) => tracing::debug!( + agreement_id = %agreement_id, + tx_hash = %tx_hash, + "Agreement ended while its offer was mining, so its offer_tx_hash isn't stored" + ), + Err(err) => tracing::warn!( agreement_id = %agreement_id, tx_hash = %tx_hash, error = %err, "Failed to persist offer_tx_hash; continuing" - ); + ), } } Err(err @ ChainClientError::TxDropped { .. }) => { diff --git a/dipper-pgregistry/src/postgres.rs b/dipper-pgregistry/src/postgres.rs index 6af8d05a..d1bc8194 100644 --- a/dipper-pgregistry/src/postgres.rs +++ b/dipper-pgregistry/src/postgres.rs @@ -914,37 +914,34 @@ impl PgRegistry { Ok(()) } - /// Persist the on-chain tx hash of the most recent `offer()` submission - /// for this agreement. Overwrites any prior value, so a resubmit after - /// mempool eviction records the live hash rather than the dropped one. - /// Observability-only: no status transition is performed here. - /// - /// Guarded on `status IN (Created, AcceptedOnChain)` so a delayed - /// receipt-confirmation cannot stamp `offer_tx_hash` onto a row that - /// has since transitioned to `Expired`, `Unresponsive`, `Rejected`, - /// or one of the cancel states. The caller treats any failure here - /// as non-fatal and just logs; a no-match result is also non-fatal - /// and silently skipped. + /// Record the hash of the latest `offer()` transaction, unless the agreement has ended. A + /// `Cancelling` row keeps its `updated_at`, which says when it was marked and paces its + /// cancel retry. Returns [`Error::NoRecordsUpdated`] when no row took the hash. pub async fn update_offer_tx_hash( &self, agreement_id: &IndexingAgreementId, tx_hash: &[u8; 32], ) -> Result<(), Error> { - sqlx::query( + let updated = sqlx::query( r#" UPDATE dipper_reg_indexing_agreements SET offer_tx_hash = $1, - updated_at = timezone('UTC', now()) - WHERE id = $2 AND status IN ($3, $4) + updated_at = CASE WHEN status = $5 THEN updated_at + ELSE timezone('UTC', now()) END + WHERE id = $2 AND status IN ($3, $4, $5) "#, ) .bind(&tx_hash[..]) .bind(agreement_id) .bind(IndexingAgreementStatus::Created) .bind(IndexingAgreementStatus::AcceptedOnChain) + .bind(IndexingAgreementStatus::Cancelling) .execute(&self.pool) .await?; + if updated.rows_affected() == 0 { + return Err(Error::NoRecordsUpdated); + } Ok(()) } diff --git a/dipper-pgregistry/tests/it_registry_postgres.rs b/dipper-pgregistry/tests/it_registry_postgres.rs index 2f4c194f..e5bd0f3a 100644 --- a/dipper-pgregistry/tests/it_registry_postgres.rs +++ b/dipper-pgregistry/tests/it_registry_postgres.rs @@ -3834,3 +3834,53 @@ async fn a_cancelling_agreement_stays_live_and_unannounced_until_it_ends() { .expect("terminated query"); assert!(terminated.iter().any(|p| p.agreement_id == cancelling)); } + +/// Reassess can mark an agreement `Cancelling` while its offer is mining, and the offer can +/// still land, so its hash is worth keeping. Once the agreement has ended it isn't. +#[tokio::test] +async fn an_offer_mined_after_the_cancel_began_keeps_its_hash() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let registry = PgRegistry::new(db.clone()); + let id = fixture_agreement(0xaa); + registry + .mark_indexing_agreement_as_cancelling(&id) + .await + .expect("mark cancelling"); + let stored = async || { + sqlx::query_as::<_, (Option>, time::OffsetDateTime)>( + "SELECT offer_tx_hash, updated_at FROM dipper_reg_indexing_agreements WHERE id = $1", + ) + .bind(id) + .fetch_one(&db) + .await + .expect("read the agreement") + }; + let (_, marked_at) = stored().await; + + registry + .update_offer_tx_hash(&id, &[0x11; 32]) + .await + .expect("a cancelling agreement takes the hash"); + assert_eq!( + stored().await, + (Some(vec![0x11; 32]), marked_at), + "hash stored, and the time it was marked kept" + ); + + sqlx::query("UPDATE dipper_reg_indexing_agreements SET status = 5 WHERE id = $1") + .bind(id) + .execute(&db) + .await + .expect("expire the agreement"); + let err = registry + .update_offer_tx_hash(&id, &[0x22; 32]) + .await + .expect_err("an ended agreement doesn't take the hash"); + assert!(matches!(err, Error::NoRecordsUpdated)); +} From 07583b1f5a27a41fd1856fb5bcd3c229bdbe78a9 Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Fri, 9 Oct 2026 19:03:42 +0100 Subject: [PATCH 23/26] fix: alert on unsendable cancels and stop paying 2 indexers (#737) * fix(cancel): count cancels that can never be sent toward the stuck alert A cancel the chain client refused to send, such as gas over the cap or a signer out of funds, counted as an outage, so it was retried forever without raising agreement_cancel_stuck. The chain is read just before each send, so every failed send now counts except an unchecked receipt. * fix(cancel): run the cancel retry as its own service Agreements marked Cancelling were only ever finished by a sweep inside the chain listener, which config can turn off, leaving them cancelling for good. The sweep now runs on its own every 5 minutes whatever the listener does, and a stop request cuts a slow sweep short. * fix(liveness): replace a stale agreement only once it can't be paid When the cancel of an agreement whose indexer stopped serving it failed, a replacement was queued anyway, so dipper paid both indexers until a retry landed the cancel. The replacement now waits until the chain shows the old one ended, and the cancel retry queues it once it ends it. * fix(listener): cancel a replaced agreement dipper can't mark cancelling When a replacement was accepted, an old agreement in a status dipper can't mark Cancelling, such as Unresponsive with its offer still open, got no on-chain cancel, so the indexer could accept it. It now gets one if the chain shows it live, and its pending row stays for retry if that fails. * fix(liveness): replace an abandoned agreement however it ends A stale agreement whose cancel failed was replaced only if the cancel retry ended it, so one the chain listener ended first was never replaced. The agreement now records that a replacement is due, and the cancel retry queues it once the agreement has ended, whatever ended it, and only once. * fix(listener): alert instead of resending a cancel the contract refuses A replaced agreement that can't be marked cancelling had its cancel resent every sweep even when the contract refused it or it mined without ending the agreement, costing gas each time with no alert. Those now raise agreement_cancel_stuck once and stop; a failed read or send still retries. * revert(listener): drop the cancel for replaced unmarkable agreements That change cancelled a replaced Unresponsive agreement on-chain, assuming its offer could still be open. An agreement is only marked Unresponsive when the gRPC proposal fails, and dipper puts an offer on-chain only after the indexer accepts that proposal, so no such offer exists. --- bin/dipper-service/src/cancel_dispatch.rs | 22 +- bin/dipper-service/src/main.rs | 16 + .../src/network/service/cancel_retry.rs | 347 ++++++++++++++++-- .../src/network/service/chain_listener.rs | 19 +- .../src/network/service/liveness_checker.rs | 223 ++++++++--- bin/dipper-service/src/registry.rs | 21 ++ bin/dipper-service/src/registry/agreement.rs | 10 + .../src/registry/agreement_stub.rs | 22 ++ .../handlers/reassess_indexing_request.rs | 11 +- .../src/worker/handlers/submit_offer.rs | 4 +- ...20261008000000_add_replacement_pending.sql | 8 + dipper-pgregistry/src/postgres.rs | 51 +++ .../tests/it_registry_postgres.rs | 50 +++ 13 files changed, 681 insertions(+), 123 deletions(-) create mode 100644 dipper-pgregistry/migrations/20261008000000_add_replacement_pending.sql diff --git a/bin/dipper-service/src/cancel_dispatch.rs b/bin/dipper-service/src/cancel_dispatch.rs index bd3674db..cedd4326 100644 --- a/bin/dipper-service/src/cancel_dispatch.rs +++ b/bin/dipper-service/src/cancel_dispatch.rs @@ -92,13 +92,17 @@ impl CancelReason { } } -/// What [`start_cancel`] left an agreement as. +/// What [`start_cancel`] left an agreement as. Unless `Ended`, it stays `Cancelling` for the +/// cancel retry to finish. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum CancelStarted { /// It was accepted and its cancel landed: now ended, as its [`CancelReason`] says. Ended, - /// Still `Cancelling`; the chain listener finishes it once it can't go live. - Cancelling, + /// The chain shows nothing live, so nobody is being paid for it. + NotLive, + /// The chain couldn't be read, or its cancel failed or couldn't be confirmed, so it may + /// still be live and paid. + MayBeLive, } /// Start ending an agreement that may be live on-chain. It is marked `Cancelling` first, so an @@ -129,18 +133,18 @@ where } let tx_hash = match cancel_if_live(chain_client, agreement, config).await { LiveCancel::Ended(tx_hash) => tx_hash, - LiveCancel::NotLive { .. } => return Ok(CancelStarted::Cancelling), + LiveCancel::NotLive { .. } => return Ok(CancelStarted::NotLive), LiveCancel::ReadFailed(err) | LiveCancel::CancelFailed(err) => { tracing::warn!( agreement_id = %agreement.id, error = %err, - "On-chain cancel failed; the chain listener retries it" + "On-chain cancel failed; the cancel retry sends it again" ); - return Ok(CancelStarted::Cancelling); + return Ok(CancelStarted::MayBeLive); } LiveCancel::Unconfirmed { tx_hash, err } => { log_unconfirmed(agreement, tx_hash, &err); - return Ok(CancelStarted::Cancelling); + return Ok(CancelStarted::MayBeLive); } }; tracing::info!( @@ -150,13 +154,13 @@ where ); // An offer never accepted could still land and be accepted until its deadline. if agreement.status != IndexingAgreementStatus::AcceptedOnChain { - return Ok(CancelStarted::Cancelling); + return Ok(CancelStarted::NotLive); } Ok( if confirm_cancelled(registry, agreement, reason, tx_hash, config).await { CancelStarted::Ended } else { - CancelStarted::Cancelling + CancelStarted::NotLive }, ) } diff --git a/bin/dipper-service/src/main.rs b/bin/dipper-service/src/main.rs index 4b52b7c2..17810755 100644 --- a/bin/dipper-service/src/main.rs +++ b/bin/dipper-service/src/main.rs @@ -107,6 +107,7 @@ pub async fn main() -> anyhow::Result<()> { let chain_listener_agreement_conf = agreement_conf.clone(); let liveness_agreement_conf = agreement_conf.clone(); let escrow_reconciler_agreement_conf = agreement_conf.clone(); + let cancel_retry_agreement_conf = agreement_conf.clone(); // Canonical chain id and RecurringCollector address, read once and shared by the // admin signer, the gRPC proposal signer, and the on-chain chain client so their @@ -545,6 +546,15 @@ pub async fn main() -> anyhow::Result<()> { _ => None, }; + //- The cancel retry, always on: it alone finishes the cancels dipper starts + let (cancel_retry_handle, cancel_retry_service) = + network::service::cancel_retry::new(network::service::cancel_retry::Ctx { + registry: registry.clone(), + chain_client: chain_client.clone(), + agreement_conf: cancel_retry_agreement_conf, + worker_queue: worker_handle.queue().clone(), + }); + //- The liveness checker service (optional, enabled by config) // Detects indexers who silently stop indexing active AcceptedOnChain agreements let liveness_checker_handle = match conf.liveness_checker { @@ -683,6 +693,9 @@ pub async fn main() -> anyhow::Result<()> { None }; + let cancel_retry_task_handle = task_tree.spawn(cancel_retry_service); + tracing::debug!(task_id=%cancel_retry_task_handle.id(), "Cancel retry service started"); + // Spawn the escrow reconciler service if enabled let escrow_reconciler_stop_handle = if let Some((handle, service)) = escrow_reconciler_handle { let task_handle = task_tree.spawn(service); @@ -763,6 +776,9 @@ pub async fn main() -> anyhow::Result<()> { all_stopped &= stop_service("Chain listener", handle.stop()).await; } + // Stop the cancel retry before worker (it queues replacements) + all_stopped &= stop_service("Cancel retry", cancel_retry_handle.stop()).await; + // Stop escrow reconciler service before the DB pool closes if let Some(handle) = escrow_reconciler_stop_handle { all_stopped &= stop_service("Escrow reconciler", handle.stop()).await; diff --git a/bin/dipper-service/src/network/service/cancel_retry.rs b/bin/dipper-service/src/network/service/cancel_retry.rs index 63cf5d50..3417b30e 100644 --- a/bin/dipper-service/src/network/service/cancel_retry.rs +++ b/bin/dipper-service/src/network/service/cancel_retry.rs @@ -1,10 +1,12 @@ -//! Finishes the cancels dipper starts. An agreement dipper wants ended is marked -//! `Cancelling` before its on-chain cancel goes out; this sweep re-sends the cancel while -//! the chain shows it live, and marks it ended once it can no longer be: `CanceledByRequester`, -//! or `AbandonedByIndexer` for one dipper ended because its indexer stopped serving it. +//! Finishes the cancels dipper starts, marked `Cancelling` before they go out: re-sends each +//! while the chain shows it live, then marks it ended, `AbandonedByIndexer` if its indexer +//! stopped serving it. Runs on its own, as nothing else finishes them. + +use std::{future::Future, sync::Arc, time::Duration}; use dipper_core::time::now_secs; use thegraph_core::alloy::primitives::B256; +use tokio::{sync::mpsc, time::MissedTickBehavior}; use crate::{ cancel_dispatch::{ @@ -12,7 +14,11 @@ use crate::{ }, chain_client::{ChainClient, ChainClientError}, config::IndexingAgreementConfig, - registry::{AgreementRegistry, CancelKind, CancellingAgreement, IndexingAgreement}, + registry::{ + AgreementRegistry, CancelKind, CancellingAgreement, IndexingAgreement, + IndexingRequestRegistry, + }, + worker::service::WorkerQueue, }; /// Failed cancels before dipper alerts an operator and retries the agreement only hourly, so a @@ -23,9 +29,16 @@ pub const MAX_CANCEL_ATTEMPTS: u32 = 10; /// below decides how many it gets through. const BATCH_SIZE: i64 = 50; -/// Time a sweep may take before leaving the rest to the next one: it holds up the chain -/// listener while it runs, and each cancel can wait up to 15 s to be mined. -const SWEEP_BUDGET: std::time::Duration = std::time::Duration::from_secs(30); +/// How often agreements still being cancelled get their cancel retried. +const SWEEP_INTERVAL: Duration = Duration::from_secs(300); + +/// Time a sweep may take before leaving the rest to the next one, as each cancel can wait up +/// to 15 s to be mined. +const SWEEP_BUDGET: Duration = Duration::from_secs(30); + +/// Time allowed to read an ended agreement's request, and to queue its replacement. +const DB_TIMEOUT: Duration = Duration::from_secs(30); +const QUEUE_TIMEOUT: Duration = Duration::from_secs(10); /// Minutes an agreement stays out of the retry after it is marked, so the cancel sent /// when it was marked can be mined first instead of being sent again. One moved back to @@ -36,6 +49,102 @@ const SETTLE_MINUTES: i32 = 2; /// record when and in which transaction it ended, before the retry marks it without them. const LISTENER_GRACE: time::Duration = time::Duration::HOUR; +/// Handle for stopping the cancel retry. +#[derive(Clone)] +pub struct Handle { + tx_stop: mpsc::Sender<()>, +} + +impl Handle { + /// Stop the cancel retry, cutting short a sweep in progress. + pub async fn stop(&self) { + if self.tx_stop.is_closed() { + return; + } + let _ = self.tx_stop.send(()).await; + self.tx_stop.closed().await; + } +} + +/// What the cancel retry needs. +pub struct Ctx { + pub registry: R, + pub chain_client: T, + pub agreement_conf: Arc, + /// Queues the replacement of an agreement ended because its indexer stopped serving it. + pub worker_queue: W, +} + +/// Create the cancel retry. Returns a handle plus a future to spawn, which sweeps at once and +/// then every [`SWEEP_INTERVAL`]. +pub fn new(ctx: Ctx) -> (Handle, impl Future>) +where + R: AgreementRegistry + IndexingRequestRegistry + Send + Sync, + T: ChainClient + Send + Sync, + W: WorkerQueue + Send + Sync, +{ + let (tx_stop, mut rx_stop) = mpsc::channel(1); + let Ctx { + registry, + chain_client, + agreement_conf, + worker_queue, + } = ctx; + let service = async move { + tracing::info!( + interval_secs = SWEEP_INTERVAL.as_secs(), + "cancel retry service started" + ); + let mut timer = tokio::time::interval(SWEEP_INTERVAL); + timer.set_missed_tick_behavior(MissedTickBehavior::Skip); + loop { + tokio::select! { + _ = rx_stop.recv() => break, + _ = timer.tick() => {}, + } + tokio::select! { + _ = rx_stop.recv() => break, + () = async { + retry_cancelling_agreements(®istry, &chain_client, &agreement_conf).await; + replace_ended_abandoned(®istry, &worker_queue).await; + } => {}, + } + } + tracing::debug!("cancel retry service stopped"); + Ok(()) + }; + (Handle { tx_stop }, service) +} + +/// Queue the replacement of each agreement whose indexer stopped serving it once it has ended, +/// whatever ended it: this retry, the chain listener or the indexer. +async fn replace_ended_abandoned(registry: &R, worker_queue: &W) +where + R: AgreementRegistry + IndexingRequestRegistry + Sync, + W: WorkerQueue + Sync, +{ + let ended = match registry + .get_ended_agreements_awaiting_replacement(BATCH_SIZE) + .await + { + Ok(ended) => ended, + Err(err) => { + tracing::warn!(error = %err, "Failed to list ended agreements awaiting replacement"); + return; + } + }; + for agreement in &ended { + super::liveness_checker::replace_abandoned( + agreement, + registry, + worker_queue, + DB_TIMEOUT, + QUEUE_TIMEOUT, + ) + .await; + } +} + /// Retry the cancel of agreements still `Cancelling`. The chain's own latest block time /// decides when an offer that was never accepted no longer can be, so a subgraph that has /// fallen behind doesn't hold that up. @@ -288,21 +397,17 @@ fn log_failed_cancel( ); } -/// How many of an agreement's cancel attempts a failure uses up. A cancel the contract -/// refused, before sending or once mined, that mined without ending the agreement, or that -/// endpoints answered had no receipt counts, and one that can never be sent uses them all. An -/// unreachable chain, including one whose receipt checks all failed, is retried freely. +/// How many of an agreement's cancel attempts a failed cancel uses up. The chain was read just +/// before, so any failure counts, a refusal to send (gas over the cap, signer out of funds) +/// included, except a cancel whose receipt checks all failed, which may have mined unseen. fn failed_attempts(err: &ChainClientError) -> u32 { match err { - ChainClientError::CancelNotConfirmed { .. } - | ChainClientError::TxReverted { .. } - | ChainClientError::TxDropped { - receipt_checked: true, + ChainClientError::TxDropped { + receipt_checked: false, .. - } - | ChainClientError::ContractRevert { .. } => 1, + } => 0, ChainClientError::MissingTermsVersionHash { .. } => MAX_CANCEL_ATTEMPTS, - _ => 0, + _ => 1, } } @@ -314,7 +419,7 @@ mod tests { }; use async_trait::async_trait; - use dipper_core::ids::IndexingAgreementId; + use dipper_core::ids::{IndexingAgreementId, IndexingRequestId}; use dipper_rpc::indexer::indexer_client::sol::RecurringCollectionAgreement; use thegraph_core::alloy::primitives::Address; @@ -323,6 +428,7 @@ mod tests { cancel_dispatch::tests::agreement, chain_client::AgreementOnChain, registry::{IndexingAgreementStatus, StubAgreementRegistry}, + worker::service::JobPriority, }; const DEADLINE: u64 = 1_000; @@ -340,6 +446,9 @@ mod tests { /// The chain listener marks it ended before the retry's own mark lands. listener_ended_it: bool, writes: Mutex>, + /// Ended agreements awaiting replacement, until noted as replaced. + awaiting_replacement: Vec, + replacements_noted: Mutex>, } #[async_trait] @@ -352,6 +461,25 @@ mod tests { ) -> crate::registry::Result> { Ok(self.cancelling.clone()) } + async fn get_ended_agreements_awaiting_replacement( + &self, + _batch_size: i64, + ) -> crate::registry::Result> { + let noted = self.replacements_noted.lock().unwrap(); + Ok(self + .awaiting_replacement + .iter() + .filter(|agreement| !noted.contains(&agreement.id)) + .cloned() + .collect()) + } + async fn mark_replacement_queued( + &self, + id: &IndexingAgreementId, + ) -> crate::registry::Result<()> { + self.replacements_noted.lock().unwrap().push(*id); + Ok(()) + } async fn mark_indexing_agreement_as_canceled_by_requester( &self, id: &IndexingAgreementId, @@ -408,7 +536,8 @@ mod tests { struct MockChain { live: AtomicBool, read_fails: bool, - send_fails: bool, + /// What every send fails with, when set. + send_error: Option ChainClientError>, mined_cancel_reverts: bool, never_mines: bool, receipt_unreadable: bool, @@ -439,8 +568,8 @@ mod tests { _version_hash: B256, _options: u16, ) -> Result, ChainClientError> { - if self.send_fails { - return Err(ChainClientError::RpcError(anyhow::anyhow!("rpc down"))); + if let Some(send_error) = self.send_error { + return Err(send_error()); } if self.reverts_before_sending { return Err(ChainClientError::ContractRevert { @@ -531,6 +660,137 @@ mod tests { retry_cancelling_agreements(registry, chain, &config).await; } + #[async_trait] + impl IndexingRequestRegistry for MockRegistry { + async fn set_indexing_target_candidates( + &self, + _requested_by: Address, + _deployment_id: thegraph_core::DeploymentId, + _deployment_chain_id: u64, + _num_candidates: usize, + ) -> crate::registry::Result { + unimplemented!() + } + async fn get_all_indexing_requests( + &self, + ) -> crate::registry::Result> { + unimplemented!() + } + async fn get_indexing_request_by_id( + &self, + id: &IndexingRequestId, + ) -> crate::registry::Result> { + let agreement = ended_abandoned(); + Ok(Some(crate::registry::IndexingRequest { + id: *id, + created_at: time::OffsetDateTime::now_utc(), + updated_at: time::OffsetDateTime::now_utc(), + status: crate::registry::IndexingRequestStatus::Open, + requested_by: Address::ZERO, + deployment_id: agreement.terms.metadata.subgraph_deployment_id, + deployment_chain_id: agreement.terms.metadata.chain_id, + num_candidates: 3, + })) + } + async fn get_indexing_requests_by_deployment_id( + &self, + _deployment_id: &thegraph_core::DeploymentId, + ) -> crate::registry::Result> { + unimplemented!() + } + async fn get_open_indexing_requests_for_reassessment( + &self, + _min_age_seconds: i64, + _batch_size: i64, + ) -> crate::registry::Result> { + unimplemented!() + } + } + + /// Records the requests it is asked to reassess. + #[derive(Default)] + struct MockQueue(Arc>>); + + #[async_trait] + impl WorkerQueue for MockQueue { + async fn send_indexing_agreement_proposal( + &self, + _candidate_url: url::Url, + _agreement_id: IndexingAgreementId, + _indexing_request_id: IndexingRequestId, + _deployment_id: thegraph_core::DeploymentId, + _deployment_chain_id: u64, + _priority: JobPriority, + ) -> anyhow::Result { + unimplemented!() + } + async fn reassess_indexing_request( + &self, + indexing_request_id: IndexingRequestId, + _deployment_id: thegraph_core::DeploymentId, + _deployment_chain_id: u64, + _num_candidates: usize, + _priority: JobPriority, + ) -> anyhow::Result { + self.0.lock().unwrap().push(indexing_request_id); + Ok(dipper_pgmq::JobId::default()) + } + async fn submit_offer( + &self, + _agreement_id: IndexingAgreementId, + _indexing_request_id: IndexingRequestId, + _indexer_url: url::Url, + _deployment_id: thegraph_core::DeploymentId, + _deployment_chain_id: u64, + _priority: JobPriority, + ) -> anyhow::Result { + unimplemented!() + } + } + + fn ended_abandoned() -> IndexingAgreement { + agreement( + IndexingAgreementStatus::AbandonedByIndexer, + Some(vec![7u8; 32]), + ) + } + + /// Nothing else finishes a cancel, so the retry can't wait on the chain listener, which + /// config can turn off. + #[tokio::test(start_paused = true)] + async fn sweeps_on_its_own_and_replaces_an_abandoned_agreement_once_it_has_ended() { + let ended = ended_abandoned(); + let registry = MockRegistry { + awaiting_replacement: vec![ended.clone()], + ..registry_with_one(true) + }; + let chain = Arc::new(live_chain()); + let queue = MockQueue::default(); + let reassessed = Arc::clone(&queue.0); + let (handle, service) = new(Ctx { + registry, + chain_client: Arc::clone(&chain), + agreement_conf: Arc::new(IndexingAgreementConfig::for_tests()), + worker_queue: queue, + }); + let service = tokio::spawn(service); + + tokio::time::sleep(SWEEP_INTERVAL + Duration::from_secs(1)).await; + handle.stop().await; + + service.await.unwrap().unwrap(); + assert_eq!( + chain.clock_reads.load(Ordering::SeqCst), + 2, + "1 sweep at start, 1 later" + ); + assert_eq!( + *reassessed.lock().unwrap(), + vec![ended.indexing_request_id], + "queued once, then noted as replaced" + ); + } + #[tokio::test] async fn waits_for_the_next_sweep_when_the_chain_time_cannot_be_read() { let registry = registry_with_one(true); @@ -804,23 +1064,42 @@ mod tests { #[tokio::test] async fn an_unreachable_chain_neither_sends_nor_counts_an_attempt() { - for chain in [ - MockChain { - read_fails: true, - ..live_chain() - }, - MockChain { - send_fails: true, - ..live_chain() + let registry = registry_with_one(true); + let chain = MockChain { + read_fails: true, + ..live_chain() + }; + + retry(®istry, &chain, 0).await; + + assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 0); + assert_eq!(registry.attempts.load(Ordering::SeqCst), 0); + assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn counts_a_cancel_that_could_not_be_sent() { + // The chain was just read, so a send that fails every time (gas over the cap, signer + // out of funds) is no outage, and only counting it reaches the alert. + let refusals: [fn() -> ChainClientError; 2] = [ + || ChainClientError::SubmitFailed(anyhow::anyhow!("Gas price exceeds maximum")), + || { + ChainClientError::RpcError(anyhow::anyhow!( + "Gas estimation failed: insufficient funds" + )) }, - ] { + ]; + for err in refusals { let registry = registry_with_one(true); + let chain = MockChain { + send_error: Some(err), + ..live_chain() + }; retry(®istry, &chain, 0).await; assert_eq!(chain.cancels_sent.load(Ordering::SeqCst), 0); - assert_eq!(registry.attempts.load(Ordering::SeqCst), 0); - assert!(registry.marked_cancelled.lock().unwrap().is_empty()); + assert_eq!(registry.attempts.load(Ordering::SeqCst), 1); } } diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index 15c00d57..aa8b761e 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -82,10 +82,6 @@ const SWEEP_BATCH_SIZE: i64 = 1000; /// crash-recovery; the steady-state fan-out fires from finalize on a /// fresh accept, so per-poll execution is wasted DB work. const SWEEP_POLLS: u64 = 60; -/// How often agreements still being cancelled get their cancel retried, whichever rate the -/// listener polls at: just under the slow poll interval, so every slow poll retries. -const CANCEL_RETRY_INTERVAL: Duration = Duration::from_secs(290); - /// Handle for controlling the chain listener service lifecycle #[derive(Clone)] pub struct Handle { @@ -226,7 +222,6 @@ where // Starts at SWEEP_POLLS so the first poll runs the sweep, // recovering any pre-startup orphans. let mut polls_since_sweep: u64 = SWEEP_POLLS; - let mut last_cancel_retry: Option = None; // Pause the event sweeps after a Kafka send failure, backing off from one // poll interval up to the idle interval, so a hung broker cannot stall the // poll loop on every iteration. @@ -292,18 +287,6 @@ where tokio::time::sleep(backoff).await; } - // Ahead of the drain, which ends the poll early while the subgraph is down: - // finishing a cancel needs only the chain. - if last_cancel_retry.is_none_or(|at| at.elapsed() >= CANCEL_RETRY_INTERVAL) { - last_cancel_retry = Some(Instant::now()); - super::cancel_retry::retry_cancelling_agreements( - ®istry, - &chain_client, - &agreement_conf, - ) - .await; - } - let outcome = match drain_once( &mut cursor, &mut last_persisted_timestamp, @@ -1156,7 +1139,7 @@ where /// /// Called from the Created -> AcceptedOnChain and Expired -> AcceptedOnChain /// transitions. Each replaced agreement is marked `Cancelling` and its on-chain cancel -/// sent once; the cancel retry in this listener finishes any that don't end at once. +/// sent once; the cancel retry finishes any that don't end at once. /// A pending row is deleted once its agreement is marked; a failed mark keeps it for retry. async fn execute_pending_cancellations( agreement_id: &IndexingAgreementId, diff --git a/bin/dipper-service/src/network/service/liveness_checker.rs b/bin/dipper-service/src/network/service/liveness_checker.rs index e2dfc5d7..a95e2682 100644 --- a/bin/dipper-service/src/network/service/liveness_checker.rs +++ b/bin/dipper-service/src/network/service/liveness_checker.rs @@ -43,7 +43,7 @@ use tokio::{sync::mpsc, time::MissedTickBehavior}; use url::Url; use crate::{ - cancel_dispatch::CancelReason, + cancel_dispatch::{CancelReason, CancelStarted}, chain_client::ChainClient, config::LivenessCheckerConfig, network::provider::NetworkProviderService, @@ -481,15 +481,10 @@ async fn record_progress( } } -/// End a stale agreement and queue a reassessment to replace it. It is marked `Cancelling` -/// before any cancel is sent, so the cancel retry finishes one that fails here, and it ends -/// `AbandonedByIndexer` once the chain confirms dipper's cancel. +/// End a stale agreement and queue a reassessment to replace it once it can't be paid. It is +/// marked `Cancelling`, and as awaiting replacement, before any cancel is sent, so the cancel +/// retry finishes one that fails here and queues its replacement once it has ended. #[allow(clippy::too_many_arguments)] -#[expect( - clippy::cognitive_complexity, - clippy::too_many_lines, - reason = "predates this lint; fix when next touched" -)] async fn cancel_and_reassess( agreement: &IndexingAgreement, registry: &R, @@ -512,8 +507,57 @@ async fn cancel_and_reassess( ); return; } + let Some(started) = + start_abandoned_cancel(agreement, registry, chain_client, agreement_conf).await + else { + return; + }; + forget_replaced_agreement(agreement, registry).await; + if started == CancelStarted::MayBeLive { + tracing::info!( + agreement_id = %agreement.id, + "Stale agreement may still be paid; the cancel retry replaces it once it has ended" + ); + return; + } + replace_abandoned(agreement, registry, worker_queue, db_timeout, queue_timeout).await; +} + +/// Queue the replacement of an agreement whose indexer stopped serving it, and note it queued +/// so it isn't queued again. +pub(crate) async fn replace_abandoned( + agreement: &IndexingAgreement, + registry: &R, + worker_queue: &W, + db_timeout: Duration, + queue_timeout: Duration, +) where + R: AgreementRegistry + IndexingRequestRegistry + Sync, + W: WorkerQueue + Sync, +{ + if !queue_replacement(agreement, registry, worker_queue, db_timeout, queue_timeout).await { + return; + } + if let Err(err) = registry.mark_replacement_queued(&agreement.id).await { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "Failed to note an abandoned agreement's replacement queued; it may be queued again" + ); + } +} - // 1. Start the cancel +/// Mark a stale agreement abandoned and send its cancel; `None`, logged, when it isn't marked. +async fn start_abandoned_cancel( + agreement: &IndexingAgreement, + registry: &R, + chain_client: &C, + agreement_conf: &crate::config::IndexingAgreementConfig, +) -> Option +where + R: AgreementRegistry + Sync, + C: ChainClient, +{ match crate::cancel_dispatch::start_cancel( registry, chain_client, @@ -523,18 +567,21 @@ async fn cancel_and_reassess( ) .await { - Ok(started) => tracing::info!( - agreement_id = %agreement.id, - ?started, - reason = "indexer_stale", - "Cancelling stale agreement" - ), + Ok(started) => { + tracing::info!( + agreement_id = %agreement.id, + ?started, + reason = "indexer_stale", + "Cancelling stale agreement" + ); + Some(started) + } Err(crate::registry::Error::NoRecordsUpdated) => { tracing::debug!( agreement_id = %agreement.id, "Stale agreement already ended or being cancelled" ); - return; + None } Err(err) => { tracing::error!( @@ -542,12 +589,17 @@ async fn cancel_and_reassess( error = %err, "failed to mark stale agreement cancelling, will retry next cycle" ); - return; + None } } +} - // Clean up pending cancellations: if this abandoned agreement was a - // replacement, the old agreement it was replacing should stay active. +/// Drop the pending cancellations an abandoned agreement holds as a replacement, so the +/// agreement it was to replace stays active. +async fn forget_replaced_agreement( + agreement: &IndexingAgreement, + registry: &R, +) { if let Err(err) = registry .delete_pending_cancellations_by_new_agreement(agreement.id) .await @@ -558,41 +610,26 @@ async fn cancel_and_reassess( "failed to clean up pending cancellations for abandoned agreement" ); } +} - // 2. Fetch the indexing request for num_candidates - let request = match tokio::time::timeout( - db_timeout, - registry.get_indexing_request_by_id(&agreement.indexing_request_id), - ) - .await - { - Ok(Ok(Some(r))) => r, - Ok(Ok(None)) => { - tracing::warn!( - agreement_id = %agreement.id, - indexing_request_id = %agreement.indexing_request_id, - "indexing request not found for abandoned agreement" - ); - return; - } - Ok(Err(err)) => { - tracing::warn!( - agreement_id = %agreement.id, - error = %err, - "failed to fetch indexing request for abandoned agreement" - ); - return; - } - Err(_) => { - tracing::warn!( - agreement_id = %agreement.id, - "timeout fetching indexing request for abandoned agreement" - ); - return; - } +/// Queue a reassessment of an abandoned agreement's request, which replaces its indexer. True +/// once nothing is left to do: queued, or its request is gone. +async fn queue_replacement( + agreement: &IndexingAgreement, + registry: &R, + worker_queue: &W, + db_timeout: Duration, + queue_timeout: Duration, +) -> bool +where + R: IndexingRequestRegistry + Sync, + W: WorkerQueue + Sync, +{ + let request = match abandoned_request(agreement, registry, db_timeout).await { + Ok(Some(request)) => request, + Ok(None) => return true, + Err(()) => return false, }; - - // 3. Queue reassessment let push_result = tokio::time::timeout( queue_timeout, worker_queue.reassess_indexing_request( @@ -613,6 +650,7 @@ async fn cancel_and_reassess( indexing_request_id = %agreement.indexing_request_id, "queued reassessment for abandoned agreement" ); + true } Ok(Err(err)) => { tracing::warn!( @@ -620,12 +658,54 @@ async fn cancel_and_reassess( error = %err, "failed to queue reassessment for abandoned agreement" ); + false } Err(_) => { tracing::warn!( agreement_id = %agreement.id, "timeout queuing reassessment for abandoned agreement" ); + false + } + } +} + +/// The request an abandoned agreement served, `None` when it is gone; `Err`, logged, when it +/// can't be read. +async fn abandoned_request( + agreement: &IndexingAgreement, + registry: &R, + db_timeout: Duration, +) -> Result, ()> { + match tokio::time::timeout( + db_timeout, + registry.get_indexing_request_by_id(&agreement.indexing_request_id), + ) + .await + { + Ok(Ok(Some(r))) => Ok(Some(r)), + Ok(Ok(None)) => { + tracing::warn!( + agreement_id = %agreement.id, + indexing_request_id = %agreement.indexing_request_id, + "indexing request not found for abandoned agreement" + ); + Ok(None) + } + Ok(Err(err)) => { + tracing::warn!( + agreement_id = %agreement.id, + error = %err, + "failed to fetch indexing request for abandoned agreement" + ); + Err(()) + } + Err(_) => { + tracing::warn!( + agreement_id = %agreement.id, + "timeout fetching indexing request for abandoned agreement" + ); + Err(()) } } } @@ -873,6 +953,8 @@ mod tests { /// Ids passed to `record_cancel_audit` -- the signal the handler drives /// the terminated event (the chain_listener sweep emits from this audit). cancel_audits: Arc>>, + /// Ids noted as having their replacement queued. + replacements_noted: Arc>>, } struct MockRegistry { @@ -925,6 +1007,11 @@ mod tests { Ok(()) } + async fn mark_replacement_queued(&self, id: &IndexingAgreementId) -> RegistryResult<()> { + self.calls.replacements_noted.lock().unwrap().push(*id); + Ok(()) + } + async fn mark_indexing_agreement_as_canceled_by_requester( &self, id: &IndexingAgreementId, @@ -1445,12 +1532,17 @@ mod tests { calls.reassessments.lock().unwrap().as_slice(), &[agreement.indexing_request_id] ); + assert_eq!( + calls.replacements_noted.lock().unwrap().as_slice(), + &[agreement.id], + "so the cancel retry doesn't queue it again" + ); } #[tokio::test] - async fn leaves_a_failed_cancel_to_the_retry_and_still_reassesses() { - // Marked cancelling, the agreement is out of the checker's sight, and its indexer out - // of selection until the cancel lands, so it is replaced now and the retry finishes it. + async fn holds_back_the_replacement_while_a_failed_cancel_leaves_it_paid() { + // Replacing it now would pay both indexers until the retry lands the cancel, which then + // queues the replacement. let agreement = stale_agreement(); let calls = MockCalls::default(); let registry = MockRegistry::new(calls.clone(), agreement.clone()); @@ -1461,10 +1553,29 @@ mod tests { assert_eq!(calls.abandoning.lock().unwrap().as_slice(), &[agreement.id]); assert!(calls.ended.lock().unwrap().is_empty()); assert!(calls.cancel_audits.lock().unwrap().is_empty()); + assert!(calls.reassessments.lock().unwrap().is_empty()); + assert!(calls.replacements_noted.lock().unwrap().is_empty()); + } + + #[tokio::test] + async fn replaces_at_once_a_stale_agreement_the_chain_shows_ended() { + let agreement = stale_agreement(); + let calls = MockCalls::default(); + let registry = MockRegistry::new(calls.clone(), agreement.clone()); + let chain = MockChainClient::rpc_error(calls.clone()); + chain.live.store(false, std::sync::atomic::Ordering::SeqCst); + + end_stale(&agreement, ®istry, &chain).await; + + assert!(calls.chain_cancels.lock().unwrap().is_empty()); assert_eq!( calls.reassessments.lock().unwrap().as_slice(), &[agreement.indexing_request_id] ); + assert_eq!( + calls.replacements_noted.lock().unwrap().as_slice(), + &[agreement.id] + ); } #[tokio::test] diff --git a/bin/dipper-service/src/registry.rs b/bin/dipper-service/src/registry.rs index e49d8c4c..86ac2a7f 100644 --- a/bin/dipper-service/src/registry.rs +++ b/bin/dipper-service/src/registry.rs @@ -690,6 +690,27 @@ impl AgreementRegistry for RegistryProvider { .collect()) } + async fn get_ended_agreements_awaiting_replacement( + &self, + batch_size: i64, + ) -> RegistryResult> { + Ok(self + .inner + .get_ended_agreements_awaiting_replacement(batch_size) + .await? + .into_iter() + .map(IndexingAgreement::try_from) + .filter_map(filter_map_with_logging) + .collect()) + } + + async fn mark_replacement_queued(&self, id: &IndexingAgreementId) -> RegistryResult<()> { + self.inner + .mark_replacement_queued(id) + .await + .map_err(Into::into) + } + async fn update_agreement_sync_progress( &self, id: &IndexingAgreementId, diff --git a/bin/dipper-service/src/registry/agreement.rs b/bin/dipper-service/src/registry/agreement.rs index 95da2b73..10cc02f8 100644 --- a/bin/dipper-service/src/registry/agreement.rs +++ b/bin/dipper-service/src/registry/agreement.rs @@ -497,6 +497,16 @@ pub trait AgreementRegistry { batch_size: i64, ) -> RegistryResult>; + /// Agreements whose indexer stopped serving them that have ended, longest ended first, + /// whose replacement is yet to be queued. + async fn get_ended_agreements_awaiting_replacement( + &self, + batch_size: i64, + ) -> RegistryResult>; + + /// Note that an agreement's replacement has been queued. + async fn mark_replacement_queued(&self, id: &IndexingAgreementId) -> RegistryResult<()>; + /// Update the sync progress for an agreement. /// /// Called when the liveness checker observes the block height has changed diff --git a/bin/dipper-service/src/registry/agreement_stub.rs b/bin/dipper-service/src/registry/agreement_stub.rs index d9092a6d..cc3c717f 100644 --- a/bin/dipper-service/src/registry/agreement_stub.rs +++ b/bin/dipper-service/src/registry/agreement_stub.rs @@ -225,6 +225,17 @@ pub trait StubAgreementRegistry: Send + Sync { unimplemented!("get_agreements_pending_chain_cancel") } + async fn get_ended_agreements_awaiting_replacement( + &self, + _batch_size: i64, + ) -> Result> { + unimplemented!("get_ended_agreements_awaiting_replacement") + } + + async fn mark_replacement_queued(&self, _id: &IndexingAgreementId) -> Result<()> { + unimplemented!("mark_replacement_queued") + } + async fn update_agreement_sync_progress( &self, _id: &IndexingAgreementId, @@ -533,6 +544,17 @@ impl AgreementRegistry for T { StubAgreementRegistry::get_agreements_pending_chain_cancel(self, batch_size).await } + async fn get_ended_agreements_awaiting_replacement( + &self, + batch_size: i64, + ) -> Result> { + StubAgreementRegistry::get_ended_agreements_awaiting_replacement(self, batch_size).await + } + + async fn mark_replacement_queued(&self, id: &IndexingAgreementId) -> Result<()> { + StubAgreementRegistry::mark_replacement_queued(self, id).await + } + async fn update_agreement_sync_progress( &self, id: &IndexingAgreementId, diff --git a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs index e07f5ba5..584e08d3 100644 --- a/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs +++ b/bin/dipper-service/src/worker/handlers/reassess_indexing_request.rs @@ -1827,8 +1827,8 @@ mod lifecycle_event_tests { } #[tokio::test] - async fn an_agreement_whose_cancel_fails_is_left_cancelling_for_the_listener() { - // The chain listener retries the cancel of every cancelling agreement until + async fn an_agreement_whose_cancel_fails_is_left_cancelling_for_the_retry() { + // The cancel retry re-sends the cancel of every cancelling agreement until // the chain shows it ended, so nothing is queued here. for status in [ IndexingAgreementStatus::Created, @@ -1910,7 +1910,7 @@ enum Unpaired { } /// Take an old agreement out of the target group. One that may be live on-chain is cancelled -/// there too, which the chain listener retries until it ends. +/// there too, which the cancel retry re-sends until it ends. async fn cancel_unpaired( ctx: &Ctx, agreement: &crate::registry::IndexingAgreement, @@ -1944,7 +1944,10 @@ where Ok(crate::cancel_dispatch::CancelStarted::Ended) => { Unpaired::Moved("CANCELED_BY_REQUESTER") } - Ok(crate::cancel_dispatch::CancelStarted::Cancelling) => Unpaired::Moved("CANCELLING"), + Ok( + crate::cancel_dispatch::CancelStarted::NotLive + | crate::cancel_dispatch::CancelStarted::MayBeLive, + ) => Unpaired::Moved("CANCELLING"), Err(err) => unmarked(agreement, &err), } } diff --git a/bin/dipper-service/src/worker/handlers/submit_offer.rs b/bin/dipper-service/src/worker/handlers/submit_offer.rs index f5ee5088..83867064 100644 --- a/bin/dipper-service/src/worker/handlers/submit_offer.rs +++ b/bin/dipper-service/src/worker/handlers/submit_offer.rs @@ -283,8 +283,8 @@ fn dipper_cancelled(status: IndexingAgreementStatus) -> bool { } /// The agreement, if it was cancelled after this job's status check. After a failed read -/// the job still finishes: retrying would send the offer again, and the chain listener's -/// cancel retry withdraws the offer of an agreement left cancelling. +/// the job still finishes: retrying would send the offer again, and the cancel retry +/// withdraws the offer of an agreement left cancelling. async fn cancelled_meanwhile( registry: &R, agreement_id: &IndexingAgreementId, diff --git a/dipper-pgregistry/migrations/20261008000000_add_replacement_pending.sql b/dipper-pgregistry/migrations/20261008000000_add_replacement_pending.sql new file mode 100644 index 00000000..bf58fbb4 --- /dev/null +++ b/dipper-pgregistry/migrations/20261008000000_add_replacement_pending.sql @@ -0,0 +1,8 @@ +-- replacement_pending: the agreement's indexer stopped serving it and dipper has yet to queue a +-- reassessment to replace it, which waits until the agreement can no longer be paid. +ALTER TABLE dipper_reg_indexing_agreements + ADD COLUMN replacement_pending BOOLEAN NOT NULL DEFAULT false; + +CREATE INDEX idx_indexing_agreements_replacement_pending + ON dipper_reg_indexing_agreements (updated_at) + WHERE replacement_pending; diff --git a/dipper-pgregistry/src/postgres.rs b/dipper-pgregistry/src/postgres.rs index d1bc8194..d2bce6bd 100644 --- a/dipper-pgregistry/src/postgres.rs +++ b/dipper-pgregistry/src/postgres.rs @@ -989,6 +989,7 @@ impl PgRegistry { SET status = $1, abandoned = true, + replacement_pending = true, updated_at = timezone('UTC', now()) WHERE id = $2 AND status = $3 "#, @@ -1858,6 +1859,56 @@ impl PgRegistry { .map_err(Into::into) } + /// Agreements whose indexer stopped serving them that have ended, longest ended first, + /// whose replacement is yet to be queued. + pub async fn get_ended_agreements_awaiting_replacement( + &self, + batch_size: i64, + ) -> Result, Error> { + sqlx::query_as( + r#" + SELECT + id, + nonce_uuid, + created_at, + updated_at, + status, + indexing_request_id, + deployment_id, + indexer_id, + indexer_url, + terms, + last_block_height, + last_progress_at, + rejection_reason, + terms_version_hash + FROM dipper_reg_indexing_agreements + WHERE replacement_pending AND status <> $1 + ORDER BY updated_at ASC + LIMIT $2 + "#, + ) + .bind(IndexingAgreementStatus::Cancelling) + .bind(batch_size) + .fetch_all(&self.pool) + .await + .map_err(Into::into) + } + + /// Note that an agreement's replacement has been queued. + pub async fn mark_replacement_queued( + &self, + agreement_id: &IndexingAgreementId, + ) -> Result<(), Error> { + sqlx::query( + "UPDATE dipper_reg_indexing_agreements SET replacement_pending = false WHERE id = $1", + ) + .bind(agreement_id) + .execute(&self.pool) + .await?; + Ok(()) + } + /// Update the sync progress for an agreement. /// /// Called when the liveness checker observes the block height has changed diff --git a/dipper-pgregistry/tests/it_registry_postgres.rs b/dipper-pgregistry/tests/it_registry_postgres.rs index e5bd0f3a..88b993ef 100644 --- a/dipper-pgregistry/tests/it_registry_postgres.rs +++ b/dipper-pgregistry/tests/it_registry_postgres.rs @@ -3884,3 +3884,53 @@ async fn an_offer_mined_after_the_cancel_began_keeps_its_hash() { .expect_err("an ended agreement doesn't take the hash"); assert!(matches!(err, Error::NoRecordsUpdated)); } + +/// An agreement whose indexer stopped serving it is replaced only once it can't be paid, and +/// only once, whatever ended it. +#[tokio::test] +async fn an_abandoned_agreement_awaits_replacement_once_ended_until_queued() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let id = fixture_agreement(0xaa); + sqlx::query("UPDATE dipper_reg_indexing_agreements SET status = $1 WHERE id = $2") + .bind(IndexingAgreementStatus::AcceptedOnChain) + .bind(id) + .execute(&db) + .await + .expect("accept the agreement"); + let registry = PgRegistry::new(db); + registry + .mark_indexing_agreement_as_abandoning(&id) + .await + .expect("mark abandoning"); + let awaiting = async || { + registry + .get_ended_agreements_awaiting_replacement(10) + .await + .expect("awaiting query") + .into_iter() + .map(|agreement| agreement.id) + .collect::>() + }; + assert!( + awaiting().await.is_empty(), + "still cancelling, so maybe paid" + ); + + registry + .mark_indexing_agreement_as_canceled_by_requester(&id) + .await + .expect("end it"); + assert_eq!(awaiting().await, vec![id]); + + registry + .mark_replacement_queued(&id) + .await + .expect("note it queued"); + assert!(awaiting().await.is_empty()); +} From 4379086b7f9e6a01e95c24d97298c3e999dc34be Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Fri, 9 Oct 2026 19:03:43 +0100 Subject: [PATCH 24/26] fix(rpc): stop a hung endpoint from holding up reads (#738) Until 2 endpoints agree on the chain's newest block, a read first asks every endpoint for it. 1 endpoint that never answered held that read for the full request timeout, once a minute. Each ask now gives up after 3 seconds and the check goes on with the answers it has. --- bin/dipper-service/src/chain_client/client.rs | 5 ++- .../src/chain_client/rpc_provider.rs | 43 +++++++++++++++++-- 2 files changed, 43 insertions(+), 5 deletions(-) diff --git a/bin/dipper-service/src/chain_client/client.rs b/bin/dipper-service/src/chain_client/client.rs index 69122dea..9376c63c 100644 --- a/bin/dipper-service/src/chain_client/client.rs +++ b/bin/dipper-service/src/chain_client/client.rs @@ -84,6 +84,9 @@ const FAR_AHEAD_OF_A_SEEN_BLOCK: &str = "too far ahead of the newest block seen" /// block it has seen, while that block is unconfirmed. const CROSS_CHECK_INTERVAL: Duration = Duration::from_secs(60); +/// An answer takes well under a second, so this only bounds how long a hung endpoint holds a read. +const CROSS_CHECK_DEADLINE: Duration = Duration::from_secs(3); + /// The newest block dipper has seen, from reads, receipts and latest-block lookups, and when it /// last moved. It never goes backwards, so a lagging endpoint can't show state from before it. /// Blocks too far ahead of it are refused only once it is confirmed, by the receipt for one of @@ -892,7 +895,7 @@ impl AlloyChainClient { if pool.endpoint_count() < 2 || !self.seen_block().cross_check_due(Instant::now()) { return; } - match agreed_head(pool.latest_blocks().await) { + match agreed_head(pool.latest_blocks(CROSS_CHECK_DEADLINE).await) { Some(head) => self.seen_block().confirm(head, Instant::now()), None => tracing::warn!( "No 2 RPC endpoints agree on the chain's latest block; reads aren't checked \ diff --git a/bin/dipper-service/src/chain_client/rpc_provider.rs b/bin/dipper-service/src/chain_client/rpc_provider.rs index d6e8bf56..f05bd1f7 100644 --- a/bin/dipper-service/src/chain_client/rpc_provider.rs +++ b/bin/dipper-service/src/chain_client/rpc_provider.rs @@ -176,13 +176,18 @@ impl RpcProviderPool { self.providers.len() } - /// Each endpoint's latest block, all asked at once with no retries. Endpoints that fail are - /// left out, so dipper can see whether the ones that answer agree. - pub async fn latest_blocks(&self) -> Vec { + /// Each endpoint's latest block, all asked at once with no retries. Endpoints that fail, or + /// don't answer within `deadline`, are left out, so dipper can see whether the rest agree. + pub async fn latest_blocks(&self, deadline: Duration) -> Vec { let mut asks = tokio::task::JoinSet::new(); for url in &self.providers { let (http, url) = (self.http.clone(), url.clone()); - asks.spawn(async move { (endpoint_name(&url), latest_block(http, &url).await) }); + asks.spawn(async move { + let head = tokio::time::timeout(deadline, latest_block(http, &url)) + .await + .unwrap_or_else(|_| Err(format!("no answer within {deadline:?}"))); + (endpoint_name(&url), head) + }); } let mut heads = Vec::with_capacity(self.providers.len()); while let Some(answer) = asks.join_next().await { @@ -618,6 +623,36 @@ mod tests { ); } + /// Reads wait on the cross-check, so 1 endpoint that never answers must not hold it for the + /// whole request timeout. The healthy endpoint's head still counts. + #[tokio::test] + async fn a_cross_check_stops_waiting_for_a_hung_endpoint() { + let healthy = server_answering_block(0x2a).await; + let hung = MockServer::start().await; + Mock::given(method("POST")) + .respond_with(ResponseTemplate::new(200).set_delay(Duration::from_secs(30))) + .mount(&hung) + .await; + let pool = RpcProviderPool::new( + vec![ + healthy.uri().parse().expect("healthy URL"), + hung.uri().parse().expect("hung URL"), + ], + Duration::from_secs(60), + 0, + ) + .expect("pool"); + + let heads = tokio::time::timeout( + Duration::from_secs(2), + pool.latest_blocks(Duration::from_millis(200)), + ) + .await + .expect("the cross-check should give up on the hung endpoint at its deadline"); + + assert_eq!(heads, vec![0x2a]); + } + /// An endpoint that answers can only describe its own refusal, so it never repeats the /// URL. One that never answers is described by the HTTP client instead, which says which /// URL it was reaching for, and that is where the key sits. Nothing listens on port 1. From d262db874f671d4ecda56b715f1f3f671d4f6d5b Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Fri, 9 Oct 2026 19:03:43 +0100 Subject: [PATCH 25/26] perf(listener): record a replayed accept and cancel in 1 write (#739) When the listener re-reads an agreement already ended that was accepted on-chain, it recorded the end, then the accept, then the end again: 3 writes, ordered so nothing saw the accept on its own. It now records both in 1 write, which nothing can see half-done, so the order no longer matters. --- .../src/network/service/chain_listener.rs | 73 ++++++++----------- bin/dipper-service/src/registry.rs | 22 ++++++ bin/dipper-service/src/registry/agreement.rs | 15 ++++ .../src/registry/agreement_stub.rs | 34 +++++++++ dipper-pgregistry/src/postgres.rs | 47 ++++++++++++ .../tests/it_registry_postgres.rs | 61 ++++++++++++++++ 6 files changed, 211 insertions(+), 41 deletions(-) diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index aa8b761e..bfc83eb4 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -931,10 +931,9 @@ where /// Record the accept and cancel of an agreement dipper had already marked /// cancelled that went live on-chain first, so its accepted and terminated -/// events go out. Cancel first: the terminated sweep waits only for the accept. +/// events go out. Both go in 1 write, as the terminated sweep waits only for the accept. /// Existing values win, so an agreement dipper already recorded is unchanged, -/// except an end recorded before the accept, such as its offer's withdrawal: the -/// cancel is recorded again once the accept is, so that end gives way to this one. +/// except an end recorded before the accept, such as its offer's withdrawal, which gives way. async fn record_accept_and_cancel_from_chain( snapshot: &AgreementStateSnapshot, agreement: &IndexingAgreement, @@ -943,36 +942,16 @@ async fn record_accept_and_cancel_from_chain( if snapshot.accepted_at == 0 { return; } - let canceled_by = snapshot.canceled_by.to_string(); - let recorded = match registry - .record_cancel_audit( + let recorded = registry + .record_accept_and_cancel_audit( &agreement.id, + snapshot.accepted_at, + &snapshot.accepted_tx, snapshot.canceled_at, - &canceled_by, + &snapshot.canceled_by.to_string(), Some(&snapshot.canceled_tx), ) - .await - { - Ok(()) => { - registry - .record_accepted_audit(&agreement.id, snapshot.accepted_at, &snapshot.accepted_tx) - .await - } - Err(err) => Err(err), - }; - let recorded = match recorded { - Ok(()) => { - registry - .record_cancel_audit( - &agreement.id, - snapshot.canceled_at, - &canceled_by, - Some(&snapshot.canceled_tx), - ) - .await - } - Err(err) => Err(err), - }; + .await; if let Err(err) = recorded { tracing::warn!( agreement_id = %agreement.id, @@ -1899,9 +1878,9 @@ mod tests { /// Ids passed to `record_cancel_audit` -- the signal a cancel path drives /// the terminated event (the sweep emits from this audit). recorded_cancel_audit: Vec, - /// Every audit write in order, as ("cancel" | "accept", id). + /// Every audit write in order, as ("cancel" | "accept" | "accept and cancel", id). audit_writes: Vec<(&'static str, IndexingAgreementId)>, - /// When true, `record_cancel_audit` fails. + /// When true, writes that record a cancel fail. fail_cancel_audit: bool, pending_cancellations: std::collections::HashMap< IndexingAgreementId, @@ -2219,6 +2198,25 @@ mod tests { Ok(()) } + async fn record_accept_and_cancel_audit( + &self, + agreement_id: &IndexingAgreementId, + _accepted_at: u64, + _accepted_tx: &str, + _canceled_at: u64, + _canceled_by: &str, + _canceled_tx: Option<&str>, + ) -> RegistryResult<()> { + let mut state = self.state.lock().unwrap(); + if state.fail_cancel_audit { + return Err(crate::registry::Error::NoRecordsUpdated); + } + state + .audit_writes + .push(("accept and cancel", *agreement_id)); + Ok(()) + } + async fn record_accepted_audit( &self, agreement_id: &IndexingAgreementId, @@ -3128,8 +3126,7 @@ mod tests { async fn test_reconcile_records_accept_and_cancel_of_cancelled_agreement_that_went_live() { // Dipper had marked the agreement cancelled, but it was accepted on-chain // before being ended there. Recording both lets the accepted and terminated - // events go out; the cancel goes first because the terminated sweep only - // waits for the accept, and again after it, to replace an end from before it. + // events go out, in 1 write, as the terminated sweep only waits for the accept. let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); @@ -3150,18 +3147,12 @@ mod tests { assert!(result.is_ok()); assert_eq!( registry.audit_writes(), - vec![ - ("cancel", agreement_id), - ("accept", agreement_id), - ("cancel", agreement_id) - ] + vec![("accept and cancel", agreement_id)] ); } #[tokio::test] - async fn test_reconcile_records_no_accept_when_the_cancel_record_fails() { - // An accept recorded without its cancel would let the terminated event go - // out with fallback cancel fields. + async fn test_reconcile_survives_a_failed_accept_and_cancel_record() { let registry = MockRegistry::new(); let chain_client = MockChainClient::default(); let agreement_id = IndexingAgreementId::from_bytes(rand::random()); diff --git a/bin/dipper-service/src/registry.rs b/bin/dipper-service/src/registry.rs index 86ac2a7f..71470cda 100644 --- a/bin/dipper-service/src/registry.rs +++ b/bin/dipper-service/src/registry.rs @@ -626,6 +626,28 @@ impl AgreementRegistry for RegistryProvider { Ok(()) } + async fn record_accept_and_cancel_audit( + &self, + agreement_id: &IndexingAgreementId, + accepted_at: u64, + accepted_tx: &str, + canceled_at: u64, + canceled_by: &str, + canceled_tx: Option<&str>, + ) -> RegistryResult<()> { + self.inner + .record_accept_and_cancel_audit( + agreement_id, + accepted_at, + accepted_tx, + canceled_at, + canceled_by, + canceled_tx, + ) + .await?; + Ok(()) + } + async fn get_expired_created_agreements( &self, batch_size: i64, diff --git a/bin/dipper-service/src/registry/agreement.rs b/bin/dipper-service/src/registry/agreement.rs index 10cc02f8..7f8618b1 100644 --- a/bin/dipper-service/src/registry/agreement.rs +++ b/bin/dipper-service/src/registry/agreement.rs @@ -427,6 +427,21 @@ pub trait AgreementRegistry { Ok(()) } + /// Record an agreement's accept and its end together, in 1 write, so nothing reads one + /// without the other. Default no-op so mocks need not override. + #[allow(clippy::too_many_arguments)] + async fn record_accept_and_cancel_audit( + &self, + _agreement_id: &IndexingAgreementId, + _accepted_at: u64, + _accepted_tx: &str, + _canceled_at: u64, + _canceled_by: &str, + _canceled_tx: Option<&str>, + ) -> RegistryResult<()> { + Ok(()) + } + /// Record the cancel audit payload for a dipper-initiated cancel so the /// emission sweep can populate the `terminated` event fields. Default no-op /// so mocks need not override. diff --git a/bin/dipper-service/src/registry/agreement_stub.rs b/bin/dipper-service/src/registry/agreement_stub.rs index cc3c717f..dd2950b4 100644 --- a/bin/dipper-service/src/registry/agreement_stub.rs +++ b/bin/dipper-service/src/registry/agreement_stub.rs @@ -327,6 +327,19 @@ pub trait StubAgreementRegistry: Send + Sync { ) -> Result<()> { Ok(()) } + + #[allow(clippy::too_many_arguments)] + async fn record_accept_and_cancel_audit( + &self, + _agreement_id: &IndexingAgreementId, + _accepted_at: u64, + _accepted_tx: &str, + _canceled_at: u64, + _canceled_by: &str, + _canceled_tx: Option<&str>, + ) -> Result<()> { + Ok(()) + } } // Every stub is a full AgreementRegistry: each method delegates to the stub @@ -650,4 +663,25 @@ impl AgreementRegistry for T { ) .await } + + async fn record_accept_and_cancel_audit( + &self, + agreement_id: &IndexingAgreementId, + accepted_at: u64, + accepted_tx: &str, + canceled_at: u64, + canceled_by: &str, + canceled_tx: Option<&str>, + ) -> Result<()> { + StubAgreementRegistry::record_accept_and_cancel_audit( + self, + agreement_id, + accepted_at, + accepted_tx, + canceled_at, + canceled_by, + canceled_tx, + ) + .await + } } diff --git a/dipper-pgregistry/src/postgres.rs b/dipper-pgregistry/src/postgres.rs index d2bce6bd..3c3fdfd2 100644 --- a/dipper-pgregistry/src/postgres.rs +++ b/dipper-pgregistry/src/postgres.rs @@ -1591,6 +1591,53 @@ impl PgRegistry { Ok(()) } + /// Record an agreement's accept and its end together, as the chain shows them, in 1 write, + /// with the rules of [`Self::record_accepted_audit`] and [`Self::record_cancel_audit`]. An + /// end recorded before the accept is judged against the accept being recorded with it. + #[expect( + clippy::cast_possible_wrap, + reason = "chain timestamps are far below i64::MAX" + )] + pub async fn record_accept_and_cancel_audit( + &self, + agreement_id: &IndexingAgreementId, + accepted_at: u64, + accepted_tx: &str, + canceled_at: u64, + canceled_by: &str, + canceled_tx: Option<&str>, + ) -> Result<(), Error> { + sqlx::query( + r#" + UPDATE dipper_reg_indexing_agreements + SET accepted_at = COALESCE(accepted_at, $2), + accepted_tx = COALESCE(accepted_tx, $3), + canceled_at = CASE + WHEN canceled_at < COALESCE(accepted_at, $2) AND $4 >= COALESCE(accepted_at, $2) + THEN $4 ELSE COALESCE(canceled_at, $4) END, + canceled_by = CASE + WHEN canceled_at < COALESCE(accepted_at, $2) AND $4 >= COALESCE(accepted_at, $2) + THEN $5 ELSE COALESCE(canceled_by, $5) END, + canceled_tx = CASE + WHEN canceled_at < COALESCE(accepted_at, $2) AND $4 >= COALESCE(accepted_at, $2) + THEN $6 ELSE COALESCE(canceled_tx, $6) END, + terminated_event_emitted_at = CASE + WHEN canceled_at < COALESCE(accepted_at, $2) AND $4 >= COALESCE(accepted_at, $2) + THEN NULL ELSE terminated_event_emitted_at END + WHERE id = $1 + "#, + ) + .bind(agreement_id) + .bind(accepted_at as i64) + .bind(accepted_tx) + .bind(canceled_at as i64) + .bind(canceled_by) + .bind(canceled_tx) + .execute(&self.pool) + .await?; + Ok(()) + } + // ========================================================================= // Reassignment operations // ========================================================================= diff --git a/dipper-pgregistry/tests/it_registry_postgres.rs b/dipper-pgregistry/tests/it_registry_postgres.rs index 88b993ef..8937d3f1 100644 --- a/dipper-pgregistry/tests/it_registry_postgres.rs +++ b/dipper-pgregistry/tests/it_registry_postgres.rs @@ -3934,3 +3934,64 @@ async fn an_abandoned_agreement_awaits_replacement_once_ended_until_queued() { .expect("note it queued"); assert!(awaiting().await.is_empty()); } + +/// The accept and end of an agreement replayed from the chain go in together: an end already +/// known stays, unless it came before the accept, and then the replayed end is announced. +#[tokio::test] +async fn an_accept_and_end_from_the_chain_are_recorded_in_1_write() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let id = fixture_agreement(0xaa); + let registry = PgRegistry::new(db.clone()); + let record = async |accepted_at: u64, canceled_at: u64, tx: &str| { + registry + .record_accept_and_cancel_audit( + &id, + accepted_at, + "0xacc", + canceled_at, + "payer", + Some(tx), + ) + .await + .expect("record"); + }; + let recorded = async || { + sqlx::query_as::<_, (Option, Option, Option, bool)>( + "SELECT accepted_at, canceled_at, canceled_tx, terminated_event_emitted_at IS NULL \ + FROM dipper_reg_indexing_agreements WHERE id = $1", + ) + .bind(id) + .fetch_one(&db) + .await + .expect("read") + }; + + // Its offer was withdrawn at 50, already announced, before an accept at 100 came to light. + sqlx::query( + "UPDATE dipper_reg_indexing_agreements SET canceled_at = 50, canceled_tx = '0xwd', \ + terminated_event_emitted_at = now() WHERE id = $1", + ) + .bind(id) + .execute(&db) + .await + .expect("withdraw"); + record(100, 200, "0xend").await; + assert_eq!( + recorded().await, + (Some(100), Some(200), Some("0xend".to_owned()), true), + "the end before the accept gives way, to be announced again" + ); + + record(300, 400, "0xlater").await; + assert_eq!( + recorded().await, + (Some(100), Some(200), Some("0xend".to_owned()), true), + "what is already known stays" + ); +} From 121a9148bbb04929f1d98007fdcb1ce947709861 Mon Sep 17 00:00:00 2001 From: Samuel Metcalfe Date: Fri, 9 Oct 2026 19:03:44 +0100 Subject: [PATCH 26/26] fix: announce an agreement's end once its transaction is known (#740) * fix(registry): hold an end's announcement until its transaction is known Dipper can mark an agreement ended before the chain listener records the transaction that ended it, and the terminated event then went out without it. The event now waits for that transaction, for up to an hour, after which it goes out without one as before. * fix(listener): record the ending transaction of an abandoned agreement The listener backfilled the accept and ending transaction of agreements dipper had cancelled, but skipped ones it had ended as abandoned, so their terminated event waited an hour for a transaction that never arrived. Abandoned agreements now get the same backfill. --- .../src/network/service/chain_listener.rs | 29 +++++++++- dipper-pgregistry/src/postgres.rs | 11 +++- .../tests/it_registry_postgres.rs | 58 +++++++++++++++++++ 3 files changed, 96 insertions(+), 2 deletions(-) diff --git a/bin/dipper-service/src/network/service/chain_listener.rs b/bin/dipper-service/src/network/service/chain_listener.rs index bfc83eb4..a3ce327c 100644 --- a/bin/dipper-service/src/network/service/chain_listener.rs +++ b/bin/dipper-service/src/network/service/chain_listener.rs @@ -881,7 +881,9 @@ where let already_terminal_cancel = matches!( agreement.status, - IndexingAgreementStatus::CanceledByRequester | IndexingAgreementStatus::CanceledByIndexer, + IndexingAgreementStatus::CanceledByRequester + | IndexingAgreementStatus::CanceledByIndexer + | IndexingAgreementStatus::AbandonedByIndexer, ); // Classify off the on-chain state, which carries the contract's own // "canceled by" flag. The canceler address is the payer, not dipper's @@ -3151,6 +3153,31 @@ mod tests { ); } + #[tokio::test] + async fn test_reconcile_records_the_end_of_an_abandoned_agreement() { + // Dipper can mark it ended without the transaction that ended it, and its terminated + // event waits for that transaction, which only this record supplies. + let registry = MockRegistry::new(); + let chain_client = MockChainClient::default(); + let agreement_id = IndexingAgreementId::from_bytes(rand::random()); + registry.add_agreement(agreement_id, IndexingAgreementStatus::AbandonedByIndexer); + + let snapshot = make_snapshot(agreement_id, AgreementState::CanceledByPayer, Address::ZERO); + reconcile_agreement( + &snapshot, + ®istry, + &chain_client, + test_agreement_conf().as_ref(), + ) + .await + .expect("reconcile ok"); + + assert_eq!( + registry.audit_writes(), + vec![("accept and cancel", agreement_id)] + ); + } + #[tokio::test] async fn test_reconcile_survives_a_failed_accept_and_cancel_record() { let registry = MockRegistry::new(); diff --git a/dipper-pgregistry/src/postgres.rs b/dipper-pgregistry/src/postgres.rs index 3c3fdfd2..e9b7411e 100644 --- a/dipper-pgregistry/src/postgres.rs +++ b/dipper-pgregistry/src/postgres.rs @@ -212,6 +212,11 @@ impl sqlx::FromRow<'_, sqlx::postgres::PgRow> for CancellingAgreement { } } +/// How long an ended agreement's `terminated` event waits for the transaction that ended it. +/// Dipper can mark an agreement ended before the chain listener records that transaction, so +/// the wait lets the event carry it; after this the event goes out without one. +const TERMINATED_TX_WAIT_MINUTES: i32 = 60; + /// Statuses an on-chain cancel by dipper ends. const CANCEL_BY_REQUESTER_FROM: &[IndexingAgreementStatus] = &[ IndexingAgreementStatus::Created, @@ -1385,7 +1390,8 @@ impl PgRegistry { /// Fetch a batch of agreements awaiting a `terminated` event: in a /// terminal-cancel state, genuinely accepted on-chain (`accepted_at IS NOT /// NULL`, so a never-accepted local cancel is excluded), and not yet - /// emitted. Oldest-marked first so the backlog drains in order. + /// emitted, once the transaction that ended it is known or + /// [`TERMINATED_TX_WAIT_MINUTES`] have passed. Oldest-marked first. pub async fn get_agreements_pending_terminated_emission( &self, limit: i64, @@ -1397,6 +1403,8 @@ impl PgRegistry { WHERE status IN ($1, $2, $3) AND accepted_at IS NOT NULL AND terminated_event_emitted_at IS NULL + AND (canceled_tx IS NOT NULL + OR updated_at < timezone('UTC', now()) - make_interval(mins => $5)) ORDER BY updated_at ASC LIMIT $4 "#, @@ -1405,6 +1413,7 @@ impl PgRegistry { .bind(IndexingAgreementStatus::CanceledByIndexer) .bind(IndexingAgreementStatus::AbandonedByIndexer) .bind(limit) + .bind(TERMINATED_TX_WAIT_MINUTES) .fetch_all(&self.pool) .await?; Ok(rows) diff --git a/dipper-pgregistry/tests/it_registry_postgres.rs b/dipper-pgregistry/tests/it_registry_postgres.rs index 8937d3f1..d5fd3cef 100644 --- a/dipper-pgregistry/tests/it_registry_postgres.rs +++ b/dipper-pgregistry/tests/it_registry_postgres.rs @@ -3995,3 +3995,61 @@ async fn an_accept_and_end_from_the_chain_are_recorded_in_1_write() { "what is already known stays" ); } + +/// An ended agreement's `terminated` event waits for the transaction that ended it, which the +/// chain listener can record after dipper marks it ended, but not for ever. +#[tokio::test] +async fn an_end_is_announced_once_its_transaction_is_known_or_after_an_hour() { + let (db, _temp_db) = temp_registry_db().await; + run_fixture( + &db, + include_str!("fixtures/0003_multi_indexer_agreements.sql"), + ) + .await + .expect("Failed to run fixture"); + let id = fixture_agreement(0xaa); + let registry = PgRegistry::new(db.clone()); + registry + .mark_indexing_agreement_as_canceled_by_requester(&id) + .await + .expect("dipper ends it"); + registry + .record_accepted_audit(&id, 1_700_000_000, "0xacc") + .await + .expect("its accept is known"); + let pending = async || { + registry + .get_agreements_pending_terminated_emission(100) + .await + .expect("terminated query") + .iter() + .any(|p| p.agreement_id == id) + }; + assert!(!pending().await, "waits for its transaction"); + + sqlx::query( + "UPDATE dipper_reg_indexing_agreements \ + SET updated_at = timezone('UTC', now()) - interval '61 minutes' WHERE id = $1", + ) + .bind(id) + .execute(&db) + .await + .expect("age it"); + assert!(pending().await, "goes out without one after an hour"); + + sqlx::query( + "UPDATE dipper_reg_indexing_agreements SET updated_at = timezone('UTC', now()) WHERE id = $1", + ) + .bind(id) + .execute(&db) + .await + .expect("make it recent again"); + registry + .record_cancel_audit(&id, 1_700_000_100, "0xmgr", Some("0xend")) + .await + .expect("its transaction is recorded"); + assert!( + pending().await, + "goes out as soon as its transaction is known" + ); +}