@@ -39,6 +39,16 @@ const SCHEMA_CHANGE_RETRY_BACKOFF: Duration = Duration::from_millis(150);
3939/// unrelated internal errors, so the message text — which is the error
4040/// type's own stable Display string, not free-form prose — is the only
4141/// durable signal available.
42+ /// The server was still reporting `RetryableSchemaChanged` when the
43+ /// client-side retry budget ran out.
44+ ///
45+ /// Deliberately carries no payload: it exists so a caller with its own longer
46+ /// deadline can tell "the descriptor drain has not settled yet" apart from a
47+ /// real error, and every real error still panics at the call site with the full
48+ /// server log and faultbox report attached.
49+ #[ derive( Debug , Clone , Copy , PartialEq , Eq ) ]
50+ pub struct RetryableSchemaChange ;
51+
4252fn is_retryable_schema_change ( e : & tokio_postgres:: Error ) -> bool {
4353 e. as_db_error ( )
4454 . is_some_and ( |db| db. message ( ) . contains ( "retryable schema change" ) )
@@ -306,20 +316,53 @@ impl Session<'_> {
306316 /// does (see [`is_retryable_schema_change`] and its budget constants) —
307317 /// this is the helper the tests' tight polling loops actually use, so
308318 /// it needs the same client-retry behavior, not just the one-shot path.
319+ ///
320+ /// Panics once that budget is exhausted. A caller that runs this inside
321+ /// its own bounded poll should use [`Self::try_query_col_idx`] instead:
322+ /// this helper's ~750ms budget is far shorter than a typical poll
323+ /// deadline, so panicking here would abort a poll that still had seconds
324+ /// of budget left for exactly the condition the server told it to retry.
309325 pub async fn query_col_idx ( & self , sql : & str , idx : usize ) -> Vec < String > {
326+ match self . try_query_col_idx ( sql, idx) . await {
327+ Ok ( rows) => rows,
328+ Err ( RetryableSchemaChange ) => {
329+ let tail = super :: diagnostics:: log_tail_section ( & self . harness . server_log ( ) ) ;
330+ let reports = super :: diagnostics:: faultbox_report_section ( self . harness . data_dir ( ) ) ;
331+ panic ! (
332+ "query on session: retryable schema change never cleared within \
333+ {SCHEMA_CHANGE_RETRY_ATTEMPTS} attempts{}\n {reports}{tail}",
334+ self . harness. keep_data_dir_note( ) ,
335+ )
336+ }
337+ }
338+ }
339+
340+ /// [`Self::query_col_idx`] that reports an unresolved retryable schema
341+ /// change to the caller instead of panicking on it.
342+ ///
343+ /// The server's own contract says a client observing this condition
344+ /// retries the statement — it is the descriptor-lease-drain race, not a
345+ /// distinct failure. A poll loop that owns a longer deadline is the right
346+ /// place to absorb it, so this returns [`RetryableSchemaChange`] and lets
347+ /// the caller decide. Every other error still panics: those are real.
348+ pub async fn try_query_col_idx (
349+ & self ,
350+ sql : & str ,
351+ idx : usize ,
352+ ) -> Result < Vec < String > , RetryableSchemaChange > {
310353 let mut schema_change_attempts = 0usize ;
311354 loop {
312355 match self . client . simple_query ( sql) . await {
313356 Ok ( messages) => {
314- return messages
357+ return Ok ( messages
315358 . iter ( )
316359 . filter_map ( |m| match m {
317360 tokio_postgres:: SimpleQueryMessage :: Row ( row) => {
318361 Some ( row. get ( idx) . unwrap_or_default ( ) . to_string ( ) )
319362 }
320363 _ => None ,
321364 } )
322- . collect ( ) ;
365+ . collect ( ) ) ;
323366 }
324367 Err ( e)
325368 if is_retryable_schema_change ( & e)
@@ -328,23 +371,13 @@ impl Session<'_> {
328371 schema_change_attempts += 1 ;
329372 tokio:: time:: sleep ( SCHEMA_CHANGE_RETRY_BACKOFF ) . await ;
330373 }
374+ // Hand the still-unresolved drain back to the caller. Callers
375+ // polling on their own deadline retry; `query_col_idx` panics.
376+ Err ( e) if is_retryable_schema_change ( & e) => return Err ( RetryableSchemaChange ) ,
331377 // Same rationale as `simple_query_ready`'s error branches: the
332378 // interesting failures here are server-side, and the harness's
333379 // tempdir is gone by the time anyone reads the panic (unless
334380 // `NODEDB_TEST_KEEP_DATA_DIR` says otherwise).
335- Err ( e) if is_retryable_schema_change ( & e) => {
336- let tail = super :: diagnostics:: log_tail_section ( & self . harness . server_log ( ) ) ;
337- let reports =
338- super :: diagnostics:: faultbox_report_section ( self . harness . data_dir ( ) ) ;
339- panic ! (
340- "query on session: retryable schema change never cleared within \
341- {SCHEMA_CHANGE_RETRY_ATTEMPTS} attempts: {e}{}{}\n {reports}{tail}",
342- e. as_db_error( )
343- . map( |db| format!( " — {}: {}" , db. code( ) . code( ) , db. message( ) ) )
344- . unwrap_or_default( ) ,
345- self . harness. keep_data_dir_note( ) ,
346- )
347- }
348381 Err ( e) => {
349382 let tail = super :: diagnostics:: log_tail_section ( & self . harness . server_log ( ) ) ;
350383 let reports =
0 commit comments