From d72418f09a89e3b7b907a377555101fb94740167 Mon Sep 17 00:00:00 2001 From: Dianjin Wang Date: Tue, 1 Sep 2026 21:23:23 +0800 Subject: [PATCH 1/7] Fix flaky autovacuum-analyze tests Replace pg_sleep() with gp_stat_force_next_flush() to force a stats flush instead of waiting. Update the expected output. Assisted-by: Claude Code Backpatch-through: REL_2_STABLE --- .../input/autovacuum-analyze.source | 6 ++--- .../output/autovacuum-analyze.source | 25 +++++++++++++++---- .../input/autovacuum-analyze.source | 6 ++--- .../output/autovacuum-analyze.source | 21 +++++++++++++--- 4 files changed, 44 insertions(+), 14 deletions(-) diff --git a/contrib/pax_storage/src/test/isolation2/input/autovacuum-analyze.source b/contrib/pax_storage/src/test/isolation2/input/autovacuum-analyze.source index 19187b107c2..7b9b2df22d4 100644 --- a/contrib/pax_storage/src/test/isolation2/input/autovacuum-analyze.source +++ b/contrib/pax_storage/src/test/isolation2/input/autovacuum-analyze.source @@ -171,7 +171,7 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' -- with auto_stats, the auto-ANALYZE still trigger 2: INSERT INTO autostatstbl select i from generate_series(1, 1000) as i; -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. SELECT count(*) FROM pg_statistic where starelid = 'autostatstbl'::regclass; @@ -205,7 +205,7 @@ SELECT gp_inject_fault('analyze_finished_one_relation', 'skip', '', '', 'autosta SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', '', 'autostatstbl', 1, -1, 0, 1); 2: INSERT INTO autostatstbl select i from generate_series(1001, 2000) as i; -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; @@ -238,7 +238,7 @@ SELECT gp_inject_fault('analyze_finished_one_relation', 'skip', '', '', 'autosta SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', '', 'autostatstbl', 1, -1, 0, 1); 2: INSERT INTO autostatstbl select i from generate_series(2001, 3000) as i; -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); -- auto_stats should not executed and auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; diff --git a/contrib/pax_storage/src/test/isolation2/output/autovacuum-analyze.source b/contrib/pax_storage/src/test/isolation2/output/autovacuum-analyze.source index be374068d33..42370fdf595 100644 --- a/contrib/pax_storage/src/test/isolation2/output/autovacuum-analyze.source +++ b/contrib/pax_storage/src/test/isolation2/output/autovacuum-analyze.source @@ -408,7 +408,12 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' -- with auto_stats, the auto-ANALYZE still trigger 2: INSERT INTO autostatstbl select i from generate_series(1, 1000) as i; INSERT 1000 -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); + gp_stat_force_next_flush +-------------------------- + +(1 row) + -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. SELECT count(*) FROM pg_statistic where starelid = 'autostatstbl'::regclass; count @@ -425,7 +430,7 @@ select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; select analyze_count, autoanalyze_count, n_mod_since_analyze from pg_stat_all_tables where relname = 'autostatstbl'; analyze_count | autoanalyze_count | n_mod_since_analyze ---------------+-------------------+--------------------- - 1 | 0 | 0 + 1 | 0 | 1000 (1 row) -- Wait until autovacuum is triggered @@ -493,7 +498,12 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' 2: INSERT INTO autostatstbl select i from generate_series(1001, 2000) as i; INSERT 1000 -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); + gp_stat_force_next_flush +-------------------------- + +(1 row) + -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; relpages | reltuples @@ -572,7 +582,12 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' 2: INSERT INTO autostatstbl select i from generate_series(2001, 3000) as i; INSERT 1000 -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); + gp_stat_force_next_flush +-------------------------- + +(1 row) + -- auto_stats should not executed and auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; relpages | reltuples @@ -584,7 +599,7 @@ select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; select analyze_count, autoanalyze_count, n_mod_since_analyze from pg_stat_all_tables where relname = 'autostatstbl'; analyze_count | autoanalyze_count | n_mod_since_analyze ---------------+-------------------+--------------------- - 2 | 2 | 0 + 2 | 2 | 1000 (1 row) -- Wait until autovacuum is triggered diff --git a/src/test/singlenode_isolation2/input/autovacuum-analyze.source b/src/test/singlenode_isolation2/input/autovacuum-analyze.source index b35c3072a85..1e1fcb69f2a 100644 --- a/src/test/singlenode_isolation2/input/autovacuum-analyze.source +++ b/src/test/singlenode_isolation2/input/autovacuum-analyze.source @@ -170,7 +170,7 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' -- with auto_stats, the auto-ANALYZE still trigger 2: INSERT INTO autostatstbl select i from generate_series(1, 1000) as i; -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. SELECT count(*) FROM pg_statistic where starelid = 'autostatstbl'::regclass; @@ -204,7 +204,7 @@ SELECT gp_inject_fault('analyze_finished_one_relation', 'skip', '', '', 'autosta SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', '', 'autostatstbl', 1, -1, 0, 1); 2: INSERT INTO autostatstbl select i from generate_series(1001, 2000) as i; -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; @@ -237,7 +237,7 @@ SELECT gp_inject_fault('analyze_finished_one_relation', 'skip', '', '', 'autosta SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', '', 'autostatstbl', 1, -1, 0, 1); 2: INSERT INTO autostatstbl select i from generate_series(2001, 3000) as i; -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); -- auto_stats should not executed and auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; diff --git a/src/test/singlenode_isolation2/output/autovacuum-analyze.source b/src/test/singlenode_isolation2/output/autovacuum-analyze.source index 382b443944c..9ab7afa90ee 100644 --- a/src/test/singlenode_isolation2/output/autovacuum-analyze.source +++ b/src/test/singlenode_isolation2/output/autovacuum-analyze.source @@ -403,7 +403,12 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' -- with auto_stats, the auto-ANALYZE still trigger 2: INSERT INTO autostatstbl select i from generate_series(1, 1000) as i; INSERT 1000 -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); + gp_stat_force_next_flush +-------------------------- + +(1 row) + -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. SELECT count(*) FROM pg_statistic where starelid = 'autostatstbl'::regclass; count @@ -488,7 +493,12 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' 2: INSERT INTO autostatstbl select i from generate_series(1001, 2000) as i; INSERT 1000 -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); + gp_stat_force_next_flush +-------------------------- + +(1 row) + -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; relpages | reltuples @@ -567,7 +577,12 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' 2: INSERT INTO autostatstbl select i from generate_series(2001, 3000) as i; INSERT 1000 -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); + gp_stat_force_next_flush +-------------------------- + +(1 row) + -- auto_stats should not executed and auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; relpages | reltuples From eee54367c32b5a420caaa8f3eb42a70b4490adf8 Mon Sep 17 00:00:00 2001 From: Dianjin Wang Date: Tue, 1 Sep 2026 23:09:19 +0800 Subject: [PATCH 2/7] Wait for the wal_sender_loop fault to trigger before resuming VACUUM. Assisted-by: Claude Code Backpatch-through: REL_2_STABLE --- .../expected/vacuum_progress_column.out | 11 ++++++++++ .../expected/vacuum_progress_row.out | 22 +++++++++++++++++++ .../isolation2/sql/vacuum_progress_column.sql | 7 ++++++ .../isolation2/sql/vacuum_progress_row.sql | 14 ++++++++++++ 4 files changed, 54 insertions(+) diff --git a/src/test/isolation2/expected/vacuum_progress_column.out b/src/test/isolation2/expected/vacuum_progress_column.out index 12a3bceb78f..8f3f5f1d033 100644 --- a/src/test/isolation2/expected/vacuum_progress_column.out +++ b/src/test/isolation2/expected/vacuum_progress_column.out @@ -305,11 +305,22 @@ select relid::regclass as relname, phase, heap_blks_total, heap_blks_scanned, he (1 row) -- Resume execution of compact phase and block at syncrep on one segment. +-- The suspend only takes effect the next time the walsender goes round its +-- loop, so wait until it is really parked before letting the vacuum go on. +-- Otherwise the compact phase can commit while the walsender is still +-- streaming, syncrep is satisfied by the live mirror, the vacuum runs to +-- completion on the same gang, no new vacuum worker ever takes over and the +-- gp_wait_until_triggered_fault() below times out after ten minutes. 2: SELECT gp_inject_fault_infinite('wal_sender_loop', 'suspend', dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; gp_inject_fault_infinite -------------------------- Success: (1 row) +2: SELECT gp_wait_until_triggered_fault('wal_sender_loop', 1, dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; + gp_wait_until_triggered_fault +------------------------------- + Success: +(1 row) 2: SELECT gp_inject_fault('vacuum_ao_after_compact', 'reset', dbid) FROM gp_segment_configuration WHERE content > -1 AND role = 'p'; gp_inject_fault ----------------- diff --git a/src/test/isolation2/expected/vacuum_progress_row.out b/src/test/isolation2/expected/vacuum_progress_row.out index ef39d8edaf3..a7b713d689e 100644 --- a/src/test/isolation2/expected/vacuum_progress_row.out +++ b/src/test/isolation2/expected/vacuum_progress_row.out @@ -359,11 +359,22 @@ select relid::regclass as relname, phase, heap_blks_total, heap_blks_scanned, he (1 row) -- Resume execution of compact phase and block at syncrep. +-- The suspend only takes effect the next time the walsender goes round its +-- loop, so wait until it is really parked before letting the vacuum go on. +-- Otherwise the compact phase can commit while the walsender is still +-- streaming, syncrep is satisfied by the live mirror, the vacuum runs to +-- completion on the same gang, no new vacuum worker ever takes over and the +-- gp_wait_until_triggered_fault() below times out after ten minutes. 2: SELECT gp_inject_fault_infinite('wal_sender_loop', 'suspend', dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; gp_inject_fault_infinite -------------------------- Success: (1 row) +2: SELECT gp_wait_until_triggered_fault('wal_sender_loop', 1, dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; + gp_wait_until_triggered_fault +------------------------------- + Success: +(1 row) 2: SELECT gp_inject_fault('vacuum_ao_after_compact', 'reset', dbid) FROM gp_segment_configuration WHERE content > -1 AND role = 'p'; gp_inject_fault ----------------- @@ -563,11 +574,22 @@ select relid::regclass as relname, phase, heap_blks_total, heap_blks_scanned, he (1 row) -- Resume execution of compact phase and block at syncrep. +-- The suspend only takes effect the next time the walsender goes round its +-- loop, so wait until it is really parked before letting the vacuum go on. +-- Otherwise the compact phase can commit while the walsender is still +-- streaming, syncrep is satisfied by the live mirror, the vacuum runs to +-- completion on the same gang, no new vacuum worker ever takes over and the +-- gp_wait_until_triggered_fault() below times out after ten minutes. 2: SELECT gp_inject_fault_infinite('wal_sender_loop', 'suspend', dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; gp_inject_fault_infinite -------------------------- Success: (1 row) +2: SELECT gp_wait_until_triggered_fault('wal_sender_loop', 1, dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; + gp_wait_until_triggered_fault +------------------------------- + Success: +(1 row) 2: SELECT gp_inject_fault('vacuum_ao_after_compact', 'reset', dbid) FROM gp_segment_configuration WHERE content > -1 AND role = 'p'; gp_inject_fault ----------------- diff --git a/src/test/isolation2/sql/vacuum_progress_column.sql b/src/test/isolation2/sql/vacuum_progress_column.sql index 60250368b46..42f20659224 100644 --- a/src/test/isolation2/sql/vacuum_progress_column.sql +++ b/src/test/isolation2/sql/vacuum_progress_column.sql @@ -131,7 +131,14 @@ select gp_segment_id, relid::regclass as relname, phase, heap_blks_total, heap_b select relid::regclass as relname, phase, heap_blks_total, heap_blks_scanned, heap_blks_vacuumed, index_vacuum_count, max_dead_tuples, num_dead_tuples from gp_stat_progress_vacuum_summary; -- Resume execution of compact phase and block at syncrep on one segment. +-- The suspend only takes effect the next time the walsender goes round its +-- loop, so wait until it is really parked before letting the vacuum go on. +-- Otherwise the compact phase can commit while the walsender is still +-- streaming, syncrep is satisfied by the live mirror, the vacuum runs to +-- completion on the same gang, no new vacuum worker ever takes over and the +-- gp_wait_until_triggered_fault() below times out after ten minutes. 2: SELECT gp_inject_fault_infinite('wal_sender_loop', 'suspend', dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; +2: SELECT gp_wait_until_triggered_fault('wal_sender_loop', 1, dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; 2: SELECT gp_inject_fault('vacuum_ao_after_compact', 'reset', dbid) FROM gp_segment_configuration WHERE content > -1 AND role = 'p'; -- stop the mirror should turn off syncrep 2: SELECT pg_ctl(datadir, 'stop', 'immediate') FROM gp_segment_configuration WHERE content = 1 AND role = 'm'; diff --git a/src/test/isolation2/sql/vacuum_progress_row.sql b/src/test/isolation2/sql/vacuum_progress_row.sql index c832a1fd0df..acf9d0635cf 100644 --- a/src/test/isolation2/sql/vacuum_progress_row.sql +++ b/src/test/isolation2/sql/vacuum_progress_row.sql @@ -136,7 +136,14 @@ select gp_segment_id, relid::regclass as relname, phase, heap_blks_total, heap_b select relid::regclass as relname, phase, heap_blks_total, heap_blks_scanned, heap_blks_vacuumed, index_vacuum_count, max_dead_tuples, num_dead_tuples from gp_stat_progress_vacuum_summary; -- Resume execution of compact phase and block at syncrep. +-- The suspend only takes effect the next time the walsender goes round its +-- loop, so wait until it is really parked before letting the vacuum go on. +-- Otherwise the compact phase can commit while the walsender is still +-- streaming, syncrep is satisfied by the live mirror, the vacuum runs to +-- completion on the same gang, no new vacuum worker ever takes over and the +-- gp_wait_until_triggered_fault() below times out after ten minutes. 2: SELECT gp_inject_fault_infinite('wal_sender_loop', 'suspend', dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; +2: SELECT gp_wait_until_triggered_fault('wal_sender_loop', 1, dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; 2: SELECT gp_inject_fault('vacuum_ao_after_compact', 'reset', dbid) FROM gp_segment_configuration WHERE content > -1 AND role = 'p'; -- stop the mirror should turn off syncrep 2: SELECT pg_ctl(datadir, 'stop', 'immediate') FROM gp_segment_configuration WHERE content=1 AND role = 'm'; @@ -210,7 +217,14 @@ select gp_segment_id, relid::regclass as relname, phase, heap_blks_total, heap_b select relid::regclass as relname, phase, heap_blks_total, heap_blks_scanned, heap_blks_vacuumed, index_vacuum_count, max_dead_tuples, num_dead_tuples from gp_stat_progress_vacuum_summary; -- Resume execution of compact phase and block at syncrep. +-- The suspend only takes effect the next time the walsender goes round its +-- loop, so wait until it is really parked before letting the vacuum go on. +-- Otherwise the compact phase can commit while the walsender is still +-- streaming, syncrep is satisfied by the live mirror, the vacuum runs to +-- completion on the same gang, no new vacuum worker ever takes over and the +-- gp_wait_until_triggered_fault() below times out after ten minutes. 2: SELECT gp_inject_fault_infinite('wal_sender_loop', 'suspend', dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; +2: SELECT gp_wait_until_triggered_fault('wal_sender_loop', 1, dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; 2: SELECT gp_inject_fault('vacuum_ao_after_compact', 'reset', dbid) FROM gp_segment_configuration WHERE content > -1 AND role = 'p'; -- stop the mirror should turn off syncrep 2: SELECT pg_ctl(datadir, 'stop', 'immediate') FROM gp_segment_configuration WHERE content=1 AND role = 'm'; From eefec3d016fcff1fee0a2ec2253f90d607c50a27 Mon Sep 17 00:00:00 2001 From: Dianjin Wang Date: Tue, 1 Sep 2026 21:23:55 +0800 Subject: [PATCH 3/7] Fix expected CPU usage in resgroup tests The test already averages CPU samples, but the expected values are still too low. The suite sets the cluster CPU limit to 100%, so expect 100% for one uncapped group and 33%/67% for groups with weights 100/200. Keep the existing tolerance of 10 percentage points. Assisted-by: Claude Code Backpatch-through: REL_2_STABLE --- .../expected/resgroup/resgroup_cpu_max_percent.out | 12 +++++++++--- .../sql/resgroup/resgroup_cpu_max_percent.sql | 12 +++++++++--- .../expected/resgroup/resgroup_cpu_max_percent.out | 12 +++++++++--- .../sql/resgroup/resgroup_cpu_max_percent.sql | 12 +++++++++--- 4 files changed, 36 insertions(+), 12 deletions(-) diff --git a/contrib/pax_storage/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out b/contrib/pax_storage/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out index 1e011ecec22..78b8bc561ed 100644 --- a/contrib/pax_storage/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out +++ b/contrib/pax_storage/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out @@ -220,7 +220,11 @@ SELECT pg_sleep(1.7); (1 row) -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 90, 10); +-- rg1_cpu_test is uncapped (cpu_max_percent=-1) and it is the only busy +-- group, so it takes essentially every core: gp_resgroup_status reports +-- ~100, not 90. Expecting 90 put the real value on the upper edge of the +-- +/- err_rate window, so any upward sampling jitter failed the test. +SELECT verify_cpu_usage('rg1_cpu_test', 100, 10); verify_cpu_usage ------------------ t @@ -395,12 +399,14 @@ SELECT pg_sleep(1.7); (1 row) -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 30, 10); +-- Both groups are uncapped, so they share the whole machine in proportion +-- to cpu_weight (100 and 200): ~33 and ~67, not ~30 and ~60. +SELECT verify_cpu_usage('rg1_cpu_test', 33, 10); verify_cpu_usage ------------------ t (1 row) -SELECT verify_cpu_usage('rg2_cpu_test', 60, 10); +SELECT verify_cpu_usage('rg2_cpu_test', 67, 10); verify_cpu_usage ------------------ t diff --git a/contrib/pax_storage/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql b/contrib/pax_storage/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql index a014108bb52..7c249c1e6c8 100644 --- a/contrib/pax_storage/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql +++ b/contrib/pax_storage/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql @@ -141,7 +141,11 @@ SELECT fetch_sample(); SELECT pg_sleep(1.7); -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 90, 10); +-- rg1_cpu_test is uncapped (cpu_max_percent=-1) and it is the only busy +-- group, so it takes essentially every core: gp_resgroup_status reports +-- ~100, not 90. Expecting 90 put the real value on the upper edge of the +-- +/- err_rate window, so any upward sampling jitter failed the test. +SELECT verify_cpu_usage('rg1_cpu_test', 100, 10); -- start_ignore SELECT * FROM cancel_all; @@ -212,8 +216,10 @@ SELECT fetch_sample(); SELECT pg_sleep(1.7); -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 30, 10); -SELECT verify_cpu_usage('rg2_cpu_test', 60, 10); +-- Both groups are uncapped, so they share the whole machine in proportion +-- to cpu_weight (100 and 200): ~33 and ~67, not ~30 and ~60. +SELECT verify_cpu_usage('rg1_cpu_test', 33, 10); +SELECT verify_cpu_usage('rg2_cpu_test', 67, 10); -- start_ignore SELECT * FROM cancel_all; diff --git a/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out b/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out index 44969ba7eea..c4706bbc9f9 100644 --- a/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out +++ b/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out @@ -220,7 +220,11 @@ SELECT pg_sleep(1.7); (1 row) -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 90, 10); +-- rg1_cpu_test is uncapped (cpu_max_percent=-1) and it is the only busy +-- group, so it takes essentially every core: gp_resgroup_status reports +-- ~100, not 90. Expecting 90 put the real value on the upper edge of the +-- +/- err_rate window, so any upward sampling jitter failed the test. +SELECT verify_cpu_usage('rg1_cpu_test', 100, 10); verify_cpu_usage ------------------ t @@ -395,13 +399,15 @@ SELECT pg_sleep(1.7); (1 row) -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 30, 10); +-- Both groups are uncapped, so they share the whole machine in proportion +-- to cpu_weight (100 and 200): ~33 and ~67, not ~30 and ~60. +SELECT verify_cpu_usage('rg1_cpu_test', 33, 10); verify_cpu_usage ------------------ t (1 row) -- start_ignore -SELECT verify_cpu_usage('rg2_cpu_test', 60, 10); +SELECT verify_cpu_usage('rg2_cpu_test', 67, 10); verify_cpu_usage ------------------ t diff --git a/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql b/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql index 89fa523cc88..164fdcd8199 100644 --- a/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql +++ b/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql @@ -141,7 +141,11 @@ SELECT fetch_sample(); SELECT pg_sleep(1.7); -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 90, 10); +-- rg1_cpu_test is uncapped (cpu_max_percent=-1) and it is the only busy +-- group, so it takes essentially every core: gp_resgroup_status reports +-- ~100, not 90. Expecting 90 put the real value on the upper edge of the +-- +/- err_rate window, so any upward sampling jitter failed the test. +SELECT verify_cpu_usage('rg1_cpu_test', 100, 10); -- start_ignore SELECT * FROM cancel_all; @@ -212,9 +216,11 @@ SELECT fetch_sample(); SELECT pg_sleep(1.7); -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 30, 10); +-- Both groups are uncapped, so they share the whole machine in proportion +-- to cpu_weight (100 and 200): ~33 and ~67, not ~30 and ~60. +SELECT verify_cpu_usage('rg1_cpu_test', 33, 10); -- start_ignore -SELECT verify_cpu_usage('rg2_cpu_test', 60, 10); +SELECT verify_cpu_usage('rg2_cpu_test', 67, 10); SELECT * FROM cancel_all; From 3eb21d07c1fcffa9dd43dfd26706fbf1ea3807a8 Mon Sep 17 00:00:00 2001 From: Dianjin Wang Date: Thu, 17 Sep 2026 11:08:10 +0800 Subject: [PATCH 4/7] Disable tasks created to test schedule syntax These tasks only check that valid schedules are accepted. Deactivate them after creation to avoid background runs interfering with cleanup and the oid_wraparound test. Assisted-by: Claude Code Backpatch-through: REL_2_STABLE --- .../pax_storage/src/test/regress/expected/task.out | 13 +++++++++++++ contrib/pax_storage/src/test/regress/sql/task.sql | 13 +++++++++++++ src/test/regress/expected/task.out | 13 +++++++++++++ src/test/regress/sql/task.sql | 13 +++++++++++++ 4 files changed, 52 insertions(+) diff --git a/contrib/pax_storage/src/test/regress/expected/task.out b/contrib/pax_storage/src/test/regress/expected/task.out index 2f052fdad1a..3f4cf9c14ed 100644 --- a/contrib/pax_storage/src/test/regress/expected/task.out +++ b/contrib/pax_storage/src/test/regress/expected/task.out @@ -69,10 +69,23 @@ ERROR: User task_cron does not have CONNECT privilege on task_dbno alter task vacuum_db user hopedoesnotexist; ERROR: role "hopedoesnotexist" does not exist -- valid interval tasks +-- +-- These four only exercise the schedule parser, they are never meant to run. +-- Deactivate each one as soon as it exists: while a second-based task stays +-- active the scheduler keeps firing it, and every run draws a run id from the +-- cluster-wide Oid counter (NextRunId -> GetNewOidWithIndex). A task still +-- firing later in the same regression run shifts the counter that +-- oid_wraparound hardcodes, and DROP TASK, which also deletes the task's +-- pg_task_run_history rows, fails with "tuple concurrently updated" when it +-- races the run the scheduler is in the middle of recording. create task valid_task_1 schedule '1 second' as 'select 1'; +alter task valid_task_1 not active; create task valid_task_2 schedule ' 30 sEcOnDs ' as 'select 1'; +alter task valid_task_2 not active; create task valid_task_3 schedule '59 seconds' as 'select 1'; +alter task valid_task_3 not active; create task valid_task_4 schedule '17 seconds ' as 'select 1'; +alter task valid_task_4 not active; -- task in function DO $$ BEGIN diff --git a/contrib/pax_storage/src/test/regress/sql/task.sql b/contrib/pax_storage/src/test/regress/sql/task.sql index bf12909c761..3e29c46adad 100644 --- a/contrib/pax_storage/src/test/regress/sql/task.sql +++ b/contrib/pax_storage/src/test/regress/sql/task.sql @@ -49,10 +49,23 @@ alter task vacuum_db database task_dbno user task_cron; alter task vacuum_db user hopedoesnotexist; -- valid interval tasks +-- +-- These four only exercise the schedule parser, they are never meant to run. +-- Deactivate each one as soon as it exists: while a second-based task stays +-- active the scheduler keeps firing it, and every run draws a run id from the +-- cluster-wide Oid counter (NextRunId -> GetNewOidWithIndex). A task still +-- firing later in the same regression run shifts the counter that +-- oid_wraparound hardcodes, and DROP TASK, which also deletes the task's +-- pg_task_run_history rows, fails with "tuple concurrently updated" when it +-- races the run the scheduler is in the middle of recording. create task valid_task_1 schedule '1 second' as 'select 1'; +alter task valid_task_1 not active; create task valid_task_2 schedule ' 30 sEcOnDs ' as 'select 1'; +alter task valid_task_2 not active; create task valid_task_3 schedule '59 seconds' as 'select 1'; +alter task valid_task_3 not active; create task valid_task_4 schedule '17 seconds ' as 'select 1'; +alter task valid_task_4 not active; -- task in function DO $$ diff --git a/src/test/regress/expected/task.out b/src/test/regress/expected/task.out index 2f052fdad1a..3f4cf9c14ed 100644 --- a/src/test/regress/expected/task.out +++ b/src/test/regress/expected/task.out @@ -69,10 +69,23 @@ ERROR: User task_cron does not have CONNECT privilege on task_dbno alter task vacuum_db user hopedoesnotexist; ERROR: role "hopedoesnotexist" does not exist -- valid interval tasks +-- +-- These four only exercise the schedule parser, they are never meant to run. +-- Deactivate each one as soon as it exists: while a second-based task stays +-- active the scheduler keeps firing it, and every run draws a run id from the +-- cluster-wide Oid counter (NextRunId -> GetNewOidWithIndex). A task still +-- firing later in the same regression run shifts the counter that +-- oid_wraparound hardcodes, and DROP TASK, which also deletes the task's +-- pg_task_run_history rows, fails with "tuple concurrently updated" when it +-- races the run the scheduler is in the middle of recording. create task valid_task_1 schedule '1 second' as 'select 1'; +alter task valid_task_1 not active; create task valid_task_2 schedule ' 30 sEcOnDs ' as 'select 1'; +alter task valid_task_2 not active; create task valid_task_3 schedule '59 seconds' as 'select 1'; +alter task valid_task_3 not active; create task valid_task_4 schedule '17 seconds ' as 'select 1'; +alter task valid_task_4 not active; -- task in function DO $$ BEGIN diff --git a/src/test/regress/sql/task.sql b/src/test/regress/sql/task.sql index bf12909c761..3e29c46adad 100644 --- a/src/test/regress/sql/task.sql +++ b/src/test/regress/sql/task.sql @@ -49,10 +49,23 @@ alter task vacuum_db database task_dbno user task_cron; alter task vacuum_db user hopedoesnotexist; -- valid interval tasks +-- +-- These four only exercise the schedule parser, they are never meant to run. +-- Deactivate each one as soon as it exists: while a second-based task stays +-- active the scheduler keeps firing it, and every run draws a run id from the +-- cluster-wide Oid counter (NextRunId -> GetNewOidWithIndex). A task still +-- firing later in the same regression run shifts the counter that +-- oid_wraparound hardcodes, and DROP TASK, which also deletes the task's +-- pg_task_run_history rows, fails with "tuple concurrently updated" when it +-- races the run the scheduler is in the middle of recording. create task valid_task_1 schedule '1 second' as 'select 1'; +alter task valid_task_1 not active; create task valid_task_2 schedule ' 30 sEcOnDs ' as 'select 1'; +alter task valid_task_2 not active; create task valid_task_3 schedule '59 seconds' as 'select 1'; +alter task valid_task_3 not active; create task valid_task_4 schedule '17 seconds ' as 'select 1'; +alter task valid_task_4 not active; -- task in function DO $$ From 5ac5ead55022226830d194bb809639e7125213d3 Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Fri, 18 Sep 2026 19:54:26 +0300 Subject: [PATCH 5/7] Fix flaky stats check in vacuum_progress_row/column tests The tests read pg_stat_all_tables right after the DELETE, while the stats of the aborted inserts may not have arrived yet and those of the DELETE may already have. Wait for the stats of the aborts and read the view before the DELETE instead. --- .../expected/vacuum_progress_column.out | 15 +++++++++++++++ .../isolation2/expected/vacuum_progress_row.out | 15 +++++++++++++++ .../isolation2/sql/vacuum_progress_column.sql | 7 +++++++ src/test/isolation2/sql/vacuum_progress_row.sql | 7 +++++++ 4 files changed, 44 insertions(+) diff --git a/src/test/isolation2/expected/vacuum_progress_column.out b/src/test/isolation2/expected/vacuum_progress_column.out index 8f3f5f1d033..30055afcaaa 100644 --- a/src/test/isolation2/expected/vacuum_progress_column.out +++ b/src/test/isolation2/expected/vacuum_progress_column.out @@ -41,6 +41,21 @@ ABORT 1: ABORT; ABORT +-- Look up the collected stats before the DELETE below. The stats collector +-- is asynchronous: wait until it has received the dead tuples of both +-- aborted inserts, and read the view before the DELETE, whose own counts +-- may reach the collector at any moment after it. +1U: SELECT wait_until_dead_tup_change_to('vacuum_progress_ao_column'::regclass::oid, 200000); + wait_until_dead_tup_change_to +------------------------------- + OK +(1 row) +SELECT n_live_tup, n_dead_tup, last_vacuum, vacuum_count FROM pg_stat_all_tables WHERE relname = 'vacuum_progress_ao_column'; + n_live_tup | n_dead_tup | last_vacuum | vacuum_count +------------+------------+-------------+-------------- + 100000 | 200000 | | 0 +(1 row) + -- Also delete half of the tuples evenly before the EOF of segno 2. DELETE FROM vacuum_progress_ao_column where j % 2 = 0; DELETE 50000 diff --git a/src/test/isolation2/expected/vacuum_progress_row.out b/src/test/isolation2/expected/vacuum_progress_row.out index a7b713d689e..d700968c324 100644 --- a/src/test/isolation2/expected/vacuum_progress_row.out +++ b/src/test/isolation2/expected/vacuum_progress_row.out @@ -40,6 +40,21 @@ ABORT 1: ABORT; ABORT +-- Look up the collected stats before the DELETE below. The stats collector +-- is asynchronous: wait until it has received the dead tuples of both +-- aborted inserts, and read the view before the DELETE, whose own counts +-- may reach the collector at any moment after it. +1U: SELECT wait_until_dead_tup_change_to('vacuum_progress_ao_row'::regclass::oid, 200000); + wait_until_dead_tup_change_to +------------------------------- + OK +(1 row) +SELECT n_live_tup, n_dead_tup, last_vacuum, vacuum_count FROM pg_stat_all_tables WHERE relname = 'vacuum_progress_ao_row'; + n_live_tup | n_dead_tup | last_vacuum | vacuum_count +------------+------------+-------------+-------------- + 100000 | 200000 | | 0 +(1 row) + -- Also delete half of the tuples evenly before the EOF of segno 2. DELETE FROM vacuum_progress_ao_row where j % 2 = 0; DELETE 50000 diff --git a/src/test/isolation2/sql/vacuum_progress_column.sql b/src/test/isolation2/sql/vacuum_progress_column.sql index 42f20659224..6ed4744a2b8 100644 --- a/src/test/isolation2/sql/vacuum_progress_column.sql +++ b/src/test/isolation2/sql/vacuum_progress_column.sql @@ -27,6 +27,13 @@ CREATE INDEX on vacuum_progress_ao_column(j); -- Abort so that segno 1 has logical EOF = 0. 1: ABORT; +-- Look up the collected stats before the DELETE below. The stats collector +-- is asynchronous: wait until it has received the dead tuples of both +-- aborted inserts, and read the view before the DELETE, whose own counts +-- may reach the collector at any moment after it. +1U: SELECT wait_until_dead_tup_change_to('vacuum_progress_ao_column'::regclass::oid, 200000); +SELECT n_live_tup, n_dead_tup, last_vacuum, vacuum_count FROM pg_stat_all_tables WHERE relname = 'vacuum_progress_ao_column'; + -- Also delete half of the tuples evenly before the EOF of segno 2. DELETE FROM vacuum_progress_ao_column where j % 2 = 0; diff --git a/src/test/isolation2/sql/vacuum_progress_row.sql b/src/test/isolation2/sql/vacuum_progress_row.sql index acf9d0635cf..7b3749d7b70 100644 --- a/src/test/isolation2/sql/vacuum_progress_row.sql +++ b/src/test/isolation2/sql/vacuum_progress_row.sql @@ -26,6 +26,13 @@ CREATE INDEX on vacuum_progress_ao_row(j); -- Abort so that segno 1 has logical EOF = 0. 1: ABORT; +-- Look up the collected stats before the DELETE below. The stats collector +-- is asynchronous: wait until it has received the dead tuples of both +-- aborted inserts, and read the view before the DELETE, whose own counts +-- may reach the collector at any moment after it. +1U: SELECT wait_until_dead_tup_change_to('vacuum_progress_ao_row'::regclass::oid, 200000); +SELECT n_live_tup, n_dead_tup, last_vacuum, vacuum_count FROM pg_stat_all_tables WHERE relname = 'vacuum_progress_ao_row'; + -- Also delete half of the tuples evenly before the EOF of segno 2. DELETE FROM vacuum_progress_ao_row where j % 2 = 0; From 15cf62344707d70f985e3628ae56adcc58646ec9 Mon Sep 17 00:00:00 2001 From: Dianjin Wang Date: Mon, 21 Sep 2026 16:06:46 +0800 Subject: [PATCH 6/7] Fix flaky 019_replslot_limit test Run CHECKPOINT before waiting for the standby to catch up, so the slot's restart_lsn is updated before checking WAL status. Ported from Greenplum. Assisted-by: Claude Code Backpatch-through: REL_2_STABLE --- src/test/recovery/t/019_replslot_limit.pl | 26 +++++++++++++++++++---- 1 file changed, 22 insertions(+), 4 deletions(-) diff --git a/src/test/recovery/t/019_replslot_limit.pl b/src/test/recovery/t/019_replslot_limit.pl index 65b5b6e665b..cb25a181830 100644 --- a/src/test/recovery/t/019_replslot_limit.pl +++ b/src/test/recovery/t/019_replslot_limit.pl @@ -47,8 +47,17 @@ $node_standby->start; +# Cloudberry: take a CHECKPOINT before recording the position to wait for. +# A physical slot's restart_lsn is set to whichever is smaller, the confirmed +# received lsn or the last checkpoint's redo lsn (ca0c1809825), so it tracks +# checkpoints rather than the standby and wait_for_catchup() alone says nothing +# about where it ends up. Waiting for a position taken after a checkpoint makes +# the reply that satisfies it arrive after that checkpoint, which pins +# restart_lsn to its redo point. +$node_primary->safe_psql('postgres', "CHECKPOINT;"); # Wait until standby has replayed enough data -$node_primary->wait_for_catchup($node_standby); +my $start_lsn = $node_primary->lsn('write'); +$node_primary->wait_for_catchup($node_standby, 'replay', $start_lsn); # Stop standby $node_standby->stop; @@ -85,7 +94,10 @@ # The standby can reconnect to primary $node_standby->start; -$node_primary->wait_for_catchup($node_standby); +# Cloudberry: CHECKPOINT first, see the comment on the first such wait above. +$node_primary->safe_psql('postgres', "CHECKPOINT;"); +$start_lsn = $node_primary->lsn('write'); +$node_primary->wait_for_catchup($node_standby, 'replay', $start_lsn); $node_standby->stop; @@ -115,7 +127,10 @@ # The standby can reconnect to primary $node_standby->start; -$node_primary->wait_for_catchup($node_standby); +# Cloudberry: CHECKPOINT first, see the comment on the first such wait above. +$node_primary->safe_psql('postgres', "CHECKPOINT;"); +$start_lsn = $node_primary->lsn('write'); +$node_primary->wait_for_catchup($node_standby, 'replay', $start_lsn); $node_standby->stop; # wal_keep_size overrides max_slot_wal_keep_size @@ -134,7 +149,10 @@ # The standby can reconnect to primary $node_standby->start; -$node_primary->wait_for_catchup($node_standby); +# Cloudberry: CHECKPOINT first, see the comment on the first such wait above. +$node_primary->safe_psql('postgres', "CHECKPOINT;"); +$start_lsn = $node_primary->lsn('write'); +$node_primary->wait_for_catchup($node_standby, 'replay', $start_lsn); $node_standby->stop; # Advance WAL again without checkpoint, reducing remain by 6 MB. From a94cecbb9a1c70f36738ff8e6779d99d673d5c26 Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Wed, 30 Sep 2026 18:21:00 +0300 Subject: [PATCH 7/7] Fix synchronization in vacuum progress isolation tests --- .../expected/vacuum_progress_column.out | 17 +++++++------- .../expected/vacuum_progress_row.out | 23 +++++++++++-------- .../isolation2/sql/vacuum_progress_column.sql | 11 ++++----- .../isolation2/sql/vacuum_progress_row.sql | 15 ++++++------ 4 files changed, 35 insertions(+), 31 deletions(-) diff --git a/src/test/isolation2/expected/vacuum_progress_column.out b/src/test/isolation2/expected/vacuum_progress_column.out index 30055afcaaa..40bce79477a 100644 --- a/src/test/isolation2/expected/vacuum_progress_column.out +++ b/src/test/isolation2/expected/vacuum_progress_column.out @@ -41,11 +41,10 @@ ABORT 1: ABORT; ABORT --- Look up the collected stats before the DELETE below. The stats collector --- is asynchronous: wait until it has received the dead tuples of both --- aborted inserts, and read the view before the DELETE, whose own counts --- may reach the collector at any moment after it. -1U: SELECT wait_until_dead_tup_change_to('vacuum_progress_ao_column'::regclass::oid, 200000); +-- Wait on the coordinator for the statistics of both aborted inserts, since +-- the following pg_stat_all_tables query also runs there. Read before DELETE +-- so its statistics cannot race with this check. +-1U: SELECT wait_until_dead_tup_change_to('vacuum_progress_ao_column'::regclass::oid, 200000); wait_until_dead_tup_change_to ------------------------------- OK @@ -237,10 +236,10 @@ VACUUM (0 rows) -- pg_class and collected stats view should be updated after the 2nd VACUUM -1U: SELECT wait_until_dead_tup_change_to('vacuum_progress_ao_column'::regclass::oid, 0); - wait_until_dead_tup_change_to -------------------------------- - OK +1U: SELECT wait_until_vacuum_count_change_to('vacuum_progress_ao_column'::regclass::oid, 2); + wait_until_vacuum_count_change_to +----------------------------------- + OK (1 row) SELECT relpages, reltuples, relallvisible FROM pg_class where relname = 'vacuum_progress_ao_column'; relpages | reltuples | relallvisible diff --git a/src/test/isolation2/expected/vacuum_progress_row.out b/src/test/isolation2/expected/vacuum_progress_row.out index d700968c324..300f9ad8949 100644 --- a/src/test/isolation2/expected/vacuum_progress_row.out +++ b/src/test/isolation2/expected/vacuum_progress_row.out @@ -40,11 +40,10 @@ ABORT 1: ABORT; ABORT --- Look up the collected stats before the DELETE below. The stats collector --- is asynchronous: wait until it has received the dead tuples of both --- aborted inserts, and read the view before the DELETE, whose own counts --- may reach the collector at any moment after it. -1U: SELECT wait_until_dead_tup_change_to('vacuum_progress_ao_row'::regclass::oid, 200000); +-- Wait on the coordinator for the statistics of both aborted inserts, since +-- the following pg_stat_all_tables query also runs there. Read before DELETE +-- so its statistics cannot race with this check. +-1U: SELECT wait_until_dead_tup_change_to('vacuum_progress_ao_row'::regclass::oid, 200000); wait_until_dead_tup_change_to ------------------------------- OK @@ -73,11 +72,14 @@ SELECT gp_inject_fault('appendonly_after_truncate_segment_file', 'suspend', '', 1: set Debug_appendonly_print_compaction to on; SET 1&: VACUUM vacuum_progress_ao_row; -SELECT gp_wait_until_triggered_fault('appendonly_after_truncate_segment_file', 2, dbid) FROM gp_segment_configuration WHERE content = 1 AND role = 'p'; +-- Wait for all primaries, since the summary below includes every segment. +SELECT gp_wait_until_triggered_fault('appendonly_after_truncate_segment_file', 2, dbid) FROM gp_segment_configuration WHERE content > -1 AND role = 'p'; gp_wait_until_triggered_fault ------------------------------- Success: -(1 row) + Success: + Success: +(3 rows) -- We are in pre_cleanup phase and some blocks should've been vacuumed by now select relid::regclass as relname, phase, heap_blks_total, heap_blks_scanned, heap_blks_vacuumed, index_vacuum_count, max_dead_tuples, num_dead_tuples from gp_stat_progress_vacuum where gp_segment_id = 1; relname | phase | heap_blks_total | heap_blks_scanned | heap_blks_vacuumed | index_vacuum_count | max_dead_tuples | num_dead_tuples @@ -105,11 +107,14 @@ SELECT gp_inject_fault('appendonly_after_truncate_segment_file', 'reset', dbid) Success: Success: (3 rows) -SELECT gp_wait_until_triggered_fault('appendonly_insert', 200, dbid) FROM gp_segment_configuration WHERE content = 1 AND role = 'p'; +-- Wait for all primaries, since the summary below includes every segment. +SELECT gp_wait_until_triggered_fault('appendonly_insert', 200, dbid) FROM gp_segment_configuration WHERE content > -1 AND role = 'p'; gp_wait_until_triggered_fault ------------------------------- Success: -(1 row) + Success: + Success: +(3 rows) -- We are in compact phase. num_dead_tuples should increase as we move and count tuples, one by one. select relid::regclass as relname, phase, heap_blks_total, heap_blks_scanned, heap_blks_vacuumed, index_vacuum_count, max_dead_tuples, num_dead_tuples from gp_stat_progress_vacuum where gp_segment_id = 1; relname | phase | heap_blks_total | heap_blks_scanned | heap_blks_vacuumed | index_vacuum_count | max_dead_tuples | num_dead_tuples diff --git a/src/test/isolation2/sql/vacuum_progress_column.sql b/src/test/isolation2/sql/vacuum_progress_column.sql index 6ed4744a2b8..19bd474a0ae 100644 --- a/src/test/isolation2/sql/vacuum_progress_column.sql +++ b/src/test/isolation2/sql/vacuum_progress_column.sql @@ -27,11 +27,10 @@ CREATE INDEX on vacuum_progress_ao_column(j); -- Abort so that segno 1 has logical EOF = 0. 1: ABORT; --- Look up the collected stats before the DELETE below. The stats collector --- is asynchronous: wait until it has received the dead tuples of both --- aborted inserts, and read the view before the DELETE, whose own counts --- may reach the collector at any moment after it. -1U: SELECT wait_until_dead_tup_change_to('vacuum_progress_ao_column'::regclass::oid, 200000); +-- Wait on the coordinator for the statistics of both aborted inserts, since +-- the following pg_stat_all_tables query also runs there. Read before DELETE +-- so its statistics cannot race with this check. +-1U: SELECT wait_until_dead_tup_change_to('vacuum_progress_ao_column'::regclass::oid, 200000); SELECT n_live_tup, n_dead_tup, last_vacuum, vacuum_count FROM pg_stat_all_tables WHERE relname = 'vacuum_progress_ao_column'; -- Also delete half of the tuples evenly before the EOF of segno 2. @@ -100,7 +99,7 @@ SELECT gp_inject_fault('appendonly_after_truncate_segment_file', 'reset', dbid) 1U: select relid::regclass as relname, phase, heap_blks_total, heap_blks_scanned, heap_blks_vacuumed, index_vacuum_count, max_dead_tuples, num_dead_tuples from pg_stat_progress_vacuum; -- pg_class and collected stats view should be updated after the 2nd VACUUM -1U: SELECT wait_until_dead_tup_change_to('vacuum_progress_ao_column'::regclass::oid, 0); +1U: SELECT wait_until_vacuum_count_change_to('vacuum_progress_ao_column'::regclass::oid, 2); SELECT relpages, reltuples, relallvisible FROM pg_class where relname = 'vacuum_progress_ao_column'; -- SELECT n_live_tup, n_dead_tup, last_vacuum is not null as has_last_vacuum, vacuum_count FROM gp_stat_all_tables WHERE relname = 'vacuum_progress_ao_column' and gp_segment_id = 1; diff --git a/src/test/isolation2/sql/vacuum_progress_row.sql b/src/test/isolation2/sql/vacuum_progress_row.sql index 7b3749d7b70..8fde34ecddb 100644 --- a/src/test/isolation2/sql/vacuum_progress_row.sql +++ b/src/test/isolation2/sql/vacuum_progress_row.sql @@ -26,11 +26,10 @@ CREATE INDEX on vacuum_progress_ao_row(j); -- Abort so that segno 1 has logical EOF = 0. 1: ABORT; --- Look up the collected stats before the DELETE below. The stats collector --- is asynchronous: wait until it has received the dead tuples of both --- aborted inserts, and read the view before the DELETE, whose own counts --- may reach the collector at any moment after it. -1U: SELECT wait_until_dead_tup_change_to('vacuum_progress_ao_row'::regclass::oid, 200000); +-- Wait on the coordinator for the statistics of both aborted inserts, since +-- the following pg_stat_all_tables query also runs there. Read before DELETE +-- so its statistics cannot race with this check. +-1U: SELECT wait_until_dead_tup_change_to('vacuum_progress_ao_row'::regclass::oid, 200000); SELECT n_live_tup, n_dead_tup, last_vacuum, vacuum_count FROM pg_stat_all_tables WHERE relname = 'vacuum_progress_ao_row'; -- Also delete half of the tuples evenly before the EOF of segno 2. @@ -43,7 +42,8 @@ SELECT gp_inject_fault('appendonly_after_truncate_segment_file', 'suspend', '', 1: set Debug_appendonly_print_compaction to on; 1&: VACUUM vacuum_progress_ao_row; -SELECT gp_wait_until_triggered_fault('appendonly_after_truncate_segment_file', 2, dbid) FROM gp_segment_configuration WHERE content = 1 AND role = 'p'; +-- Wait for all primaries, since the summary below includes every segment. +SELECT gp_wait_until_triggered_fault('appendonly_after_truncate_segment_file', 2, dbid) FROM gp_segment_configuration WHERE content > -1 AND role = 'p'; -- We are in pre_cleanup phase and some blocks should've been vacuumed by now select relid::regclass as relname, phase, heap_blks_total, heap_blks_scanned, heap_blks_vacuumed, index_vacuum_count, max_dead_tuples, num_dead_tuples from gp_stat_progress_vacuum where gp_segment_id = 1; select relid::regclass as relname, phase, heap_blks_total, heap_blks_scanned, heap_blks_vacuumed, index_vacuum_count, max_dead_tuples, num_dead_tuples from gp_stat_progress_vacuum_summary; @@ -51,7 +51,8 @@ select relid::regclass as relname, phase, heap_blks_total, heap_blks_scanned, he -- Resume execution and suspend again in the middle of compact phase SELECT gp_inject_fault('appendonly_insert', 'suspend', '', '', '', 200, 200, 0, dbid) FROM gp_segment_configuration WHERE content > -1 AND role = 'p'; SELECT gp_inject_fault('appendonly_after_truncate_segment_file', 'reset', dbid) FROM gp_segment_configuration WHERE content > -1 AND role = 'p'; -SELECT gp_wait_until_triggered_fault('appendonly_insert', 200, dbid) FROM gp_segment_configuration WHERE content = 1 AND role = 'p'; +-- Wait for all primaries, since the summary below includes every segment. +SELECT gp_wait_until_triggered_fault('appendonly_insert', 200, dbid) FROM gp_segment_configuration WHERE content > -1 AND role = 'p'; -- We are in compact phase. num_dead_tuples should increase as we move and count tuples, one by one. select relid::regclass as relname, phase, heap_blks_total, heap_blks_scanned, heap_blks_vacuumed, index_vacuum_count, max_dead_tuples, num_dead_tuples from gp_stat_progress_vacuum where gp_segment_id = 1; select relid::regclass as relname, phase, heap_blks_total, heap_blks_scanned, heap_blks_vacuumed, index_vacuum_count, max_dead_tuples, num_dead_tuples from gp_stat_progress_vacuum_summary;