From f697f2a83b2401f61f7752ffced9c949a8398c65 Mon Sep 17 00:00:00 2001 From: Dianjin Wang Date: Tue, 1 Sep 2026 21:23:23 +0800 Subject: [PATCH 1/4] Fix flaky autovacuum-analyze tests Replace pg_sleep() with gp_stat_force_next_flush() to force a stats flush instead of waiting. Update the expected output. Assisted-by: Claude Code Backpatch-through: REL_2_STABLE --- .../input/autovacuum-analyze.source | 6 +++--- .../output/autovacuum-analyze.source | 21 ++++++++++++++++--- .../input/autovacuum-analyze.source | 6 +++--- .../output/autovacuum-analyze.source | 21 ++++++++++++++++--- .../input/autovacuum-analyze.source | 6 +++--- .../output/autovacuum-analyze.source | 21 ++++++++++++++++--- 6 files changed, 63 insertions(+), 18 deletions(-) diff --git a/contrib/pax_storage/src/test/isolation2/input/autovacuum-analyze.source b/contrib/pax_storage/src/test/isolation2/input/autovacuum-analyze.source index 32e79cdd491..7b4bb955d5d 100644 --- a/contrib/pax_storage/src/test/isolation2/input/autovacuum-analyze.source +++ b/contrib/pax_storage/src/test/isolation2/input/autovacuum-analyze.source @@ -171,7 +171,7 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' -- with auto_stats, the auto-ANALYZE still trigger 2: INSERT INTO autostatstbl select i from generate_series(1, 1000) as i; -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. SELECT count(*) FROM pg_statistic where starelid = 'autostatstbl'::regclass; @@ -205,7 +205,7 @@ SELECT gp_inject_fault('analyze_finished_one_relation', 'skip', '', '', 'autosta SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', '', 'autostatstbl', 1, -1, 0, 1); 2: INSERT INTO autostatstbl select i from generate_series(1001, 2000) as i; -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; @@ -238,7 +238,7 @@ SELECT gp_inject_fault('analyze_finished_one_relation', 'skip', '', '', 'autosta SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', '', 'autostatstbl', 1, -1, 0, 1); 2: INSERT INTO autostatstbl select i from generate_series(2001, 3000) as i; -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); -- auto_stats should not executed and auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; diff --git a/contrib/pax_storage/src/test/isolation2/output/autovacuum-analyze.source b/contrib/pax_storage/src/test/isolation2/output/autovacuum-analyze.source index 0c4b87746e4..b835fdbcc96 100644 --- a/contrib/pax_storage/src/test/isolation2/output/autovacuum-analyze.source +++ b/contrib/pax_storage/src/test/isolation2/output/autovacuum-analyze.source @@ -408,7 +408,12 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' -- with auto_stats, the auto-ANALYZE still trigger 2: INSERT INTO autostatstbl select i from generate_series(1, 1000) as i; INSERT 1000 -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); + gp_stat_force_next_flush +-------------------------- + +(1 row) + -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. SELECT count(*) FROM pg_statistic where starelid = 'autostatstbl'::regclass; count @@ -493,7 +498,12 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' 2: INSERT INTO autostatstbl select i from generate_series(1001, 2000) as i; INSERT 1000 -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); + gp_stat_force_next_flush +-------------------------- + +(1 row) + -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; relpages | reltuples @@ -572,7 +582,12 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' 2: INSERT INTO autostatstbl select i from generate_series(2001, 3000) as i; INSERT 1000 -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); + gp_stat_force_next_flush +-------------------------- + +(1 row) + -- auto_stats should not executed and auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; relpages | reltuples diff --git a/src/test/isolation2/input/autovacuum-analyze.source b/src/test/isolation2/input/autovacuum-analyze.source index 32e79cdd491..7b4bb955d5d 100644 --- a/src/test/isolation2/input/autovacuum-analyze.source +++ b/src/test/isolation2/input/autovacuum-analyze.source @@ -171,7 +171,7 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' -- with auto_stats, the auto-ANALYZE still trigger 2: INSERT INTO autostatstbl select i from generate_series(1, 1000) as i; -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. SELECT count(*) FROM pg_statistic where starelid = 'autostatstbl'::regclass; @@ -205,7 +205,7 @@ SELECT gp_inject_fault('analyze_finished_one_relation', 'skip', '', '', 'autosta SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', '', 'autostatstbl', 1, -1, 0, 1); 2: INSERT INTO autostatstbl select i from generate_series(1001, 2000) as i; -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; @@ -238,7 +238,7 @@ SELECT gp_inject_fault('analyze_finished_one_relation', 'skip', '', '', 'autosta SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', '', 'autostatstbl', 1, -1, 0, 1); 2: INSERT INTO autostatstbl select i from generate_series(2001, 3000) as i; -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); -- auto_stats should not executed and auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; diff --git a/src/test/isolation2/output/autovacuum-analyze.source b/src/test/isolation2/output/autovacuum-analyze.source index e071f9e7007..c2e7877b324 100644 --- a/src/test/isolation2/output/autovacuum-analyze.source +++ b/src/test/isolation2/output/autovacuum-analyze.source @@ -408,7 +408,12 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' -- with auto_stats, the auto-ANALYZE still trigger 2: INSERT INTO autostatstbl select i from generate_series(1, 1000) as i; INSERT 1000 -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); + gp_stat_force_next_flush +-------------------------- + +(1 row) + -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. SELECT count(*) FROM pg_statistic where starelid = 'autostatstbl'::regclass; count @@ -493,7 +498,12 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' 2: INSERT INTO autostatstbl select i from generate_series(1001, 2000) as i; INSERT 1000 -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); + gp_stat_force_next_flush +-------------------------- + +(1 row) + -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; relpages | reltuples @@ -572,7 +582,12 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' 2: INSERT INTO autostatstbl select i from generate_series(2001, 3000) as i; INSERT 1000 -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); + gp_stat_force_next_flush +-------------------------- + +(1 row) + -- auto_stats should not executed and auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; relpages | reltuples diff --git a/src/test/singlenode_isolation2/input/autovacuum-analyze.source b/src/test/singlenode_isolation2/input/autovacuum-analyze.source index b35c3072a85..1e1fcb69f2a 100644 --- a/src/test/singlenode_isolation2/input/autovacuum-analyze.source +++ b/src/test/singlenode_isolation2/input/autovacuum-analyze.source @@ -170,7 +170,7 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' -- with auto_stats, the auto-ANALYZE still trigger 2: INSERT INTO autostatstbl select i from generate_series(1, 1000) as i; -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. SELECT count(*) FROM pg_statistic where starelid = 'autostatstbl'::regclass; @@ -204,7 +204,7 @@ SELECT gp_inject_fault('analyze_finished_one_relation', 'skip', '', '', 'autosta SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', '', 'autostatstbl', 1, -1, 0, 1); 2: INSERT INTO autostatstbl select i from generate_series(1001, 2000) as i; -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; @@ -237,7 +237,7 @@ SELECT gp_inject_fault('analyze_finished_one_relation', 'skip', '', '', 'autosta SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', '', 'autostatstbl', 1, -1, 0, 1); 2: INSERT INTO autostatstbl select i from generate_series(2001, 3000) as i; -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); -- auto_stats should not executed and auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; diff --git a/src/test/singlenode_isolation2/output/autovacuum-analyze.source b/src/test/singlenode_isolation2/output/autovacuum-analyze.source index 382b443944c..9ab7afa90ee 100644 --- a/src/test/singlenode_isolation2/output/autovacuum-analyze.source +++ b/src/test/singlenode_isolation2/output/autovacuum-analyze.source @@ -403,7 +403,12 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' -- with auto_stats, the auto-ANALYZE still trigger 2: INSERT INTO autostatstbl select i from generate_series(1, 1000) as i; INSERT 1000 -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); + gp_stat_force_next_flush +-------------------------- + +(1 row) + -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. SELECT count(*) FROM pg_statistic where starelid = 'autostatstbl'::regclass; count @@ -488,7 +493,12 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' 2: INSERT INTO autostatstbl select i from generate_series(1001, 2000) as i; INSERT 1000 -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); + gp_stat_force_next_flush +-------------------------- + +(1 row) + -- auto_stats executed but auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; relpages | reltuples @@ -567,7 +577,12 @@ SELECT gp_inject_fault('auto_vac_worker_after_report_activity', 'suspend', '', ' 2: INSERT INTO autostatstbl select i from generate_series(2001, 3000) as i; INSERT 1000 -2: select pg_sleep(0.77); -- Force pgstat_report_stat() to send tabstat. +2: select gp_stat_force_next_flush(); + gp_stat_force_next_flush +-------------------------- + +(1 row) + -- auto_stats should not executed and auto-ANALYZE not execute yet since we suspend before finish ANALYZE. select relpages, reltuples from pg_class where oid = 'autostatstbl'::regclass; relpages | reltuples From 734c0cb51b4907d9579309cba5e927fc8bcd64ef Mon Sep 17 00:00:00 2001 From: Dianjin Wang Date: Tue, 1 Sep 2026 23:09:19 +0800 Subject: [PATCH 2/4] Wait for the wal_sender_loop fault to trigger before resuming VACUUM. Assisted-by: Claude Code Backpatch-through: REL_2_STABLE --- .../isolation2/expected/vacuum_progress_column.out | 11 +++++++++++ src/test/isolation2/sql/vacuum_progress_column.sql | 7 +++++++ 2 files changed, 18 insertions(+) diff --git a/src/test/isolation2/expected/vacuum_progress_column.out b/src/test/isolation2/expected/vacuum_progress_column.out index 587bd35f7c8..566dc64e503 100644 --- a/src/test/isolation2/expected/vacuum_progress_column.out +++ b/src/test/isolation2/expected/vacuum_progress_column.out @@ -334,11 +334,22 @@ LINE 1: ...cuum_count, max_dead_tuples, num_dead_tuples from gp_stat_pr... ^ -- Resume execution of compact phase and block at syncrep on one segment. +-- The suspend only takes effect the next time the walsender goes round its +-- loop, so wait until it is really parked before letting the vacuum go on. +-- Otherwise the compact phase can commit while the walsender is still +-- streaming, syncrep is satisfied by the live mirror, the vacuum runs to +-- completion on the same gang, no new vacuum worker ever takes over and the +-- gp_wait_until_triggered_fault() below times out after ten minutes. 2: SELECT gp_inject_fault_infinite('wal_sender_loop', 'suspend', dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; gp_inject_fault_infinite -------------------------- Success: (1 row) +2: SELECT gp_wait_until_triggered_fault('wal_sender_loop', 1, dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; + gp_wait_until_triggered_fault +------------------------------- + Success: +(1 row) 2: SELECT gp_inject_fault('vacuum_ao_after_compact', 'reset', dbid) FROM gp_segment_configuration WHERE content > -1 AND role = 'p'; gp_inject_fault ----------------- diff --git a/src/test/isolation2/sql/vacuum_progress_column.sql b/src/test/isolation2/sql/vacuum_progress_column.sql index 648d43303be..d89ccb658d5 100644 --- a/src/test/isolation2/sql/vacuum_progress_column.sql +++ b/src/test/isolation2/sql/vacuum_progress_column.sql @@ -139,7 +139,14 @@ select gp_segment_id, relid::regclass as relname, phase, heap_blks_total, heap_b select relid::regclass as relname, phase, heap_blks_total, heap_blks_scanned, heap_blks_vacuumed, index_vacuum_count, max_dead_tuples, num_dead_tuples from gp_stat_progress_vacuum_summary; -- Resume execution of compact phase and block at syncrep on one segment. +-- The suspend only takes effect the next time the walsender goes round its +-- loop, so wait until it is really parked before letting the vacuum go on. +-- Otherwise the compact phase can commit while the walsender is still +-- streaming, syncrep is satisfied by the live mirror, the vacuum runs to +-- completion on the same gang, no new vacuum worker ever takes over and the +-- gp_wait_until_triggered_fault() below times out after ten minutes. 2: SELECT gp_inject_fault_infinite('wal_sender_loop', 'suspend', dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; +2: SELECT gp_wait_until_triggered_fault('wal_sender_loop', 1, dbid) FROM gp_segment_configuration WHERE role = 'p' and content = 1; 2: SELECT gp_inject_fault('vacuum_ao_after_compact', 'reset', dbid) FROM gp_segment_configuration WHERE content > -1 AND role = 'p'; -- stop the mirror should turn off syncrep 2: SELECT pg_ctl(datadir, 'stop', 'immediate') FROM gp_segment_configuration WHERE content = 1 AND role = 'm'; From e01628894b25b740dabd81036a4bea0c9761afa9 Mon Sep 17 00:00:00 2001 From: Dianjin Wang Date: Tue, 1 Sep 2026 21:23:55 +0800 Subject: [PATCH 3/4] Fix expected CPU usage in resgroup tests The test already averages CPU samples, but the expected values are still too low. The suite sets the cluster CPU limit to 100%, so expect 100% for one uncapped group and 33%/67% for groups with weights 100/200. Keep the existing tolerance of 10 percentage points. Assisted-by: Claude Code Backpatch-through: REL_2_STABLE --- .../expected/resgroup/resgroup_cpu_max_percent.out | 12 +++++++++--- .../sql/resgroup/resgroup_cpu_max_percent.sql | 12 +++++++++--- .../expected/resgroup/resgroup_cpu_max_percent.out | 12 +++++++++--- .../sql/resgroup/resgroup_cpu_max_percent.sql | 12 +++++++++--- 4 files changed, 36 insertions(+), 12 deletions(-) diff --git a/contrib/pax_storage/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out b/contrib/pax_storage/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out index ae21059fcb6..bcabd53f840 100644 --- a/contrib/pax_storage/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out +++ b/contrib/pax_storage/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out @@ -220,7 +220,11 @@ SELECT pg_sleep(1.7); (1 row) -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 90, 10); +-- rg1_cpu_test is uncapped (cpu_max_percent=-1) and it is the only busy +-- group, so it takes essentially every core: gp_resgroup_status reports +-- ~100, not 90. Expecting 90 put the real value on the upper edge of the +-- +/- err_rate window, so any upward sampling jitter failed the test. +SELECT verify_cpu_usage('rg1_cpu_test', 100, 10); verify_cpu_usage ------------------ t @@ -395,12 +399,14 @@ SELECT pg_sleep(1.7); (1 row) -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 30, 10); +-- Both groups are uncapped, so they share the whole machine in proportion +-- to cpu_weight (100 and 200): ~33 and ~67, not ~30 and ~60. +SELECT verify_cpu_usage('rg1_cpu_test', 33, 10); verify_cpu_usage ------------------ t (1 row) -SELECT verify_cpu_usage('rg2_cpu_test', 60, 10); +SELECT verify_cpu_usage('rg2_cpu_test', 67, 10); verify_cpu_usage ------------------ t diff --git a/contrib/pax_storage/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql b/contrib/pax_storage/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql index 14220c38a42..df04bac4436 100644 --- a/contrib/pax_storage/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql +++ b/contrib/pax_storage/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql @@ -142,7 +142,11 @@ SELECT fetch_sample(); SELECT pg_sleep(1.7); -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 90, 10); +-- rg1_cpu_test is uncapped (cpu_max_percent=-1) and it is the only busy +-- group, so it takes essentially every core: gp_resgroup_status reports +-- ~100, not 90. Expecting 90 put the real value on the upper edge of the +-- +/- err_rate window, so any upward sampling jitter failed the test. +SELECT verify_cpu_usage('rg1_cpu_test', 100, 10); -- start_ignore SELECT * FROM cancel_all; @@ -213,8 +217,10 @@ SELECT fetch_sample(); SELECT pg_sleep(1.7); -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 30, 10); -SELECT verify_cpu_usage('rg2_cpu_test', 60, 10); +-- Both groups are uncapped, so they share the whole machine in proportion +-- to cpu_weight (100 and 200): ~33 and ~67, not ~30 and ~60. +SELECT verify_cpu_usage('rg1_cpu_test', 33, 10); +SELECT verify_cpu_usage('rg2_cpu_test', 67, 10); -- start_ignore SELECT * FROM cancel_all; diff --git a/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out b/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out index ae21059fcb6..bcabd53f840 100644 --- a/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out +++ b/src/test/isolation2/expected/resgroup/resgroup_cpu_max_percent.out @@ -220,7 +220,11 @@ SELECT pg_sleep(1.7); (1 row) -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 90, 10); +-- rg1_cpu_test is uncapped (cpu_max_percent=-1) and it is the only busy +-- group, so it takes essentially every core: gp_resgroup_status reports +-- ~100, not 90. Expecting 90 put the real value on the upper edge of the +-- +/- err_rate window, so any upward sampling jitter failed the test. +SELECT verify_cpu_usage('rg1_cpu_test', 100, 10); verify_cpu_usage ------------------ t @@ -395,12 +399,14 @@ SELECT pg_sleep(1.7); (1 row) -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 30, 10); +-- Both groups are uncapped, so they share the whole machine in proportion +-- to cpu_weight (100 and 200): ~33 and ~67, not ~30 and ~60. +SELECT verify_cpu_usage('rg1_cpu_test', 33, 10); verify_cpu_usage ------------------ t (1 row) -SELECT verify_cpu_usage('rg2_cpu_test', 60, 10); +SELECT verify_cpu_usage('rg2_cpu_test', 67, 10); verify_cpu_usage ------------------ t diff --git a/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql b/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql index 14220c38a42..df04bac4436 100644 --- a/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql +++ b/src/test/isolation2/sql/resgroup/resgroup_cpu_max_percent.sql @@ -142,7 +142,11 @@ SELECT fetch_sample(); SELECT pg_sleep(1.7); -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 90, 10); +-- rg1_cpu_test is uncapped (cpu_max_percent=-1) and it is the only busy +-- group, so it takes essentially every core: gp_resgroup_status reports +-- ~100, not 90. Expecting 90 put the real value on the upper edge of the +-- +/- err_rate window, so any upward sampling jitter failed the test. +SELECT verify_cpu_usage('rg1_cpu_test', 100, 10); -- start_ignore SELECT * FROM cancel_all; @@ -213,8 +217,10 @@ SELECT fetch_sample(); SELECT pg_sleep(1.7); -- end_ignore -SELECT verify_cpu_usage('rg1_cpu_test', 30, 10); -SELECT verify_cpu_usage('rg2_cpu_test', 60, 10); +-- Both groups are uncapped, so they share the whole machine in proportion +-- to cpu_weight (100 and 200): ~33 and ~67, not ~30 and ~60. +SELECT verify_cpu_usage('rg1_cpu_test', 33, 10); +SELECT verify_cpu_usage('rg2_cpu_test', 67, 10); -- start_ignore SELECT * FROM cancel_all; From 335738d0f385151cb956fe234b7477a3ad086bcf Mon Sep 17 00:00:00 2001 From: Dianjin Wang Date: Thu, 17 Sep 2026 11:08:10 +0800 Subject: [PATCH 4/4] Disable tasks created to test schedule syntax These tasks only check that valid schedules are accepted. Deactivate them after creation to avoid background runs interfering with cleanup and the oid_wraparound test. Assisted-by: Claude Code Backpatch-through: REL_2_STABLE --- .../pax_storage/src/test/regress/expected/task.out | 13 +++++++++++++ contrib/pax_storage/src/test/regress/sql/task.sql | 13 +++++++++++++ src/test/regress/expected/task.out | 13 +++++++++++++ src/test/regress/sql/task.sql | 13 +++++++++++++ 4 files changed, 52 insertions(+) diff --git a/contrib/pax_storage/src/test/regress/expected/task.out b/contrib/pax_storage/src/test/regress/expected/task.out index bb9a72ac503..c8a7a2d4aa4 100644 --- a/contrib/pax_storage/src/test/regress/expected/task.out +++ b/contrib/pax_storage/src/test/regress/expected/task.out @@ -69,10 +69,23 @@ ERROR: User task_cron does not have CONNECT privilege on task_dbno alter task vacuum_db user hopedoesnotexist; ERROR: role "hopedoesnotexist" does not exist -- valid interval tasks +-- +-- These four only exercise the schedule parser, they are never meant to run. +-- Deactivate each one as soon as it exists: while a second-based task stays +-- active the scheduler keeps firing it, and every run draws a run id from the +-- cluster-wide Oid counter (NextRunId -> GetNewOidWithIndex). A task still +-- firing later in the same regression run shifts the counter that +-- oid_wraparound hardcodes, and DROP TASK, which also deletes the task's +-- pg_task_run_history rows, fails with "tuple concurrently updated" when it +-- races the run the scheduler is in the middle of recording. create task valid_task_1 schedule '1 second' as 'select 1'; +alter task valid_task_1 not active; create task valid_task_2 schedule ' 30 sEcOnDs ' as 'select 1'; +alter task valid_task_2 not active; create task valid_task_3 schedule '59 seconds' as 'select 1'; +alter task valid_task_3 not active; create task valid_task_4 schedule '17 seconds ' as 'select 1'; +alter task valid_task_4 not active; -- clean up drop database task_dbno; drop user task_cron; diff --git a/contrib/pax_storage/src/test/regress/sql/task.sql b/contrib/pax_storage/src/test/regress/sql/task.sql index 217b265daf9..46cdcf4198b 100644 --- a/contrib/pax_storage/src/test/regress/sql/task.sql +++ b/contrib/pax_storage/src/test/regress/sql/task.sql @@ -49,10 +49,23 @@ alter task vacuum_db database task_dbno user task_cron; alter task vacuum_db user hopedoesnotexist; -- valid interval tasks +-- +-- These four only exercise the schedule parser, they are never meant to run. +-- Deactivate each one as soon as it exists: while a second-based task stays +-- active the scheduler keeps firing it, and every run draws a run id from the +-- cluster-wide Oid counter (NextRunId -> GetNewOidWithIndex). A task still +-- firing later in the same regression run shifts the counter that +-- oid_wraparound hardcodes, and DROP TASK, which also deletes the task's +-- pg_task_run_history rows, fails with "tuple concurrently updated" when it +-- races the run the scheduler is in the middle of recording. create task valid_task_1 schedule '1 second' as 'select 1'; +alter task valid_task_1 not active; create task valid_task_2 schedule ' 30 sEcOnDs ' as 'select 1'; +alter task valid_task_2 not active; create task valid_task_3 schedule '59 seconds' as 'select 1'; +alter task valid_task_3 not active; create task valid_task_4 schedule '17 seconds ' as 'select 1'; +alter task valid_task_4 not active; -- clean up drop database task_dbno; diff --git a/src/test/regress/expected/task.out b/src/test/regress/expected/task.out index 2f052fdad1a..3f4cf9c14ed 100644 --- a/src/test/regress/expected/task.out +++ b/src/test/regress/expected/task.out @@ -69,10 +69,23 @@ ERROR: User task_cron does not have CONNECT privilege on task_dbno alter task vacuum_db user hopedoesnotexist; ERROR: role "hopedoesnotexist" does not exist -- valid interval tasks +-- +-- These four only exercise the schedule parser, they are never meant to run. +-- Deactivate each one as soon as it exists: while a second-based task stays +-- active the scheduler keeps firing it, and every run draws a run id from the +-- cluster-wide Oid counter (NextRunId -> GetNewOidWithIndex). A task still +-- firing later in the same regression run shifts the counter that +-- oid_wraparound hardcodes, and DROP TASK, which also deletes the task's +-- pg_task_run_history rows, fails with "tuple concurrently updated" when it +-- races the run the scheduler is in the middle of recording. create task valid_task_1 schedule '1 second' as 'select 1'; +alter task valid_task_1 not active; create task valid_task_2 schedule ' 30 sEcOnDs ' as 'select 1'; +alter task valid_task_2 not active; create task valid_task_3 schedule '59 seconds' as 'select 1'; +alter task valid_task_3 not active; create task valid_task_4 schedule '17 seconds ' as 'select 1'; +alter task valid_task_4 not active; -- task in function DO $$ BEGIN diff --git a/src/test/regress/sql/task.sql b/src/test/regress/sql/task.sql index bf12909c761..3e29c46adad 100644 --- a/src/test/regress/sql/task.sql +++ b/src/test/regress/sql/task.sql @@ -49,10 +49,23 @@ alter task vacuum_db database task_dbno user task_cron; alter task vacuum_db user hopedoesnotexist; -- valid interval tasks +-- +-- These four only exercise the schedule parser, they are never meant to run. +-- Deactivate each one as soon as it exists: while a second-based task stays +-- active the scheduler keeps firing it, and every run draws a run id from the +-- cluster-wide Oid counter (NextRunId -> GetNewOidWithIndex). A task still +-- firing later in the same regression run shifts the counter that +-- oid_wraparound hardcodes, and DROP TASK, which also deletes the task's +-- pg_task_run_history rows, fails with "tuple concurrently updated" when it +-- races the run the scheduler is in the middle of recording. create task valid_task_1 schedule '1 second' as 'select 1'; +alter task valid_task_1 not active; create task valid_task_2 schedule ' 30 sEcOnDs ' as 'select 1'; +alter task valid_task_2 not active; create task valid_task_3 schedule '59 seconds' as 'select 1'; +alter task valid_task_3 not active; create task valid_task_4 schedule '17 seconds ' as 'select 1'; +alter task valid_task_4 not active; -- task in function DO $$