diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index befdfbd..06ad616 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -177,10 +177,9 @@ jobs: (cd $GITHUB_WORKSPACE/test/lua && $GITHUB_WORKSPACE/build/bin/unit_tests) ./bin/flua - # 单独一步而不是内联 if:这样 PR 里 GitHub 会把它显式标成 skipped,看得出是按策略 - # 跳过而不是没配。限定 release 是因为这是 matrix 构建,不限定的话一次 push 要跑四遍。 + # 限定 release:matrix 有 debug/asan/cov,不限定的话一次 push 要跑四遍。 - name: Run benchmarks - if: github.event_name != 'pull_request' && matrix.build_type == 'release' + if: matrix.build_type == 'release' run: | set -x cd build/bin diff --git a/.github/workflows/build_with_macos.yml b/.github/workflows/build_with_macos.yml index 860047d..b380475 100644 --- a/.github/workflows/build_with_macos.yml +++ b/.github/workflows/build_with_macos.yml @@ -99,10 +99,7 @@ jobs: (cd $GITHUB_WORKSPACE/test/lua && $GITHUB_WORKSPACE/build/bin/unit_tests) ./bin/flua - # 单独一步而不是内联 if:这样 PR 里 GitHub 会把它显式标成 skipped,看得出是按策略 - # 跳过而不是没配。 - name: Run benchmarks - if: github.event_name != 'pull_request' run: | set -x cd build/bin diff --git a/.github/workflows/build_with_windows.yml b/.github/workflows/build_with_windows.yml index 13f62bc..7feb255 100644 --- a/.github/workflows/build_with_windows.yml +++ b/.github/workflows/build_with_windows.yml @@ -178,7 +178,6 @@ jobs: ./flua.exe - name: Run benchmarks - if: github.event_name != 'pull_request' shell: msys2 {0} run: | set -eux diff --git a/benchmark/README.md b/benchmark/README.md index f43e559..77d20b7 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -16,8 +16,6 @@ build/bin/bench_mark --benchmark_filter='BM_Lua_|BM_FakeLua_.*_INTERP' \ build/bin/bench_mark --benchmark_repetitions=1 --benchmark_report_aggregates_only=true ``` -`BM_FakeLua_TailRecursion_INTERP/5000` is a 5000-deep C++ call chain (the interpreter does not turn tail calls into loops). The default 8 MiB C stack overflows; run with `ulimit -s unlimited` if that case is included. - --- ## Interpreter vs Lua 5.4 (2026-09-16) diff --git a/benchmark/README.zh.md b/benchmark/README.zh.md index d3abc60..aa3b7bc 100644 --- a/benchmark/README.zh.md +++ b/benchmark/README.zh.md @@ -14,8 +14,6 @@ build/bin/bench_mark --benchmark_filter='BM_Lua_|BM_FakeLua_.*_INTERP' \ build/bin/bench_mark --benchmark_repetitions=1 --benchmark_report_aggregates_only=true ``` -`BM_FakeLua_TailRecursion_INTERP/5000` 是 5000 层 C++ 调用(解释器没有把尾调用收成循环)。默认 8 MiB C 栈会溢出;若包含该用例请先 `ulimit -s unlimited`。 - --- ## 解释器 vs Lua 5.4(2026-09-16) diff --git a/benchmark/benchmark_function.cpp b/benchmark/benchmark_function.cpp index 65093dd..8a71abb 100644 --- a/benchmark/benchmark_function.cpp +++ b/benchmark/benchmark_function.cpp @@ -70,19 +70,6 @@ function bench_closure(n) end )"; -constexpr const char *kTailRecursionScript = R"( -function bench_tail_sum_acc(x, acc) - if x <= 0 then - return acc - end - return bench_tail_sum_acc(x - 1, acc + x) -end - -function bench_tail_sum(n) - return bench_tail_sum_acc(n, 0) -end -)"; - // C++ reference implementations inline int64_t CppEmpty(int64_t n) { @@ -129,18 +116,10 @@ int64_t CppClosure(int64_t n) { return total; } -int64_t CppTailSum(int64_t n) { - int64_t acc = 0; - for (int64_t x = n; x > 0; --x) { - acc += x; - } - return acc; -} - // Lua helpers const char *const kFunctionScripts[] = { - kEmptyCallScript, kRecursionScript, kVariadicScript, kMultiReturnScript, kClosureScript, kTailRecursionScript, + kEmptyCallScript, kRecursionScript, kVariadicScript, kMultiReturnScript, kClosureScript, }; constexpr size_t kFunctionScriptCount = sizeof(kFunctionScripts) / sizeof(kFunctionScripts[0]); @@ -159,8 +138,6 @@ struct Ctx : RuntimeContext { Call(flua, JIT_INTERP, "bench_multi_return", w, 10); Call(flua, JIT_TCC, "bench_closure", w, 10); Call(flua, JIT_INTERP, "bench_closure", w, 10); - Call(flua, JIT_TCC, "bench_tail_sum", w, 10); - Call(flua, JIT_INTERP, "bench_tail_sum", w, 10); } ~Ctx() { @@ -392,51 +369,6 @@ static void BM_FakeLua_Closure_INTERP(benchmark::State &state) { } } -// Benchmarks: tail recursion - -static void BM_CPP_TailRecursion(benchmark::State &state) { - const int64_t n = state.range(0); - for (auto _: state) { - int64_t ret = CppTailSum(n); - benchmark::DoNotOptimize(ret); - } -} - -static void BM_Lua_TailRecursion(benchmark::State &state) { - const int64_t n = state.range(0); - for (auto _: state) { - int64_t ret = CallLuaInt(g_ctx.lua, "bench_tail_sum", n); - benchmark::DoNotOptimize(ret); - } -} - -static void BM_FakeLua_TailRecursion_TCC(benchmark::State &state) { - const int64_t n = state.range(0); - for (auto _: state) { - int64_t ret = 0; - Call(g_ctx.flua, JIT_TCC, "bench_tail_sum", ret, n); - benchmark::DoNotOptimize(ret); - } -} - -static void BM_FakeLua_TailRecursion_GCC(benchmark::State &state) { - const int64_t n = state.range(0); - for (auto _: state) { - int64_t ret = 0; - Call(g_ctx.flua, JIT_GCC, "bench_tail_sum", ret, n); - benchmark::DoNotOptimize(ret); - } -} - -static void BM_FakeLua_TailRecursion_INTERP(benchmark::State &state) { - const int64_t n = state.range(0); - for (auto _: state) { - int64_t ret = 0; - Call(g_ctx.flua, JIT_INTERP, "bench_tail_sum", ret, n); - benchmark::DoNotOptimize(ret); - } -} - }// namespace // Benchmark registrations @@ -446,7 +378,6 @@ static void BM_FakeLua_TailRecursion_INTERP(benchmark::State &state) { #define VARIADIC_ARGS ->Arg(1) #define MULTI_RETURN_ARGS ->Arg(1000)->Arg(10000) #define CLOSURE_ARGS ->Arg(100)->Arg(1000) -#define TAIL_RECURSION_ARGS ->Arg(100)->Arg(1000)->Arg(5000) BENCHMARK(BM_CPP_EmptyCall) EMPTY_CALL_ARGS; BENCHMARK(BM_Lua_EmptyCall) EMPTY_CALL_ARGS; @@ -472,9 +403,4 @@ BENCHMARK(BM_CPP_Closure) CLOSURE_ARGS; BENCHMARK(BM_Lua_Closure) CLOSURE_ARGS; BENCHMARK(BM_FakeLua_Closure_TCC) CLOSURE_ARGS; BENCHMARK(BM_FakeLua_Closure_GCC) CLOSURE_ARGS; -BENCHMARK(BM_FakeLua_Closure_INTERP) CLOSURE_ARGS; -BENCHMARK(BM_CPP_TailRecursion) TAIL_RECURSION_ARGS; -BENCHMARK(BM_Lua_TailRecursion) TAIL_RECURSION_ARGS; -BENCHMARK(BM_FakeLua_TailRecursion_TCC) TAIL_RECURSION_ARGS; -BENCHMARK(BM_FakeLua_TailRecursion_GCC) TAIL_RECURSION_ARGS; -BENCHMARK(BM_FakeLua_TailRecursion_INTERP) TAIL_RECURSION_ARGS; \ No newline at end of file +BENCHMARK(BM_FakeLua_Closure_INTERP) CLOSURE_ARGS; \ No newline at end of file diff --git a/test/lua/jit/test_tail_call.lua b/test/lua/jit/test_tail_call.lua new file mode 100644 index 0000000..d8a74d9 --- /dev/null +++ b/test/lua/jit/test_tail_call.lua @@ -0,0 +1,107 @@ +-- Tail-call correctness. Depth stays modest so INTERP (no TCO) does not overflow the C stack. + +function tail_sum(x, acc) + if x <= 0 then + return acc + end + return tail_sum(x - 1, acc + x) +end + +function test_tail_sum() + if tail_sum(0, 0) ~= 0 then return 0 end + if tail_sum(1, 0) ~= 1 then return 0 end + if tail_sum(10, 0) ~= 55 then return 0 end + if tail_sum(100, 0) ~= 5050 then return 0 end + return 1 +end + +function even(n) + if n == 0 then return true end + return odd(n - 1) +end + +function odd(n) + if n == 0 then return false end + return even(n - 1) +end + +function test_tail_even_odd() + if even(0) ~= true then return 0 end + if odd(0) ~= false then return 0 end + if even(10) ~= true then return 0 end + if odd(11) ~= true then return 0 end + if even(99) ~= false then return 0 end + return 1 +end + +function ident(x) + return x +end + +function tail_through(x) + return ident(x + 1) +end + +function test_tail_other_func() + if tail_through(41) ~= 42 then return 0 end + return 1 +end + +function two_rets(a, b) + return a, b +end + +function tail_multi(x) + return two_rets(x, x + 1) +end + +function test_tail_multi_ret() + local a, b = tail_multi(7) + if a ~= 7 then return 0 end + if b ~= 8 then return 0 end + return 1 +end + +function add1(a) + return a +end + +function add3(a, b, c) + return a + b + c +end + +function tail_to_add1(x) + return add1(x) +end + +function tail_to_add3(x) + return add3(x, 2, 3) +end + +function test_tail_arity() + if tail_to_add1(5) ~= 5 then return 0 end + if tail_to_add3(5) ~= 10 then return 0 end + return 1 +end + +function tail_count(...) + local n = select('#', ...) + if n <= 1 then return n end + return tail_count(select(2, ...)) +end + +function test_tail_vararg() + if tail_count() ~= 0 then return 0 end + if tail_count(1) ~= 1 then return 0 end + if tail_count(1, 2, 3, 4, 5) ~= 1 then return 0 end + return 1 +end + +function test_tail_closure() + local function walk(n, acc) + if n <= 0 then return acc end + return walk(n - 1, acc + n) + end + if walk(20, 0) ~= 210 then return 0 end + return 1 +end diff --git a/test/test_jitter.cpp b/test/test_jitter.cpp index 28a2dc5..00b8ab2 100644 --- a/test/test_jitter.cpp +++ b/test/test_jitter.cpp @@ -3006,3 +3006,24 @@ TEST(jitter, spec_field_merge_sanitize) { ASSERT_EQ(ret, 2); }); } + +TEST(jitter, tail_call) { + JitterRunHelper([](State *s, JITType type, bool debug_mode) { + CompileFile(s, "./jit/test_tail_call.lua", {.debug_mode = debug_mode}); + int64_t ret = 0; + Call(s, type, "test_tail_sum", ret); + ASSERT_EQ(ret, 1); + Call(s, type, "test_tail_even_odd", ret); + ASSERT_EQ(ret, 1); + Call(s, type, "test_tail_other_func", ret); + ASSERT_EQ(ret, 1); + Call(s, type, "test_tail_multi_ret", ret); + ASSERT_EQ(ret, 1); + Call(s, type, "test_tail_arity", ret); + ASSERT_EQ(ret, 1); + Call(s, type, "test_tail_vararg", ret); + ASSERT_EQ(ret, 1); + Call(s, type, "test_tail_closure", ret); + ASSERT_EQ(ret, 1); + }); +}