From 65772c6be36aee53f623f63c6e7df51bcf86defe Mon Sep 17 00:00:00 2001 From: Matthias Goergens Date: Sun, 27 Sep 2026 14:34:38 +0000 Subject: [PATCH] gh-158283: Force dispatch tail duplication with Clang 19 and Apple clang 17 LLVM 19 limits tail duplication of blocks ending in an indirect branch (llvm/llvm-project#78582), so the computed-goto interpreter is compiled with a single shared dispatch jump instead of one per instruction. Apple clang from Xcode 16.3-16.4 has the same bug; Xcode 26.0-26.3 merges most of the dispatch jumps. LLVM 20.1.1 fixed it (llvm/llvm-project#114990). Detect the affected compilers and pass -mllvm -tail-dup-pred-size=1000 when compiling ceval.c, and to the linker's LTO backend under --with-lto. With Clang 19 this made pyperformance 8.4-8.7% faster (PGO+LTO, and thin LTO without PGO). --- ...-09-27-18-00-00.gh-issue-158283.cLg19d.rst | 4 ++ configure | 49 +++++++++++++++++++ configure.ac | 38 ++++++++++++++ 3 files changed, 91 insertions(+) create mode 100644 Misc/NEWS.d/next/Build/2026-09-27-18-00-00.gh-issue-158283.cLg19d.rst diff --git a/Misc/NEWS.d/next/Build/2026-09-27-18-00-00.gh-issue-158283.cLg19d.rst b/Misc/NEWS.d/next/Build/2026-09-27-18-00-00.gh-issue-158283.cLg19d.rst new file mode 100644 index 000000000000000..758361267d36fd1 --- /dev/null +++ b/Misc/NEWS.d/next/Build/2026-09-27-18-00-00.gh-issue-158283.cLg19d.rst @@ -0,0 +1,4 @@ +When building with Clang 19, or with Apple clang from Xcode 16.3 to 26.3, force +tail duplication of the interpreter's computed-goto dispatch jumps. These +compilers merge most or all of the per-opcode dispatch jumps into one, which +made the interpreter about 9% slower on pyperformance with Clang 19. diff --git a/configure b/configure index e5d3f0091058acb..0437cbb1b189464 100755 --- a/configure +++ b/configure @@ -34445,6 +34445,55 @@ if test "$block_huge_inlining_in_ceval" = yes && test "$ac_cv_computed_gotos" = # interpreter. CFLAGS_CEVAL="$CFLAGS_CEVAL -finline-max-stacksize=512" fi +{ printf "%s\n" "$as_me:${as_lineno-$LINENO}: checking if tail duplication of the dispatch jumps needs to be forced" >&5 +printf %s "checking if tail duplication of the dispatch jumps needs to be forced... " >&6; } +cat confdefs.h - <<_ACEOF >conftest.$ac_ext +/* end confdefs.h. */ + +// gh-158283: LLVM 19 limits tail duplication of blocks ending in an +// indirect branch (llvm/llvm-project#78582), so the computed-goto +// interpreter is compiled with a single shared dispatch jump instead of one +// per instruction, which defeats per-opcode branch prediction (~9% slower +// on pyperformance). Fully fixed in LLVM 20.1.1 (llvm/llvm-project#114990). +// Apple clang 1700.0.x (Xcode 16.3-16.4) has the same bug, and 1700.3-1700.6 +// (Xcode 26.0-26.3) still merges most of the dispatch jumps. Older +// compilers are not affected and reject the option. +#if defined(__apple_build_version__) +# if __apple_build_version__ < 17000000 || __apple_build_version__ >= 18000000 +# error not affected +# endif +#elif !defined(__clang__) || __clang_major__ != 19 +# error not affected +#endif + +_ACEOF +if ac_fn_c_try_compile "$LINENO" +then : + force_dispatch_tail_dup=yes +else case e in #( + e) force_dispatch_tail_dup=no ;; +esac +fi +rm -f core conftest.err conftest.$ac_objext conftest.beam conftest.$ac_ext +{ printf "%s\n" "$as_me:${as_lineno-$LINENO}: result: $force_dispatch_tail_dup" >&5 +printf "%s\n" "$force_dispatch_tail_dup" >&6; } + +if test "$force_dispatch_tail_dup" = yes && test "$ac_cv_computed_gotos" = yes; then + CFLAGS_CEVAL="$CFLAGS_CEVAL -mllvm -tail-dup-pred-size=1000" + # With LTO, code generation happens in the linker's LTO plugin, and the + # clang driver does not forward -mllvm options there: pass the option to + # the linker directly (ld64 takes -mllvm; ld.lld and GNU ld/gold with + # LLVMgold take -plugin-opt). + if test "$Py_LTO" = 'true'; then + case $ac_sys_system in + Darwin*) + LDFLAGS_NODIST="$LDFLAGS_NODIST -Wl,-mllvm,-tail-dup-pred-size=1000" ;; + *) + LDFLAGS_NODIST="$LDFLAGS_NODIST -Wl,-plugin-opt=-tail-dup-pred-size=1000" ;; + esac + fi +fi + if test "$ac_cv_gcc_asm_for_x87" = yes; then diff --git a/configure.ac b/configure.ac index 82a623fedb8192a..bf0a9a2ae58956c 100644 --- a/configure.ac +++ b/configure.ac @@ -7813,6 +7813,44 @@ if test "$block_huge_inlining_in_ceval" = yes && test "$ac_cv_computed_gotos" = # interpreter. CFLAGS_CEVAL="$CFLAGS_CEVAL -finline-max-stacksize=512" fi +AC_MSG_CHECKING([if tail duplication of the dispatch jumps needs to be forced]) +AC_COMPILE_IFELSE([AC_LANG_SOURCE([[ +// gh-158283: LLVM 19 limits tail duplication of blocks ending in an +// indirect branch (llvm/llvm-project#78582), so the computed-goto +// interpreter is compiled with a single shared dispatch jump instead of one +// per instruction, which defeats per-opcode branch prediction (~9% slower +// on pyperformance). Fully fixed in LLVM 20.1.1 (llvm/llvm-project#114990). +// Apple clang 1700.0.x (Xcode 16.3-16.4) has the same bug, and 1700.3-1700.6 +// (Xcode 26.0-26.3) still merges most of the dispatch jumps. Older +// compilers are not affected and reject the option. +#if defined(__apple_build_version__) +# if __apple_build_version__ < 17000000 || __apple_build_version__ >= 18000000 +# error not affected +# endif +#elif !defined(__clang__) || __clang_major__ != 19 +# error not affected +#endif +]])], +[force_dispatch_tail_dup=yes], +[force_dispatch_tail_dup=no]) +AC_MSG_RESULT([$force_dispatch_tail_dup]) + +if test "$force_dispatch_tail_dup" = yes && test "$ac_cv_computed_gotos" = yes; then + CFLAGS_CEVAL="$CFLAGS_CEVAL -mllvm -tail-dup-pred-size=1000" + # With LTO, code generation happens in the linker's LTO plugin, and the + # clang driver does not forward -mllvm options there: pass the option to + # the linker directly (ld64 takes -mllvm; ld.lld and GNU ld/gold with + # LLVMgold take -plugin-opt). + if test "$Py_LTO" = 'true'; then + case $ac_sys_system in + Darwin*) + LDFLAGS_NODIST="$LDFLAGS_NODIST -Wl,-mllvm,-tail-dup-pred-size=1000" ;; + *) + LDFLAGS_NODIST="$LDFLAGS_NODIST -Wl,-plugin-opt=-tail-dup-pred-size=1000" ;; + esac + fi +fi + AC_SUBST([CFLAGS_CEVAL]) if test "$ac_cv_gcc_asm_for_x87" = yes; then