commit 98794d9bd4907cea20bd5b7bb2ca2ba8fdc57830 Author: Jakub Jelinek Date: Thu Jun 20 10:44:08 2024 +0200 Bump BASE-VER. 2024-06-20 Jakub Jelinek * BASE-VER: Set to 14.4.1. diff --git a/gcc/BASE-VER b/gcc/BASE-VER index f8c17e78090..c3b55577163 100644 --- a/gcc/BASE-VER +++ b/gcc/BASE-VER @@ -1 +1 @@ -12.4.0 +12.4.1 commit 8f612e6559b39569747894ec0f8b4694b96492a7 Author: Jakub Jelinek Date: Mon Jun 17 19:24:05 2024 +0200 c-family: Fix -Warray-compare warning ICE [PR115290] The warning code uses %D to print the ARRAY_REF first operands. That works in the most common case where those operands are decls, but as can be seen on the following testcase, they can be other expressions with array type. Just changing %D to %E isn't enough, because then the diagnostics can suggest something like note: use '&(x) != 0 ? (int (*)[32])&a : (int (*)[32])&b[0] == &(y) != 0 ? (int (*)[32])&a : (int (*)[32])&b[0]' to compare the addresses which is a bad suggestion, the %E printing doesn't know that the warning code will want to add & before it and [0] after it. So, the following patch adds ()s around the operand as well, but does that only for non-decls, for decls keeps it as &arr[0] like before. 2024-06-17 Jakub Jelinek PR c/115290 * c-warn.cc (do_warn_array_compare): Use %E rather than %D for printing op0 and op1; if those operands aren't decls, also print parens around them. * c-c++-common/Warray-compare-3.c: New test. (cherry picked from commit b63c7d92012f92e0517190cf263d29bbef8a06bf) diff --git a/gcc/c-family/c-warn.cc b/gcc/c-family/c-warn.cc index 9634351a4d3..7b45448df4e 100644 --- a/gcc/c-family/c-warn.cc +++ b/gcc/c-family/c-warn.cc @@ -3806,10 +3806,15 @@ do_warn_array_compare (location_t location, tree_code code, tree op0, tree op1) /* C doesn't allow +arr. */ if (c_dialect_cxx ()) inform (location, "use unary %<+%> which decays operands to pointers " - "or %<&%D[0] %s &%D[0]%> to compare the addresses", - op0, op_symbol_code (code), op1); + "or %<&%s%E%s[0] %s &%s%E%s[0]%> to compare the addresses", + DECL_P (op0) ? "" : "(", op0, DECL_P (op0) ? "" : ")", + op_symbol_code (code), + DECL_P (op1) ? "" : "(", op1, DECL_P (op1) ? "" : ")"); else - inform (location, "use %<&%D[0] %s &%D[0]%> to compare the addresses", - op0, op_symbol_code (code), op1); + inform (location, + "use %<&%s%E%s[0] %s &%s%E%s[0]%> to compare the addresses", + DECL_P (op0) ? "" : "(", op0, DECL_P (op0) ? "" : ")", + op_symbol_code (code), + DECL_P (op1) ? "" : "(", op1, DECL_P (op1) ? "" : ")"); } } diff --git a/gcc/testsuite/c-c++-common/Warray-compare-3.c b/gcc/testsuite/c-c++-common/Warray-compare-3.c new file mode 100644 index 00000000000..4725aa2b38b --- /dev/null +++ b/gcc/testsuite/c-c++-common/Warray-compare-3.c @@ -0,0 +1,13 @@ +/* PR c/115290 */ +/* { dg-do compile } */ +/* { dg-options "-Warray-compare" } */ + +int a[32][32], b[32][32]; + +int +foo (int x, int y) +{ + return (x ? a : b) == (y ? a : b); /* { dg-warning "comparison between two arrays" } */ +/* { dg-message "use '&\\\(\[^\n\r]*\\\)\\\[0\\\] == &\\\(\[^\n\r]*\\\)\\\[0\\\]' to compare the addresses" "" { target c } .-1 } */ +/* { dg-message "use unary '\\\+' which decays operands to pointers or '&\\\(\[^\n\r]*\\\)\\\[0\\\] == &\\\(\[^\n\r]*\\\)\\\[0\\\]' to compare the addresses" "" { target c++ } .-2 } */ +} commit c60dd0eb28eff3deaa389b0aafa689d423fc12f1 Author: Jakub Jelinek Date: Mon Jun 17 22:02:46 2024 +0200 diagnostics: Fix add_misspelling_candidates [PR115440] The option_map array for most entries contains just non-NULL opt0 { "-Wno-", NULL, "-W", false, true }, { "-fno-", NULL, "-f", false, true }, { "-gno-", NULL, "-g", false, true }, { "-mno-", NULL, "-m", false, true }, { "--debug=", NULL, "-g", false, false }, { "--machine-", NULL, "-m", true, false }, { "--machine-no-", NULL, "-m", false, true }, { "--machine=", NULL, "-m", false, false }, { "--machine=no-", NULL, "-m", false, true }, { "--machine", "", "-m", false, false }, { "--machine", "no-", "-m", false, true }, { "--optimize=", NULL, "-O", false, false }, { "--std=", NULL, "-std=", false, false }, { "--std", "", "-std=", false, false }, { "--warn-", NULL, "-W", true, false }, { "--warn-no-", NULL, "-W", false, true }, { "--", NULL, "-f", true, false }, { "--no-", NULL, "-f", false, true } and so add_misspelling_candidates works correctly for it, but 3 out of these, { "--machine", "", "-m", false, false }, { "--machine", "no-", "-m", false, true }, and { "--std", "", "-std=", false, false }, use non-NULL opt1. That says that --machine foo should map to -mfoo and --machine no-foo should map to -mno-foo and --std c++17 should map to -std=c++17 add_misspelling_canidates was not handling this, so it hapilly registered say --stdc++17 or --machineavx512 (twice) as spelling alternatives, when those options aren't recognized. Instead we support --std c++17 or --machine avx512 --machine no-avx512 The following patch fixes that. On this particular testcase, we no longer suggest anything, even when among the suggestion is say that --std c++17 or -std=c++17 etc. 2024-06-17 Jakub Jelinek PR driver/115440 * opts-common.cc (add_misspelling_candidates): If opt1 is non-NULL, add a space and opt1 to the alternative suggestion text. * g++.dg/cpp1z/pr115440.C: New test. (cherry picked from commit 96db57948b50f45235ae4af3b46db66cae7ea859) diff --git a/gcc/opts-common.cc b/gcc/opts-common.cc index 7c07d504696..99688009214 100644 --- a/gcc/opts-common.cc +++ b/gcc/opts-common.cc @@ -502,6 +502,7 @@ add_misspelling_candidates (auto_vec *candidates, for (unsigned i = 0; i < ARRAY_SIZE (option_map); i++) { const char *opt0 = option_map[i].opt0; + const char *opt1 = option_map[i].opt1; const char *new_prefix = option_map[i].new_prefix; size_t new_prefix_len = strlen (new_prefix); @@ -510,8 +511,9 @@ add_misspelling_candidates (auto_vec *candidates, if (strncmp (opt_text, new_prefix, new_prefix_len) == 0) { - char *alternative = concat (opt0 + 1, opt_text + new_prefix_len, - NULL); + char *alternative + = concat (opt0 + 1, opt1 ? " " : "", opt1 ? opt1 : "", + opt_text + new_prefix_len, NULL); candidates->safe_push (alternative); } } diff --git a/gcc/testsuite/g++.dg/cpp1z/pr115440.C b/gcc/testsuite/g++.dg/cpp1z/pr115440.C new file mode 100644 index 00000000000..788d4806fe2 --- /dev/null +++ b/gcc/testsuite/g++.dg/cpp1z/pr115440.C @@ -0,0 +1,8 @@ +// PR driver/115440 +// { dg-do compile { target c++17_only } } +// { dg-options "--c++17" } + +int i; + +// { dg-bogus "unrecognized command-line option '--c\\\+\\\+17'; did you mean '--stdc\\\+\\\+17'" "" { target *-*-* } 0 } +// { dg-error "unrecognized command-line option '--c\\\+\\\+17'" "" { target *-*-* } 0 } commit fb067547e401940b433cf0d2ae30749b4c21492e Author: Matthias Kretz Date: Mon May 6 12:13:55 2024 +0200 libstdc++: Use __builtin_shufflevector for simd split and concat Signed-off-by: Matthias Kretz libstdc++-v3/ChangeLog: PR libstdc++/114958 * include/experimental/bits/simd.h (__as_vector): Return scalar simd as one-element vector. Return vector from single-vector fixed_size simd. (__vec_shuffle): New. (__extract_part): Adjust return type signature. (split): Use __extract_part for any split into non-fixed_size simds. (concat): If the return type stores a single vector, use __vec_shuffle (which calls __builtin_shufflevector) to produce the return value. * include/experimental/bits/simd_builtin.h (__shift_elements_right): Removed. (__extract_part): Return single elements directly. Use __vec_shuffle (which calls __builtin_shufflevector) to for all non-trivial cases. * include/experimental/bits/simd_fixed_size.h (__extract_part): Return single elements directly. * testsuite/experimental/simd/pr114958.cc: New test. (cherry picked from commit fb1649f8b4ad5043dd0e65e4e3a643a0ced018a9) diff --git a/libstdc++-v3/include/experimental/bits/simd.h b/libstdc++-v3/include/experimental/bits/simd.h index 439545869be..60606298440 100644 --- a/libstdc++-v3/include/experimental/bits/simd.h +++ b/libstdc++-v3/include/experimental/bits/simd.h @@ -1616,7 +1616,24 @@ template if constexpr (__is_vector_type_v<_V>) return __x; else if constexpr (is_simd<_V>::value || is_simd_mask<_V>::value) - return __data(__x)._M_data; + { + if constexpr (__is_fixed_size_abi_v) + { + static_assert(is_simd<_V>::value); + static_assert(_V::abi_type::template __traits< + typename _V::value_type>::_SimdMember::_S_tuple_size == 1); + return __as_vector(__data(__x).first); + } + else if constexpr (_V::size() > 1) + return __data(__x)._M_data; + else + { + static_assert(is_simd<_V>::value); + using _Tp = typename _V::value_type; + using _RV [[__gnu__::__vector_size__(sizeof(_Tp))]] = _Tp; + return _RV{__data(__x)}; + } + } else if constexpr (__is_vectorizable_v<_V>) return __vector_type_t<_V, 2>{__x}; else @@ -2026,6 +2043,60 @@ template > return ~__a; } +// }}} +// __vec_shuffle{{{ +template + _GLIBCXX_SIMD_INTRINSIC constexpr auto + __vec_shuffle(_T0 __x, _T1 __y, index_sequence<_Is...> __seq, _Fun __idx_perm) + { + constexpr int _N0 = sizeof(__x) / sizeof(__x[0]); + constexpr int _N1 = sizeof(__y) / sizeof(__y[0]); +#if __has_builtin(__builtin_shufflevector) +#ifdef __clang__ + // Clang requires _T0 == _T1 + if constexpr (sizeof(__x) > sizeof(__y) and _N1 == 1) + return __vec_shuffle(__x, _T0{__y[0]}, __seq, __idx_perm); + else if constexpr (sizeof(__x) > sizeof(__y)) + return __vec_shuffle(__x, __intrin_bitcast<_T0>(__y), __seq, __idx_perm); + else if constexpr (sizeof(__x) < sizeof(__y) and _N0 == 1) + return __vec_shuffle(_T1{__x[0]}, __y, __seq, [=](int __i) { + __i = __idx_perm(__i); + return __i < _N0 ? __i : __i - _N0 + _N1; + }); + else if constexpr (sizeof(__x) < sizeof(__y)) + return __vec_shuffle(__intrin_bitcast<_T1>(__x), __y, __seq, [=](int __i) { + __i = __idx_perm(__i); + return __i < _N0 ? __i : __i - _N0 + _N1; + }); + else +#endif + return __builtin_shufflevector(__x, __y, [=] { + constexpr int __j = __idx_perm(_Is); + static_assert(__j < _N0 + _N1); + return __j; + }()...); +#else + using _Tp = __remove_cvref_t; + return __vector_type_t<_Tp, sizeof...(_Is)> { + [=]() -> _Tp { + constexpr int __j = __idx_perm(_Is); + static_assert(__j < _N0 + _N1); + if constexpr (__j < 0) + return 0; + else if constexpr (__j < _N0) + return __x[__j]; + else + return __y[__j - _N0]; + }()... + }; +#endif + } + +template + _GLIBCXX_SIMD_INTRINSIC constexpr auto + __vec_shuffle(_T0 __x, _Seq __seq, _Fun __idx_perm) + { return __vec_shuffle(__x, _T0(), __seq, __idx_perm); } + // }}} // __concat{{{ template , @@ -3884,7 +3955,7 @@ template _GLIBCXX_SIMD_INTRINSIC _GLIBCXX_CONST constexpr - _SimdWrapper<_Tp, _Np / _Total * _Combine> + conditional_t<_Np == _Total and _Combine == 1, _Tp, _SimdWrapper<_Tp, _Np / _Total * _Combine>> __extract_part(const _SimdWrapper<_Tp, _Np> __x); template @@ -4148,48 +4219,21 @@ template __split_wrapper(_SL::template _S_pop_front<1>(), __data(__x).second)); } - else if constexpr ((!is_same_v> && ...) - && (!__is_fixed_size_abi_v< - simd_abi::deduce_t<_Tp, _Sizes>> && ...)) + else if constexpr ((!__is_fixed_size_abi_v> && ...)) { - if constexpr (((_Sizes * 2 == _Np) && ...)) - return {{__private_init, __extract_part<0, 2>(__data(__x))}, - {__private_init, __extract_part<1, 2>(__data(__x))}}; - else if constexpr (is_same_v<_SizeList<_Sizes...>, - _SizeList<_Np / 3, _Np / 3, _Np / 3>>) - return {{__private_init, __extract_part<0, 3>(__data(__x))}, - {__private_init, __extract_part<1, 3>(__data(__x))}, - {__private_init, __extract_part<2, 3>(__data(__x))}}; - else if constexpr (is_same_v<_SizeList<_Sizes...>, - _SizeList<2 * _Np / 3, _Np / 3>>) - return {{__private_init, __extract_part<0, 3, 2>(__data(__x))}, - {__private_init, __extract_part<2, 3>(__data(__x))}}; - else if constexpr (is_same_v<_SizeList<_Sizes...>, - _SizeList<_Np / 3, 2 * _Np / 3>>) - return {{__private_init, __extract_part<0, 3>(__data(__x))}, - {__private_init, __extract_part<1, 3, 2>(__data(__x))}}; - else if constexpr (is_same_v<_SizeList<_Sizes...>, - _SizeList<_Np / 2, _Np / 4, _Np / 4>>) - return {{__private_init, __extract_part<0, 2>(__data(__x))}, - {__private_init, __extract_part<2, 4>(__data(__x))}, - {__private_init, __extract_part<3, 4>(__data(__x))}}; - else if constexpr (is_same_v<_SizeList<_Sizes...>, - _SizeList<_Np / 4, _Np / 4, _Np / 2>>) - return {{__private_init, __extract_part<0, 4>(__data(__x))}, - {__private_init, __extract_part<1, 4>(__data(__x))}, - {__private_init, __extract_part<1, 2>(__data(__x))}}; - else if constexpr (is_same_v<_SizeList<_Sizes...>, - _SizeList<_Np / 4, _Np / 2, _Np / 4>>) - return {{__private_init, __extract_part<0, 4>(__data(__x))}, - {__private_init, __extract_center(__data(__x))}, - {__private_init, __extract_part<3, 4>(__data(__x))}}; - else if constexpr (((_Sizes * 4 == _Np) && ...)) - return {{__private_init, __extract_part<0, 4>(__data(__x))}, - {__private_init, __extract_part<1, 4>(__data(__x))}, - {__private_init, __extract_part<2, 4>(__data(__x))}, - {__private_init, __extract_part<3, 4>(__data(__x))}}; - // else fall through + constexpr array __size = {_Sizes...}; + return __generate_from_n_evaluations( + [&](auto __i) constexpr { + constexpr size_t __offset = [&]() { + size_t __r = 0; + for (unsigned __j = 0; __j < __i; ++__j) + __r += __size[__j]; + return __r; + }(); + return __deduced_simd<_Tp, __size[__i]>( + __private_init, + __extract_part<__offset, _Np, __size[__i]>(__data(__x))); + }); } #ifdef _GLIBCXX_SIMD_USE_ALIASING_LOADS const __may_alias<_Tp>* const __element_ptr @@ -4251,14 +4295,37 @@ template simd<_Tp, simd_abi::deduce_t<_Tp, (simd_size_v<_Tp, _As> + ...)>> concat(const simd<_Tp, _As>&... __xs) { - using _Rp = __deduced_simd<_Tp, (simd_size_v<_Tp, _As> + ...)>; + constexpr int _Np = (simd_size_v<_Tp, _As> + ...); + using _Abi = simd_abi::deduce_t<_Tp, _Np>; + using _Rp = simd<_Tp, _Abi>; + using _RW = typename _SimdTraits<_Tp, _Abi>::_SimdMember; if constexpr (sizeof...(__xs) == 1) return simd_cast<_Rp>(__xs...); else if ((... && __xs._M_is_constprop())) - return simd<_Tp, - simd_abi::deduce_t<_Tp, (simd_size_v<_Tp, _As> + ...)>>( - [&](auto __i) constexpr _GLIBCXX_SIMD_ALWAYS_INLINE_LAMBDA + return _Rp([&](auto __i) constexpr _GLIBCXX_SIMD_ALWAYS_INLINE_LAMBDA { return __subscript_in_pack<__i>(__xs...); }); + else if constexpr (__is_simd_wrapper_v<_RW> and sizeof...(__xs) == 2) + { + return {__private_init, + __vec_shuffle(__as_vector(__xs)..., std::make_index_sequence<_RW::_S_full_size>(), + [](int __i) { + constexpr int __sizes[2] = {int(simd_size_v<_Tp, _As>)...}; + constexpr int __padding0 + = sizeof(__vector_type_t<_Tp, __sizes[0]>) / sizeof(_Tp) + - __sizes[0]; + return __i >= _Np ? -1 : __i < __sizes[0] ? __i : __i + __padding0; + })}; + } + else if constexpr (__is_simd_wrapper_v<_RW> and sizeof...(__xs) == 3) + return [](const auto& __x0, const auto& __x1, const auto& __x2) + _GLIBCXX_SIMD_ALWAYS_INLINE_LAMBDA { + return concat(concat(__x0, __x1), __x2); + }(__xs...); + else if constexpr (__is_simd_wrapper_v<_RW> and sizeof...(__xs) > 3) + return [](const auto& __x0, const auto& __x1, const auto&... __rest) + _GLIBCXX_SIMD_ALWAYS_INLINE_LAMBDA { + return concat(concat(__x0, __x1), concat(__rest...)); + }(__xs...); else { _Rp __r{}; diff --git a/libstdc++-v3/include/experimental/bits/simd_builtin.h b/libstdc++-v3/include/experimental/bits/simd_builtin.h index 57a5640643d..069d607bcd7 100644 --- a/libstdc++-v3/include/experimental/bits/simd_builtin.h +++ b/libstdc++-v3/include/experimental/bits/simd_builtin.h @@ -92,124 +92,16 @@ template >(__x._M_data); } -// }}} -// __shift_elements_right{{{ -// if (__shift % 2ⁿ == 0) => the low n Bytes are correct -template > - _GLIBCXX_SIMD_INTRINSIC _Tp - __shift_elements_right(_Tp __v) - { - [[maybe_unused]] const auto __iv = __to_intrin(__v); - static_assert(__shift <= sizeof(_Tp)); - if constexpr (__shift == 0) - return __v; - else if constexpr (__shift == sizeof(_Tp)) - return _Tp(); -#if _GLIBCXX_SIMD_X86INTRIN // {{{ - else if constexpr (__have_sse && __shift == 8 - && _TVT::template _S_is) - return _mm_movehl_ps(__iv, __iv); - else if constexpr (__have_sse2 && __shift == 8 - && _TVT::template _S_is) - return _mm_unpackhi_pd(__iv, __iv); - else if constexpr (__have_sse2 && sizeof(_Tp) == 16) - return reinterpret_cast( - _mm_srli_si128(reinterpret_cast<__m128i>(__iv), __shift)); - else if constexpr (__shift == 16 && sizeof(_Tp) == 32) - { - /*if constexpr (__have_avx && _TVT::template _S_is) - return _mm256_permute2f128_pd(__iv, __iv, 0x81); - else if constexpr (__have_avx && _TVT::template _S_is) - return _mm256_permute2f128_ps(__iv, __iv, 0x81); - else if constexpr (__have_avx) - return reinterpret_cast( - _mm256_permute2f128_si256(__iv, __iv, 0x81)); - else*/ - return __zero_extend(__hi128(__v)); - } - else if constexpr (__have_avx2 && sizeof(_Tp) == 32 && __shift < 16) - { - const auto __vll = __vector_bitcast<_LLong>(__v); - return reinterpret_cast( - _mm256_alignr_epi8(_mm256_permute2x128_si256(__vll, __vll, 0x81), - __vll, __shift)); - } - else if constexpr (__have_avx && sizeof(_Tp) == 32 && __shift < 16) - { - const auto __vll = __vector_bitcast<_LLong>(__v); - return reinterpret_cast( - __concat(_mm_alignr_epi8(__hi128(__vll), __lo128(__vll), __shift), - _mm_srli_si128(__hi128(__vll), __shift))); - } - else if constexpr (sizeof(_Tp) == 32 && __shift > 16) - return __zero_extend(__shift_elements_right<__shift - 16>(__hi128(__v))); - else if constexpr (sizeof(_Tp) == 64 && __shift == 32) - return __zero_extend(__hi256(__v)); - else if constexpr (__have_avx512f && sizeof(_Tp) == 64) - { - if constexpr (__shift >= 48) - return __zero_extend( - __shift_elements_right<__shift - 48>(__extract<3, 4>(__v))); - else if constexpr (__shift >= 32) - return __zero_extend( - __shift_elements_right<__shift - 32>(__hi256(__v))); - else if constexpr (__shift % 8 == 0) - return reinterpret_cast( - _mm512_alignr_epi64(__m512i(), __intrin_bitcast<__m512i>(__v), - __shift / 8)); - else if constexpr (__shift % 4 == 0) - return reinterpret_cast( - _mm512_alignr_epi32(__m512i(), __intrin_bitcast<__m512i>(__v), - __shift / 4)); - else if constexpr (__have_avx512bw && __shift < 16) - { - const auto __vll = __vector_bitcast<_LLong>(__v); - return reinterpret_cast( - _mm512_alignr_epi8(_mm512_shuffle_i32x4(__vll, __vll, 0xf9), - __vll, __shift)); - } - else if constexpr (__have_avx512bw && __shift < 32) - { - const auto __vll = __vector_bitcast<_LLong>(__v); - return reinterpret_cast( - _mm512_alignr_epi8(_mm512_shuffle_i32x4(__vll, __m512i(), 0xee), - _mm512_shuffle_i32x4(__vll, __vll, 0xf9), - __shift - 16)); - } - else - __assert_unreachable<_Tp>(); - } - /* - } else if constexpr (__shift % 16 == 0 && sizeof(_Tp) == 64) - return __auto_bitcast(__extract<__shift / 16, 4>(__v)); - */ -#endif // _GLIBCXX_SIMD_X86INTRIN }}} - else - { - constexpr int __chunksize = __shift % 8 == 0 ? 8 - : __shift % 4 == 0 ? 4 - : __shift % 2 == 0 ? 2 - : 1; - auto __w = __vector_bitcast<__int_with_sizeof_t<__chunksize>>(__v); - using _Up = decltype(__w); - return __intrin_bitcast<_Tp>( - __call_with_n_evaluations<(sizeof(_Tp) - __shift) / __chunksize>( - [](auto... __chunks) _GLIBCXX_SIMD_ALWAYS_INLINE_LAMBDA { - return _Up{__chunks...}; - }, [&](auto __i) _GLIBCXX_SIMD_ALWAYS_INLINE_LAMBDA { - return __w[__shift / __chunksize + __i]; - })); - } - } - // }}} // __extract_part(_SimdWrapper<_Tp, _Np>) {{{ template _GLIBCXX_SIMD_INTRINSIC _GLIBCXX_CONST constexpr - _SimdWrapper<_Tp, _Np / _Total * _Combine> + conditional_t<_Np == _Total and _Combine == 1, _Tp, _SimdWrapper<_Tp, _Np / _Total * _Combine>> __extract_part(const _SimdWrapper<_Tp, _Np> __x) { - if constexpr (_Index % 2 == 0 && _Total % 2 == 0 && _Combine % 2 == 0) + if constexpr (_Np == _Total and _Combine == 1) + return __x[_Index]; + else if constexpr (_Index % 2 == 0 && _Total % 2 == 0 && _Combine % 2 == 0) return __extract_part<_Index / 2, _Total / 2, _Combine / 2>(__x); else { @@ -235,39 +127,11 @@ template return __x; else if constexpr (_Index == 0) return __intrin_bitcast<_R>(__as_vector(__x)); -#if _GLIBCXX_SIMD_X86INTRIN // {{{ - else if constexpr (sizeof(__x) == 32 - && __return_size * sizeof(_Tp) <= 16) - { - constexpr size_t __bytes_to_skip = __values_to_skip * sizeof(_Tp); - if constexpr (__bytes_to_skip == 16) - return __vector_bitcast<_Tp, __return_size>( - __hi128(__as_vector(__x))); - else - return __vector_bitcast<_Tp, __return_size>( - _mm_alignr_epi8(__hi128(__vector_bitcast<_LLong>(__x)), - __lo128(__vector_bitcast<_LLong>(__x)), - __bytes_to_skip)); - } -#endif // _GLIBCXX_SIMD_X86INTRIN }}} - else if constexpr (_Index > 0 - && (__values_to_skip % __return_size != 0 - || sizeof(_R) >= 8) - && (__values_to_skip + __return_size) * sizeof(_Tp) - <= 64 - && sizeof(__x) >= 16) - return __intrin_bitcast<_R>( - __shift_elements_right<__values_to_skip * sizeof(_Tp)>( - __as_vector(__x))); else - { - _R __r = {}; - __builtin_memcpy(&__r, - reinterpret_cast(&__x) - + sizeof(_Tp) * __values_to_skip, - __return_size * sizeof(_Tp)); - return __r; - } + return __vec_shuffle(__as_vector(__x), make_index_sequence<__bit_ceil(__return_size)>(), + [](size_t __i) { + return __i + __values_to_skip; + }); } } diff --git a/libstdc++-v3/include/experimental/bits/simd_fixed_size.h b/libstdc++-v3/include/experimental/bits/simd_fixed_size.h index 9df82802ae7..2300397557c 100644 --- a/libstdc++-v3/include/experimental/bits/simd_fixed_size.h +++ b/libstdc++-v3/include/experimental/bits/simd_fixed_size.h @@ -927,7 +927,9 @@ template ; // handle (optimize) the simple cases - if constexpr (_Index == 0 && _Tuple::_S_first_size == __return_size) + if constexpr (__return_size == 1) + return __x[integral_constant()]; + else if constexpr (_Index == 0 && _Tuple::_S_first_size == __return_size) return __x.first._M_data; else if constexpr (_Index == 0 && _Total == _Combine) return __x; diff --git a/libstdc++-v3/testsuite/experimental/simd/pr114958.cc b/libstdc++-v3/testsuite/experimental/simd/pr114958.cc new file mode 100644 index 00000000000..f7b09ad9ba7 --- /dev/null +++ b/libstdc++-v3/testsuite/experimental/simd/pr114958.cc @@ -0,0 +1,20 @@ +// { dg-options "-std=c++17" } +// { dg-do compile { target x86_64-*-* } } +// { dg-require-effective-target c++17 } +// { dg-additional-options "-march=x86-64-v3" { target x86_64-*-* } } +// { dg-require-cmath "" } +// { dg-final { scan-assembler-times "vperm(?:q|pd)\\s+\\\$144" 1 } } + +#include + +namespace stdx = std::experimental; + +using T = std::uint64_t; +using V = stdx::simd>; +using V1 = stdx::simd; + +V perm(V data) +{ + auto [carry, _] = stdx::split<3, 1>(data); + return concat(V1(), carry); +} commit f79b273d4145961133ef8b0344469e77425629f6 Author: Matthias Kretz Date: Wed May 15 11:02:22 2024 +0200 libstdc++: Avoid MMX return types from __builtin_shufflevector This resolves a regression on i686 that was introduced with r15-429-gfb1649f8b4ad50. Signed-off-by: Matthias Kretz libstdc++-v3/ChangeLog: PR libstdc++/115247 * include/experimental/bits/simd.h (__as_vector): Don't use vector_size(8) on __i386__. (__vec_shuffle): Never return MMX vectors, widen to 16 bytes instead. (concat): Fix padding calculation to pick up widening logic from __as_vector. (cherry picked from commit 241a6cc88d866fb36bd35ddb3edb659453d6322e) diff --git a/libstdc++-v3/include/experimental/bits/simd.h b/libstdc++-v3/include/experimental/bits/simd.h index 60606298440..2cc280762cd 100644 --- a/libstdc++-v3/include/experimental/bits/simd.h +++ b/libstdc++-v3/include/experimental/bits/simd.h @@ -1630,7 +1630,12 @@ template { static_assert(is_simd<_V>::value); using _Tp = typename _V::value_type; +#ifdef __i386__ + constexpr auto __bytes = sizeof(_Tp) == 8 ? 16 : sizeof(_Tp); + using _RV [[__gnu__::__vector_size__(__bytes)]] = _Tp; +#else using _RV [[__gnu__::__vector_size__(sizeof(_Tp))]] = _Tp; +#endif return _RV{__data(__x)}; } } @@ -2046,11 +2051,14 @@ template > // }}} // __vec_shuffle{{{ template - _GLIBCXX_SIMD_INTRINSIC constexpr auto + _GLIBCXX_SIMD_INTRINSIC constexpr + __vector_type_t()[0])>, sizeof...(_Is)> __vec_shuffle(_T0 __x, _T1 __y, index_sequence<_Is...> __seq, _Fun __idx_perm) { constexpr int _N0 = sizeof(__x) / sizeof(__x[0]); constexpr int _N1 = sizeof(__y) / sizeof(__y[0]); + using _Tp = remove_reference_t()[0])>; + using _RV [[maybe_unused]] = __vector_type_t<_Tp, sizeof...(_Is)>; #if __has_builtin(__builtin_shufflevector) #ifdef __clang__ // Clang requires _T0 == _T1 @@ -2070,14 +2078,23 @@ template }); else #endif - return __builtin_shufflevector(__x, __y, [=] { - constexpr int __j = __idx_perm(_Is); - static_assert(__j < _N0 + _N1); - return __j; - }()...); + { + const auto __r = __builtin_shufflevector(__x, __y, [=] { + constexpr int __j = __idx_perm(_Is); + static_assert(__j < _N0 + _N1); + return __j; + }()...); +#ifdef __i386__ + if constexpr (sizeof(__r) == sizeof(_RV)) + return __r; + else + return _RV {__r[_Is]...}; +#else + return __r; +#endif + } #else - using _Tp = __remove_cvref_t; - return __vector_type_t<_Tp, sizeof...(_Is)> { + return _RV { [=]() -> _Tp { constexpr int __j = __idx_perm(_Is); static_assert(__j < _N0 + _N1); @@ -4310,9 +4327,9 @@ template __vec_shuffle(__as_vector(__xs)..., std::make_index_sequence<_RW::_S_full_size>(), [](int __i) { constexpr int __sizes[2] = {int(simd_size_v<_Tp, _As>)...}; - constexpr int __padding0 - = sizeof(__vector_type_t<_Tp, __sizes[0]>) / sizeof(_Tp) - - __sizes[0]; + constexpr int __vsizes[2] + = {int(sizeof(__as_vector(__xs)) / sizeof(_Tp))...}; + constexpr int __padding0 = __vsizes[0] - __sizes[0]; return __i >= _Np ? -1 : __i < __sizes[0] ? __i : __i + __padding0; })}; } commit b9569e7a829d054336d2704ccff61eece5437baf Author: Matthias Kretz Date: Mon Jun 3 12:02:07 2024 +0200 libstdc++: Fix simd conversion for -fno-signed-char for Clang The special case for Clang in the trait producing a signed integer type lead to the trait returning 'char' where it should have been 'signed char'. This workaround was introduced because on Clang the return type of vector compares was not convertible to '_SimdWrapper< __int_for_sizeof_t<...' unless '__int_for_sizeof_t' was an alias for 'char'. In order to not rewrite the complete mask type code (there is code scattered around the implementation assuming signed integers), this needs to be 'signed char'; so the special case for Clang needs to be removed. The conversion issue is now solved in _SimdWrapper, which now additionally allows conversion from vector types with compatible integral type. Signed-off-by: Matthias Kretz libstdc++-v3/ChangeLog: PR libstdc++/115308 * include/experimental/bits/simd.h (__int_for_sizeof): Remove special cases for __clang__. (_SimdWrapper): Change constructor overload set to allow conversion from vector types with integral conversions via bit reinterpretation. (cherry picked from commit 8e36cf4c5c9140915d0019999db132a900b48037) diff --git a/libstdc++-v3/include/experimental/bits/simd.h b/libstdc++-v3/include/experimental/bits/simd.h index 2cc280762cd..365f5158f0d 100644 --- a/libstdc++-v3/include/experimental/bits/simd.h +++ b/libstdc++-v3/include/experimental/bits/simd.h @@ -584,19 +584,12 @@ template static_assert(_Bytes > 0); if constexpr (_Bytes == sizeof(int)) return int(); - #ifdef __clang__ - else if constexpr (_Bytes == sizeof(char)) - return char(); - #else else if constexpr (_Bytes == sizeof(_SChar)) return _SChar(); - #endif else if constexpr (_Bytes == sizeof(short)) return short(); - #ifndef __clang__ else if constexpr (_Bytes == sizeof(long)) return long(); - #endif else if constexpr (_Bytes == sizeof(_LLong)) return _LLong(); #ifdef __SIZEOF_INT128__ @@ -2712,6 +2705,8 @@ template // }}} // _SimdWrapper{{{ +struct _DisabledSimdWrapper; + template struct _SimdWrapper< _Tp, _Width, @@ -2721,16 +2716,17 @@ template == sizeof(__vector_type_t<_Tp, _Width>), __vector_type_t<_Tp, _Width>> { - using _Base - = _SimdWrapperBase<__has_iec559_behavior<__signaling_NaN, _Tp>::value - && sizeof(_Tp) * _Width - == sizeof(__vector_type_t<_Tp, _Width>), - __vector_type_t<_Tp, _Width>>; + static constexpr bool _S_need_default_init + = __has_iec559_behavior<__signaling_NaN, _Tp>::value + and sizeof(_Tp) * _Width == sizeof(__vector_type_t<_Tp, _Width>); + + using _BuiltinType = __vector_type_t<_Tp, _Width>; + + using _Base = _SimdWrapperBase<_S_need_default_init, _BuiltinType>; static_assert(__is_vectorizable_v<_Tp>); static_assert(_Width >= 2); // 1 doesn't make sense, use _Tp directly then - using _BuiltinType = __vector_type_t<_Tp, _Width>; using value_type = _Tp; static inline constexpr size_t _S_full_size @@ -2766,13 +2762,26 @@ template _GLIBCXX_SIMD_INTRINSIC constexpr _SimdWrapper& operator=(_SimdWrapper&&) = default; - template >, - is_same<_V, __intrinsic_type_t<_Tp, _Width>>>>> + // Convert from exactly matching __vector_type_t + using _SimdWrapperBase<_S_need_default_init, _BuiltinType>::_SimdWrapperBase; + + // Convert from __intrinsic_type_t if __intrinsic_type_t and __vector_type_t differ, otherwise + // this ctor should not exist. Making the argument type unusable is our next best solution. + _GLIBCXX_SIMD_INTRINSIC constexpr + _SimdWrapper(conditional_t>, + _DisabledSimdWrapper, __intrinsic_type_t<_Tp, _Width>> __x) + : _Base(__vector_bitcast<_Tp, _Width>(__x)) {} + + // Convert from different __vector_type_t, but only if bit reinterpretation is a correct + // conversion of the value_type + template , + typename = enable_if_t + and is_integral_v>> _GLIBCXX_SIMD_INTRINSIC constexpr _SimdWrapper(_V __x) - // __vector_bitcast can convert e.g. __m128 to __vector(2) float - : _Base(__vector_bitcast<_Tp, _Width>(__x)) {} + : _Base(reinterpret_cast<_BuiltinType>(__x)) {} template && ...) commit cdbff5f14669129384ea810b1fb2b7c0207b3742 Author: GCC Administrator Date: Fri Jun 21 00:21:02 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 07e553d0b3f..4f9e7b01c5e 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2024-06-20 Jakub Jelinek + + Backported from master: + 2024-06-17 Jakub Jelinek + + PR driver/115440 + * opts-common.cc (add_misspelling_candidates): If opt1 is non-NULL, + add a space and opt1 to the alternative suggestion text. + 2024-06-20 Release Manager * GCC 12.4.0 released. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 9df1831b6e3..e778c427d11 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240620 +20240621 diff --git a/gcc/c-family/ChangeLog b/gcc/c-family/ChangeLog index 748902653af..a6539dbd40b 100644 --- a/gcc/c-family/ChangeLog +++ b/gcc/c-family/ChangeLog @@ -1,3 +1,13 @@ +2024-06-20 Jakub Jelinek + + Backported from master: + 2024-06-17 Jakub Jelinek + + PR c/115290 + * c-warn.cc (do_warn_array_compare): Use %E rather than %D for + printing op0 and op1; if those operands aren't decls, also print + parens around them. + 2024-06-20 Release Manager * GCC 12.4.0 released. diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 876c7186069..87c83d10865 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,19 @@ +2024-06-20 Jakub Jelinek + + Backported from master: + 2024-06-17 Jakub Jelinek + + PR driver/115440 + * g++.dg/cpp1z/pr115440.C: New test. + +2024-06-20 Jakub Jelinek + + Backported from master: + 2024-06-17 Jakub Jelinek + + PR c/115290 + * c-c++-common/Warray-compare-3.c: New test. + 2024-06-20 Release Manager * GCC 12.4.0 released. diff --git a/libstdc++-v3/ChangeLog b/libstdc++-v3/ChangeLog index 201f46804cc..cc6f6a88902 100644 --- a/libstdc++-v3/ChangeLog +++ b/libstdc++-v3/ChangeLog @@ -1,3 +1,53 @@ +2024-06-20 Matthias Kretz + + Backported from master: + 2024-06-04 Matthias Kretz + + PR libstdc++/115308 + * include/experimental/bits/simd.h (__int_for_sizeof): Remove + special cases for __clang__. + (_SimdWrapper): Change constructor overload set to allow + conversion from vector types with integral conversions via bit + reinterpretation. + +2024-06-20 Matthias Kretz + + Backported from master: + 2024-05-29 Matthias Kretz + + PR libstdc++/115247 + * include/experimental/bits/simd.h (__as_vector): Don't use + vector_size(8) on __i386__. + (__vec_shuffle): Never return MMX vectors, widen to 16 bytes + instead. + (concat): Fix padding calculation to pick up widening logic from + __as_vector. + +2024-06-20 Matthias Kretz + + Backported from master: + 2024-05-13 Matthias Kretz + + PR libstdc++/114958 + * include/experimental/bits/simd.h (__as_vector): Return scalar + simd as one-element vector. Return vector from single-vector + fixed_size simd. + (__vec_shuffle): New. + (__extract_part): Adjust return type signature. + (split): Use __extract_part for any split into non-fixed_size + simds. + (concat): If the return type stores a single vector, use + __vec_shuffle (which calls __builtin_shufflevector) to produce + the return value. + * include/experimental/bits/simd_builtin.h + (__shift_elements_right): Removed. + (__extract_part): Return single elements directly. Use + __vec_shuffle (which calls __builtin_shufflevector) to for all + non-trivial cases. + * include/experimental/bits/simd_fixed_size.h (__extract_part): + Return single elements directly. + * testsuite/experimental/simd/pr114958.cc: New test. + 2024-06-20 Release Manager * GCC 12.4.0 released. commit 8b5bdeb8aa2c2f6dbd448a8f7d500d9eaece48e1 Author: Matthias Kretz Date: Fri Jun 14 15:11:25 2024 +0200 libstdc++: Fix find_last_set(simd_mask) to ignore padding bits With the change to the AVX512 find_last_set implementation, the change to AVX512 operator!= is unnecessary. However, the latter was not producing optimal code and unnecessarily set the padding bits. In theory, the compiler could determine that with the new != implementation, the bit operation for clearing the padding bits is a no-op and can be elided. Signed-off-by: Matthias Kretz libstdc++-v3/ChangeLog: PR libstdc++/115454 * include/experimental/bits/simd_x86.h (_S_not_equal_to): Use neq comparison instead of bitwise negation after eq. (_S_find_last_set): Clear unused high bits before computing bit_width. * testsuite/experimental/simd/pr115454_find_last_set.cc: New test. (cherry picked from commit 1340ddea0158de3f49aeb75b4013e5fc313ff6f4) diff --git a/libstdc++-v3/include/experimental/bits/simd_x86.h b/libstdc++-v3/include/experimental/bits/simd_x86.h index 7cda7f7d0e0..a88cc535f50 100644 --- a/libstdc++-v3/include/experimental/bits/simd_x86.h +++ b/libstdc++-v3/include/experimental/bits/simd_x86.h @@ -2339,29 +2339,29 @@ template __assert_unreachable<_Tp>(); } else if constexpr (sizeof(__xi) == 64 && sizeof(_Tp) == 8) - return ~_mm512_mask_cmpeq_epi64_mask(__k1, __xi, __yi); + return _mm512_mask_cmpneq_epi64_mask(__k1, __xi, __yi); else if constexpr (sizeof(__xi) == 64 && sizeof(_Tp) == 4) - return ~_mm512_mask_cmpeq_epi32_mask(__k1, __xi, __yi); + return _mm512_mask_cmpneq_epi32_mask(__k1, __xi, __yi); else if constexpr (sizeof(__xi) == 64 && sizeof(_Tp) == 2) - return ~_mm512_mask_cmpeq_epi16_mask(__k1, __xi, __yi); + return _mm512_mask_cmpneq_epi16_mask(__k1, __xi, __yi); else if constexpr (sizeof(__xi) == 64 && sizeof(_Tp) == 1) - return ~_mm512_mask_cmpeq_epi8_mask(__k1, __xi, __yi); + return _mm512_mask_cmpneq_epi8_mask(__k1, __xi, __yi); else if constexpr (sizeof(__xi) == 32 && sizeof(_Tp) == 8) - return ~_mm256_mask_cmpeq_epi64_mask(__k1, __xi, __yi); + return _mm256_mask_cmpneq_epi64_mask(__k1, __xi, __yi); else if constexpr (sizeof(__xi) == 32 && sizeof(_Tp) == 4) - return ~_mm256_mask_cmpeq_epi32_mask(__k1, __xi, __yi); + return _mm256_mask_cmpneq_epi32_mask(__k1, __xi, __yi); else if constexpr (sizeof(__xi) == 32 && sizeof(_Tp) == 2) - return ~_mm256_mask_cmpeq_epi16_mask(__k1, __xi, __yi); + return _mm256_mask_cmpneq_epi16_mask(__k1, __xi, __yi); else if constexpr (sizeof(__xi) == 32 && sizeof(_Tp) == 1) - return ~_mm256_mask_cmpeq_epi8_mask(__k1, __xi, __yi); + return _mm256_mask_cmpneq_epi8_mask(__k1, __xi, __yi); else if constexpr (sizeof(__xi) == 16 && sizeof(_Tp) == 8) - return ~_mm_mask_cmpeq_epi64_mask(__k1, __xi, __yi); + return _mm_mask_cmpneq_epi64_mask(__k1, __xi, __yi); else if constexpr (sizeof(__xi) == 16 && sizeof(_Tp) == 4) - return ~_mm_mask_cmpeq_epi32_mask(__k1, __xi, __yi); + return _mm_mask_cmpneq_epi32_mask(__k1, __xi, __yi); else if constexpr (sizeof(__xi) == 16 && sizeof(_Tp) == 2) - return ~_mm_mask_cmpeq_epi16_mask(__k1, __xi, __yi); + return _mm_mask_cmpneq_epi16_mask(__k1, __xi, __yi); else if constexpr (sizeof(__xi) == 16 && sizeof(_Tp) == 1) - return ~_mm_mask_cmpeq_epi8_mask(__k1, __xi, __yi); + return _mm_mask_cmpneq_epi8_mask(__k1, __xi, __yi); else __assert_unreachable<_Tp>(); } // }}} @@ -5292,7 +5292,7 @@ template _S_find_last_set(simd_mask<_Tp, _Abi> __k) { if constexpr (__is_avx512_abi<_Abi>()) - return std::__bit_width(__k._M_data._M_data) - 1; + return std::__bit_width(_Abi::_S_masked(__k._M_data)._M_data) - 1; else return _Base::_S_find_last_set(__k); } diff --git a/libstdc++-v3/testsuite/experimental/simd/pr115454_find_last_set.cc b/libstdc++-v3/testsuite/experimental/simd/pr115454_find_last_set.cc new file mode 100644 index 00000000000..b47f19d3067 --- /dev/null +++ b/libstdc++-v3/testsuite/experimental/simd/pr115454_find_last_set.cc @@ -0,0 +1,49 @@ +// { dg-options "-std=gnu++17" } +// { dg-do run { target *-*-* } } +// { dg-require-effective-target c++17 } +// { dg-additional-options "-march=x86-64-v4" { target avx512f } } +// { dg-require-cmath "" } + +#include + +namespace stdx = std::experimental; + +using T = std::uint64_t; + +template +using V = stdx::simd>; + +[[gnu::noinline, gnu::noipa]] +int reduce(V x) +{ + static_assert(stdx::find_last_set(V([](unsigned i) { return i; }) != V(0)) == 3); + return stdx::find_last_set(x != -1); +} + +[[gnu::noinline, gnu::noipa]] +int reduce2() +{ + using M8 = typename V::mask_type; + using M4 = typename V::mask_type; + if constexpr (sizeof(M8) == sizeof(M4)) + { + M4 k; + __builtin_memcpy(&__data(k), &__data(M8(true)), sizeof(M4)); + return stdx::find_last_set(k); + } + return 3; +} + + +int main() +{ + const V x {}; + + const int r = reduce(x); + if (r != 3) + __builtin_abort(); + + const int r2 = reduce2(); + if (r2 != 3) + __builtin_abort(); +} commit 169d4d1addaac7eef6cde4049aa8b4f3d81c28b0 Author: Matthias Kretz Date: Fri Jun 21 16:22:22 2024 +0200 libstdc++: Fix test on x86_64 and non-simd targets * Running a test compiled with AVX512 instructions requires avx512f_runtime not just avx512f. * The 'reduce2' test violated an invariant of fixed_size_simd_mask and thus failed on all targets without 16-Byte vector builtins enabled (in bits/simd.h). Signed-off-by: Matthias Kretz libstdc++-v3/ChangeLog: PR libstdc++/115575 * testsuite/experimental/simd/pr115454_find_last_set.cc: Require avx512f_runtime. Don't memcpy fixed_size masks. (cherry picked from commit 77f321435b4ac37992c2ed6737ca0caa1dd50551) diff --git a/libstdc++-v3/testsuite/experimental/simd/pr115454_find_last_set.cc b/libstdc++-v3/testsuite/experimental/simd/pr115454_find_last_set.cc index b47f19d3067..25a713b4e94 100644 --- a/libstdc++-v3/testsuite/experimental/simd/pr115454_find_last_set.cc +++ b/libstdc++-v3/testsuite/experimental/simd/pr115454_find_last_set.cc @@ -1,7 +1,7 @@ // { dg-options "-std=gnu++17" } // { dg-do run { target *-*-* } } // { dg-require-effective-target c++17 } -// { dg-additional-options "-march=x86-64-v4" { target avx512f } } +// { dg-additional-options "-march=x86-64-v4" { target avx512f_runtime } } // { dg-require-cmath "" } #include @@ -25,7 +25,9 @@ int reduce2() { using M8 = typename V::mask_type; using M4 = typename V::mask_type; - if constexpr (sizeof(M8) == sizeof(M4)) + if constexpr (sizeof(M8) == sizeof(M4) + && !std::is_same_v>) + // fixed_size invariant: padding bits of masks are zero, the memcpy would violate that { M4 k; __builtin_memcpy(&__data(k), &__data(M8(true)), sizeof(M4)); commit 218adac0fce6135fcb5c0c56911272687f05872b Author: GCC Administrator Date: Sat Jun 22 00:21:36 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e778c427d11..d4f60539db5 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240621 +20240622 diff --git a/libstdc++-v3/ChangeLog b/libstdc++-v3/ChangeLog index cc6f6a88902..954608173b1 100644 --- a/libstdc++-v3/ChangeLog +++ b/libstdc++-v3/ChangeLog @@ -1,3 +1,25 @@ +2024-06-21 Matthias Kretz + + Backported from master: + 2024-06-21 Matthias Kretz + + PR libstdc++/115575 + * testsuite/experimental/simd/pr115454_find_last_set.cc: Require + avx512f_runtime. Don't memcpy fixed_size masks. + +2024-06-21 Matthias Kretz + + Backported from master: + 2024-06-20 Matthias Kretz + + PR libstdc++/115454 + * include/experimental/bits/simd_x86.h (_S_not_equal_to): Use + neq comparison instead of bitwise negation after eq. + (_S_find_last_set): Clear unused high bits before computing + bit_width. + * testsuite/experimental/simd/pr115454_find_last_set.cc: New + test. + 2024-06-20 Matthias Kretz Backported from master: commit 723716cba0ab284e075abce55d0e5bd69a6dd9c9 Author: GCC Administrator Date: Sun Jun 23 00:20:18 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d4f60539db5..7b41634677c 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240622 +20240623 commit f4affb90730abbe83eaca527f2752eb8490e7eb1 Author: GCC Administrator Date: Mon Jun 24 00:21:23 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7b41634677c..149499c5447 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240623 +20240624 commit 0fd6ae9b20913ab84d596448e14411eedbd324f9 Author: Kewen Lin Date: Tue May 28 21:13:40 2024 -0500 rs6000: Don't clobber return value when eh_return called [PR114846] As the associated test case in PR114846 shows, currently with eh_return involved some register restoring for EH RETURN DATA in epilogue can clobber the one which holding the return value. Referring to the existing handlings in some other targets, this patch makes eh_return expander call one new define_insn_and_split eh_return_internal which directly calls rs6000_emit_epilogue with epilogue_type EPILOGUE_TYPE_EH_RETURN instead of the previous treating normal return with crtl->calls_eh_return specially. PR target/114846 gcc/ChangeLog: * config/rs6000/rs6000-logue.cc (rs6000_emit_epilogue): As EPILOGUE_TYPE_EH_RETURN would be passed as epilogue_type directly now, adjust the relevant handlings on it. * config/rs6000/rs6000.md (eh_return expander): Append by calling gen_eh_return_internal and emit_barrier. (eh_return_internal): New define_insn_and_split, call function rs6000_emit_epilogue with epilogue type EPILOGUE_TYPE_EH_RETURN. gcc/testsuite/ChangeLog: * gcc.target/powerpc/pr114846.c: New test. (cherry picked from commit e5fc5d42d25c86ae48178db04ce64d340a834614) diff --git a/gcc/config/rs6000/rs6000-logue.cc b/gcc/config/rs6000/rs6000-logue.cc index a868ede24fb..33077b72611 100644 --- a/gcc/config/rs6000/rs6000-logue.cc +++ b/gcc/config/rs6000/rs6000-logue.cc @@ -4283,9 +4283,6 @@ rs6000_emit_epilogue (enum epilogue_type epilogue_type) rs6000_stack_t *info = rs6000_stack_info (); - if (epilogue_type == EPILOGUE_TYPE_NORMAL && crtl->calls_eh_return) - epilogue_type = EPILOGUE_TYPE_EH_RETURN; - int strategy = info->savres_strategy; bool using_load_multiple = !!(strategy & REST_MULTIPLE); bool restoring_GPRs_inline = !!(strategy & REST_INLINE_GPRS); @@ -4763,7 +4760,9 @@ rs6000_emit_epilogue (enum epilogue_type epilogue_type) /* In the ELFv2 ABI we need to restore all call-saved CR fields from *separate* slots if the routine calls __builtin_eh_return, so - that they can be independently restored by the unwinder. */ + that they can be independently restored by the unwinder. Since + it is for CR fields restoring, it should be done for any epilogue + types (not EPILOGUE_TYPE_EH_RETURN specific). */ if (DEFAULT_ABI == ABI_ELFv2 && crtl->calls_eh_return) { int i, cr_off = info->ehcr_offset; diff --git a/gcc/config/rs6000/rs6000.md b/gcc/config/rs6000/rs6000.md index 2f69c89c689..c38bebde185 100644 --- a/gcc/config/rs6000/rs6000.md +++ b/gcc/config/rs6000/rs6000.md @@ -14039,6 +14039,8 @@ "" { emit_insn (gen_eh_set_lr (Pmode, operands[0])); + emit_jump_insn (gen_eh_return_internal ()); + emit_barrier (); DONE; }) @@ -14055,6 +14057,19 @@ DONE; }) +(define_insn_and_split "eh_return_internal" + [(eh_return)] + "" + "#" + "epilogue_completed" + [(const_int 0)] +{ + if (!TARGET_SCHED_PROLOG) + emit_insn (gen_blockage ()); + rs6000_emit_epilogue (EPILOGUE_TYPE_EH_RETURN); + DONE; +}) + (define_insn "prefetch" [(prefetch (match_operand 0 "indexed_or_indirect_address" "a") (match_operand:SI 1 "const_int_operand" "n") diff --git a/gcc/testsuite/gcc.target/powerpc/pr114846.c b/gcc/testsuite/gcc.target/powerpc/pr114846.c new file mode 100644 index 00000000000..efe2300b73a --- /dev/null +++ b/gcc/testsuite/gcc.target/powerpc/pr114846.c @@ -0,0 +1,20 @@ +/* { dg-do run } */ +/* { dg-require-effective-target builtin_eh_return } */ + +/* Ensure it runs successfully. */ + +__attribute__ ((noipa)) +int f (int *a, long offset, void *handler) +{ + if (*a == 5) + return 5; + __builtin_eh_return (offset, handler); +} + +int main () +{ + int t = 5; + if (f (&t, 0, 0) != 5) + __builtin_abort (); + return 0; +} commit 814b8cce9f3c83b6764bcb0b32cbfff7ac7ce88f Author: GCC Administrator Date: Tue Jun 25 00:20:57 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 4f9e7b01c5e..de45aac0538 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,17 @@ +2024-06-24 Kewen Lin + + Backported from master: + 2024-05-29 Kewen Lin + + PR target/114846 + * config/rs6000/rs6000-logue.cc (rs6000_emit_epilogue): As + EPILOGUE_TYPE_EH_RETURN would be passed as epilogue_type directly + now, adjust the relevant handlings on it. + * config/rs6000/rs6000.md (eh_return expander): Append by calling + gen_eh_return_internal and emit_barrier. + (eh_return_internal): New define_insn_and_split, call function + rs6000_emit_epilogue with epilogue type EPILOGUE_TYPE_EH_RETURN. + 2024-06-20 Jakub Jelinek Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 149499c5447..271b3eb540d 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240624 +20240625 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 87c83d10865..def1a220cd2 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-06-24 Kewen Lin + + Backported from master: + 2024-05-29 Kewen Lin + + PR target/114846 + * gcc.target/powerpc/pr114846.c: New test. + 2024-06-20 Jakub Jelinek Backported from master: commit 4b69afd72ea8aaa8921c8049049f412614e1327d Author: Jonathan Wakely Date: Tue Jun 25 23:25:54 2024 +0100 libstdc++: Remove confusing text from status tables for release branch When I tried to make the release branch versions of these docs refer to the release branch instead of "mainline GCC", for some reason I left the text "not any particular release" there. That's just confusing, because the docs are for a particular release, the latest on that branch. Remove that confusing text in several places. libstdc++-v3/ChangeLog: * doc/xml/manual/status_cxx2023.xml: Change reference from mainline GCC to the release branch. * doc/xml/manual/status_cxx1998.xml: Remove confusing "not in any particular release" text. * doc/xml/manual/status_cxx2011.xml: Likewise. * doc/xml/manual/status_cxx2014.xml: Likewise. * doc/xml/manual/status_cxx2017.xml: Likewise. * doc/xml/manual/status_cxx2020.xml: Likewise. * doc/xml/manual/status_cxxtr1.xml: Likewise. * doc/xml/manual/status_cxxtr24733.xml: Likewise. * doc/html/manual/status.html: Regenerate. diff --git a/libstdc++-v3/doc/html/manual/status.html b/libstdc++-v3/doc/html/manual/status.html index a6494912c07..c212ea64fe6 100644 --- a/libstdc++-v3/doc/html/manual/status.html +++ b/libstdc++-v3/doc/html/manual/status.html @@ -5,8 +5,7 @@  Next

Chapter 1. Status

Implementation Status

C++ 1998/2003

Implementation Status

This status table is based on the table of contents of ISO/IEC 14882:2003.

-This section describes the C++ support in the GCC 12 release series, -not in any particular release. +This section describes the C++ support in the GCC 12 release series.

Table 1.1. C++ 1998/2003 Implementation Status

SectionDescriptionStatusComments
18 @@ -160,8 +159,7 @@ since that release.

This status table is based on the table of contents of ISO/IEC 14882:2011.

-This section describes the C++11 support in the GCC 12 release series, -not in any particular release. +This section describes the C++11 support in the GCC 12 release series.

Table 1.2. C++ 2011 Implementation Status

SectionDescriptionStatusComments
18 @@ -433,8 +431,7 @@ This status table is based on the table of contents of ISO/IEC 14882:2014. Some subclauses are not shown in the table where the content is unchanged since C++11 and the implementation is complete.

-This section describes the C++14 and library TS support in the GCC 12 release series, -not in any particular release. +This section describes the C++14 and library TS support in the GCC 12 release series.

Table 1.3. C++ 2014 Implementation Status

SectionDescriptionStatusComments
18 @@ -578,8 +575,7 @@ GCC 9.1 was the first release with non-experimental C++17 support, so the API and ABI of features added in C++17 is only stable since that release.

-This section describes the C++17 and library TS support in the GCC 12 release series, -not in any particular release. +This section describes the C++17 and library TS support in the GCC 12 release series.

The following table lists new library features that are included in the C++17 standard. The "Proposal" column provides a link to the @@ -1254,8 +1250,7 @@ options. The pre-defined symbol __cplusplus is used to check for the presence of the required flag.

-This section describes the C++20 and library TS support in the GCC 12 release series, -not in any particular release. +This section describes the C++20 and library TS support in the GCC 12 release series.

The following table lists new library features that are included in the C++20 standard. The "Proposal" column provides a link to the @@ -1720,8 +1715,7 @@ options. The pre-defined symbol __cplusplus is used to check for the presence of the required flag.

-This section describes the C++23 and library TS support in mainline GCC, -not in any particular release. +This section describes the C++23 and library TS support in the GCC 12 release series.

The following table lists new library features that have been accepted into the C++23 working draft. The "Proposal" column provides a link to the @@ -1925,8 +1919,7 @@ In this implementation the header names are prefixed by tr1/, for instance <tr1/functional>, <tr1/memory>, and so on.

-This page describes the TR1 support in the GCC 12 release series, -not in any particular release. +This page describes the TR1 support in the GCC 12 release series.

Table 1.11. C++ TR1 Implementation Status

SectionDescriptionStatusComments
2General Utilities
2.1Reference wrappers  
2.1.1Additions to header <functional> synopsisY 
2.1.2Class template reference_wrapper  
2.1.2.1reference_wrapper construct/copy/destroyY 
2.1.2.2reference_wrapper assignmentY 
2.1.2.3reference_wrapper accessY 
2.1.2.4reference_wrapper invocationY 
2.1.2.5reference_wrapper helper functionsY 
2.2Smart pointers  
2.2.1Additions to header <memory> synopsisY 
2.2.2Class bad_weak_ptrY 
2.2.3Class template shared_ptr 

Uses code from @@ -1946,8 +1939,7 @@ ISO/IEC TR 24733:2011, "Extensions for the programming language C++ to support decimal floating-point arithmetic".

-This page describes the TR 24733 support in the GCC 12 release series, -not in any particular release. +This page describes the TR 24733 support in the GCC 12 release series.

Table 1.12. C++ TR 24733 Implementation Status

SectionDescriptionStatusComments
0 diff --git a/libstdc++-v3/doc/xml/manual/status_cxx1998.xml b/libstdc++-v3/doc/xml/manual/status_cxx1998.xml index 6d3f6b09c41..107fd5ea398 100644 --- a/libstdc++-v3/doc/xml/manual/status_cxx1998.xml +++ b/libstdc++-v3/doc/xml/manual/status_cxx1998.xml @@ -18,8 +18,7 @@ This status table is based on the table of contents of ISO/IEC 14882:2003. -This section describes the C++ support in the GCC 12 release series, -not in any particular release. +This section describes the C++ support in the GCC 12 release series. diff --git a/libstdc++-v3/doc/xml/manual/status_cxx2011.xml b/libstdc++-v3/doc/xml/manual/status_cxx2011.xml index d929d4b5bac..5b1841c4f00 100644 --- a/libstdc++-v3/doc/xml/manual/status_cxx2011.xml +++ b/libstdc++-v3/doc/xml/manual/status_cxx2011.xml @@ -34,8 +34,7 @@ This status table is based on the table of contents of ISO/IEC 14882:2011. -This section describes the C++11 support in the GCC 12 release series, -not in any particular release. +This section describes the C++11 support in the GCC 12 release series. diff --git a/libstdc++-v3/doc/xml/manual/status_cxx2014.xml b/libstdc++-v3/doc/xml/manual/status_cxx2014.xml index becc4ed2684..48823633cf2 100644 --- a/libstdc++-v3/doc/xml/manual/status_cxx2014.xml +++ b/libstdc++-v3/doc/xml/manual/status_cxx2014.xml @@ -29,8 +29,7 @@ since C++11 and the implementation is complete. -This section describes the C++14 and library TS support in the GCC 12 release series, -not in any particular release. +This section describes the C++14 and library TS support in the GCC 12 release series. diff --git a/libstdc++-v3/doc/xml/manual/status_cxx2017.xml b/libstdc++-v3/doc/xml/manual/status_cxx2017.xml index e4a3a2ae0ed..0e219e4c46c 100644 --- a/libstdc++-v3/doc/xml/manual/status_cxx2017.xml +++ b/libstdc++-v3/doc/xml/manual/status_cxx2017.xml @@ -23,8 +23,7 @@ since that release. -This section describes the C++17 and library TS support in the GCC 12 release series, -not in any particular release. +This section describes the C++17 and library TS support in the GCC 12 release series. diff --git a/libstdc++-v3/doc/xml/manual/status_cxx2020.xml b/libstdc++-v3/doc/xml/manual/status_cxx2020.xml index d85fa59ae59..2be9984bdc6 100644 --- a/libstdc++-v3/doc/xml/manual/status_cxx2020.xml +++ b/libstdc++-v3/doc/xml/manual/status_cxx2020.xml @@ -20,8 +20,7 @@ presence of the required flag. -This section describes the C++20 and library TS support in the GCC 12 release series, -not in any particular release. +This section describes the C++20 and library TS support in the GCC 12 release series. diff --git a/libstdc++-v3/doc/xml/manual/status_cxx2023.xml b/libstdc++-v3/doc/xml/manual/status_cxx2023.xml index a6049e7b0d4..74a714c9a8e 100644 --- a/libstdc++-v3/doc/xml/manual/status_cxx2023.xml +++ b/libstdc++-v3/doc/xml/manual/status_cxx2023.xml @@ -20,8 +20,7 @@ presence of the required flag. -This section describes the C++23 and library TS support in mainline GCC, -not in any particular release. +This section describes the C++23 and library TS support in the GCC 12 release series. diff --git a/libstdc++-v3/doc/xml/manual/status_cxxtr1.xml b/libstdc++-v3/doc/xml/manual/status_cxxtr1.xml index 56fb3a86a0c..5d7e9edd3a4 100644 --- a/libstdc++-v3/doc/xml/manual/status_cxxtr1.xml +++ b/libstdc++-v3/doc/xml/manual/status_cxxtr1.xml @@ -22,8 +22,7 @@ In this implementation the header names are prefixed by -This page describes the TR1 support in the GCC 12 release series, -not in any particular release. +This page describes the TR1 support in the GCC 12 release series. diff --git a/libstdc++-v3/doc/xml/manual/status_cxxtr24733.xml b/libstdc++-v3/doc/xml/manual/status_cxxtr24733.xml index 26726a23217..02a5db30062 100644 --- a/libstdc++-v3/doc/xml/manual/status_cxxtr24733.xml +++ b/libstdc++-v3/doc/xml/manual/status_cxxtr24733.xml @@ -17,8 +17,7 @@ decimal floating-point arithmetic". -This page describes the TR 24733 support in the GCC 12 release series, -not in any particular release. +This page describes the TR 24733 support in the GCC 12 release series. commit 809d911415338207df6a405ebcd3a0cbbed8e968 Author: GCC Administrator Date: Wed Jun 26 00:20:42 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 271b3eb540d..c2b916de874 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240625 +20240626 diff --git a/libstdc++-v3/ChangeLog b/libstdc++-v3/ChangeLog index 954608173b1..a81c1ed112d 100644 --- a/libstdc++-v3/ChangeLog +++ b/libstdc++-v3/ChangeLog @@ -1,3 +1,17 @@ +2024-06-25 Jonathan Wakely + + * doc/xml/manual/status_cxx2023.xml: Change reference from + mainline GCC to the release branch. + * doc/xml/manual/status_cxx1998.xml: Remove confusing "not in + any particular release" text. + * doc/xml/manual/status_cxx2011.xml: Likewise. + * doc/xml/manual/status_cxx2014.xml: Likewise. + * doc/xml/manual/status_cxx2017.xml: Likewise. + * doc/xml/manual/status_cxx2020.xml: Likewise. + * doc/xml/manual/status_cxxtr1.xml: Likewise. + * doc/xml/manual/status_cxxtr24733.xml: Likewise. + * doc/html/manual/status.html: Regenerate. + 2024-06-21 Matthias Kretz Backported from master: commit 4f8dc81c175513132992a43e033404962c80b408 Author: GCC Administrator Date: Thu Jun 27 00:20:21 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c2b916de874..3402e552106 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240626 +20240627 commit 25cb13649b1765a21f21907f2d7a0aa2135accb5 Author: Kyrylo Tkachov Date: Wed Jun 19 14:56:02 2024 +0530 Add support for -mcpu=grace This adds support for the NVIDIA Grace CPU to aarch64. We reuse the tuning decisions for the Neoverse V2 core, but include a number of architecture features that are not enabled by default in -mcpu=neoverse-v2. This allows Grace users to more simply target the CPU with -mcpu=grace rather than remembering what extensions to tag on top of -mcpu=neoverse-v2. Bootstrapped and tested on aarch64-none-linux-gnu. gcc/ * config/aarch64/aarch64-cores.def (grace): New entry. * config/aarch64/aarch64-tune.md: Regenerate. * doc/invoke.texi (AArch64 Options): Document the above. Signed-off-by: Kyrylo Tkachov diff --git a/gcc/config/aarch64/aarch64-cores.def b/gcc/config/aarch64/aarch64-cores.def index 956afa70714..6532bdaafb5 100644 --- a/gcc/config/aarch64/aarch64-cores.def +++ b/gcc/config/aarch64/aarch64-cores.def @@ -176,5 +176,6 @@ AARCH64_CORE("cobalt-100", cobalt100, cortexa57, 9A, AARCH64_FL_FOR_ARCH9 | AA AARCH64_CORE("demeter", demeter, cortexa57, 9A, AARCH64_FL_FOR_ARCH9 | AARCH64_FL_I8MM | AARCH64_FL_BF16 | AARCH64_FL_SVE2_BITPERM | AARCH64_FL_RNG | AARCH64_FL_MEMTAG | AARCH64_FL_PROFILE, neoversev2, 0x41, 0xd4f, -1) AARCH64_CORE("neoverse-v2", neoversev2, cortexa57, 9A, AARCH64_FL_FOR_ARCH9 | AARCH64_FL_I8MM | AARCH64_FL_BF16 | AARCH64_FL_SVE2_BITPERM | AARCH64_FL_RNG | AARCH64_FL_MEMTAG | AARCH64_FL_PROFILE, neoversev2, 0x41, 0xd4f, -1) +AARCH64_CORE("grace", grace, cortexa57, 9A, AARCH64_FL_FOR_ARCH9 | AARCH64_FL_I8MM | AARCH64_FL_BF16 | AARCH64_FL_SVE2_BITPERM | AARCH64_FL_CRYPTO | AARCH64_FL_SHA3 | AARCH64_FL_SM4 | AARCH64_FL_SVE2_AES | AARCH64_FL_SVE2_SHA3 | AARCH64_FL_SVE2_SM4 | AARCH64_FL_PROFILE, neoversev2, 0x41, 0xd4f, -1) #undef AARCH64_CORE diff --git a/gcc/config/aarch64/aarch64-tune.md b/gcc/config/aarch64/aarch64-tune.md index 2c1852c8fe6..0c139e3e729 100644 --- a/gcc/config/aarch64/aarch64-tune.md +++ b/gcc/config/aarch64/aarch64-tune.md @@ -1,5 +1,5 @@ ;; -*- buffer-read-only: t -*- ;; Generated automatically by gentune.sh from aarch64-cores.def (define_attr "tune" - "cortexa34,cortexa35,cortexa53,cortexa57,cortexa72,cortexa73,thunderx,thunderxt88p1,thunderxt88,octeontx,octeontxt81,octeontxt83,thunderxt81,thunderxt83,ampere1,ampere1a,emag,xgene1,falkor,qdf24xx,exynosm1,phecda,thunderx2t99p1,vulcan,thunderx2t99,cortexa55,cortexa75,cortexa76,cortexa76ae,cortexa77,cortexa78,cortexa78ae,cortexa78c,cortexa65,cortexa65ae,cortexx1,ares,neoversen1,neoversee1,octeontx2,octeontx2t98,octeontx2t96,octeontx2t93,octeontx2f95,octeontx2f95n,octeontx2f95mm,a64fx,tsv110,thunderx3t110,zeus,neoversev1,neoverse512tvb,saphira,cortexa57cortexa53,cortexa72cortexa53,cortexa73cortexa35,cortexa73cortexa53,cortexa75cortexa55,cortexa76cortexa55,cortexr82,cortexa510,cortexa710,cortexx2,neoversen2,cobalt100,demeter,neoversev2" + "cortexa34,cortexa35,cortexa53,cortexa57,cortexa72,cortexa73,thunderx,thunderxt88p1,thunderxt88,octeontx,octeontxt81,octeontxt83,thunderxt81,thunderxt83,ampere1,ampere1a,emag,xgene1,falkor,qdf24xx,exynosm1,phecda,thunderx2t99p1,vulcan,thunderx2t99,cortexa55,cortexa75,cortexa76,cortexa76ae,cortexa77,cortexa78,cortexa78ae,cortexa78c,cortexa65,cortexa65ae,cortexx1,ares,neoversen1,neoversee1,octeontx2,octeontx2t98,octeontx2t96,octeontx2t93,octeontx2f95,octeontx2f95n,octeontx2f95mm,a64fx,tsv110,thunderx3t110,zeus,neoversev1,neoverse512tvb,saphira,cortexa57cortexa53,cortexa72cortexa53,cortexa73cortexa35,cortexa73cortexa53,cortexa75cortexa55,cortexa76cortexa55,cortexr82,cortexa510,cortexa710,cortexx2,neoversen2,cobalt100,demeter,neoversev2,grace" (const (symbol_ref "((enum attr_tune) aarch64_tune)"))) diff --git a/gcc/doc/invoke.texi b/gcc/doc/invoke.texi index c83f667260e..fbfa3241e7f 100644 --- a/gcc/doc/invoke.texi +++ b/gcc/doc/invoke.texi @@ -19203,8 +19203,8 @@ performance of the code. Permissible values for this option are: @samp{cortex-a78}, @samp{cortex-a78ae}, @samp{cortex-a78c}, @samp{ares}, @samp{exynos-m1}, @samp{emag}, @samp{falkor}, @samp{neoverse-512tvb}, @samp{neoverse-e1}, @samp{neoverse-n1}, -@samp{neoverse-n2}, @samp{neoverse-v1}, @samp{neoverse-v2}, @samp{qdf24xx}, -@samp{saphira}, @samp{phecda}, @samp{xgene1}, @samp{vulcan}, +@samp{neoverse-n2}, @samp{neoverse-v1}, @samp{neoverse-v2}, @samp{grace}, +@samp{qdf24xx}, @samp{saphira}, @samp{phecda}, @samp{xgene1}, @samp{vulcan}, @samp{octeontx}, @samp{octeontx81}, @samp{octeontx83}, @samp{octeontx2}, @samp{octeontx2t98}, @samp{octeontx2t96} @samp{octeontx2t93}, @samp{octeontx2f95}, @samp{octeontx2f95n}, commit 95ca5f458251e21123e45ec52c38d629d39cd0e4 Author: Alexandre Oliva Date: Thu Jun 27 08:44:54 2024 -0300 [libstdc++] [testsuite] defer to check_vect_support* [PR115454] The newly-added testcase overrides the default dg-do action set by check_vect_support_and_set_flags (in libstdc++-dg/conformance.exp), so it attempts to run the test even if runtime vector support is not available. Remove the explicit dg-do directive, so that the default is honored, and the test is run if vector support is found, and only compiled otherwise. for libstdc++-v3/ChangeLog PR libstdc++/115454 * testsuite/experimental/simd/pr115454_find_last_set.cc: Defer to check_vect_support_and_set_flags's default dg-do action. (cherry picked from commit 95faa1bea7bdc7f92fcccb3543bfcbc8184c5e5b) diff --git a/libstdc++-v3/testsuite/experimental/simd/pr115454_find_last_set.cc b/libstdc++-v3/testsuite/experimental/simd/pr115454_find_last_set.cc index 25a713b4e94..4ade8601f27 100644 --- a/libstdc++-v3/testsuite/experimental/simd/pr115454_find_last_set.cc +++ b/libstdc++-v3/testsuite/experimental/simd/pr115454_find_last_set.cc @@ -1,5 +1,4 @@ // { dg-options "-std=gnu++17" } -// { dg-do run { target *-*-* } } // { dg-require-effective-target c++17 } // { dg-additional-options "-march=x86-64-v4" { target avx512f_runtime } } // { dg-require-cmath "" } commit aba733582ecf079a1dd7f58f9dd47959b7e66d4f Author: GCC Administrator Date: Fri Jun 28 00:20:48 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index de45aac0538..ca82b09f87c 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,9 @@ +2024-06-27 Kyrylo Tkachov + + * config/aarch64/aarch64-cores.def (grace): New entry. + * config/aarch64/aarch64-tune.md: Regenerate. + * doc/invoke.texi (AArch64 Options): Document the above. + 2024-06-24 Kewen Lin Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 3402e552106..b08c5cee3e2 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240627 +20240628 diff --git a/libstdc++-v3/ChangeLog b/libstdc++-v3/ChangeLog index a81c1ed112d..62345b0d11e 100644 --- a/libstdc++-v3/ChangeLog +++ b/libstdc++-v3/ChangeLog @@ -1,3 +1,12 @@ +2024-06-27 Alexandre Oliva + + Backported from master: + 2024-06-27 Alexandre Oliva + + PR libstdc++/115454 + * testsuite/experimental/simd/pr115454_find_last_set.cc: Defer + to check_vect_support_and_set_flags's default dg-do action. + 2024-06-25 Jonathan Wakely * doc/xml/manual/status_cxx2023.xml: Change reference from commit 96ef3367067219c8e3eb88c0474a1090cc7749b4 Author: Kewen Lin Date: Thu Jun 20 20:23:56 2024 -0500 rs6000: Fix wrong RTL patterns for vector merge high/low word on LE Commit r12-4496 changes some define_expands and define_insns for vector merge high/low word, which are altivec_vmrg[hl]w, vsx_xxmrg[hl]w_. These defines are mainly for built-in function vec_merge{h,l}, __builtin_vsx_xxmrghw, __builtin_vsx_xxmrghw_4si and some internal gen function needs. These functions should consider endianness, taking vec_mergeh as example, as PVIPR defines, vec_mergeh "Merges the first halves (in element order) of two vectors", it does note it's in element order. So it's mapped into vmrghw on BE while vmrglw on LE respectively. Although the mapped insns are different, as the discussion in PR106069, the RTL pattern should be still the same, it is conformed before commit r12-4496, define_expand altivec_vmrghw got expanded into: (vec_select:VSX_W (vec_concat: (match_operand:VSX_W 1 "register_operand" "wa,v") (match_operand:VSX_W 2 "register_operand" "wa,v")) (parallel [(const_int 0) (const_int 4) (const_int 1) (const_int 5)])))] on both BE and LE then. But commit r12-4496 changed it to expand into: (vec_select:VSX_W (vec_concat: (match_operand:VSX_W 1 "register_operand" "wa,v") (match_operand:VSX_W 2 "register_operand" "wa,v")) (parallel [(const_int 0) (const_int 4) (const_int 1) (const_int 5)])))] on BE, and (vec_select:VSX_W (vec_concat: (match_operand:VSX_W 1 "register_operand" "wa,v") (match_operand:VSX_W 2 "register_operand" "wa,v")) (parallel [(const_int 2) (const_int 6) (const_int 3) (const_int 7)])))] on LE, although the mapped insn are still vmrghw on BE and vmrglw on LE, the associated RTL pattern is completely wrong and inconsistent with the mapped insn. If optimization passes leave this pattern alone, even if its pattern doesn't represent its mapped insn, it's still fine, that's why simple testing on bif doesn't expose this issue. But once some optimization pass such as combine does some changes basing on this wrong pattern, because the pattern doesn't match the semantics that the expanded insn is intended to represent, it would cause the unexpected result. So this patch is to fix the wrong RTL pattern, ensure the associated RTL patterns become the same as before which can have the same semantic as their mapped insns. With the proposed patch, the expanders like altivec_vmrghw expands into altivec_vmrghb_direct_be or altivec_vmrglb_direct_le depending on endianness, "direct" can easily show which insn would be generated, _be and _le are mainly for the different RTL patterns as endianness. Co-authored-by: Xionghu Luo PR target/106069 PR target/115355 gcc/ChangeLog: * config/rs6000/altivec.md (altivec_vmrghw_direct_): Rename to ... (altivec_vmrghw_direct__be): ... this. Add the condition BYTES_BIG_ENDIAN. (altivec_vmrghw_direct__le): New define_insn. (altivec_vmrglw_direct_): Rename to ... (altivec_vmrglw_direct__be): ... this. Add the condition BYTES_BIG_ENDIAN. (altivec_vmrglw_direct__le): New define_insn. (altivec_vmrghw): Adjust by calling gen_altivec_vmrghw_direct_v4si_be for BE and gen_altivec_vmrglw_direct_v4si_le for LE. (altivec_vmrglw): Adjust by calling gen_altivec_vmrglw_direct_v4si_be for BE and gen_altivec_vmrghw_direct_v4si_le for LE. (vec_widen_umult_hi_v8hi): Adjust the call to gen_altivec_vmrghw_direct_v4si by gen_altivec_vmrghw for BE and by gen_altivec_vmrglw for LE. (vec_widen_smult_hi_v8hi): Likewise. (vec_widen_umult_lo_v8hi): Adjust the call to gen_altivec_vmrglw_direct_v4si by gen_altivec_vmrglw for BE and by gen_altivec_vmrghw for LE (vec_widen_smult_lo_v8hi): Likewise. * config/rs6000/rs6000.cc (altivec_expand_vec_perm_const): Replace CODE_FOR_altivec_vmrghw_direct_v4si by CODE_FOR_altivec_vmrghw_direct_v4si_be for BE and CODE_FOR_altivec_vmrghw_direct_v4si_le for LE. And replace CODE_FOR_altivec_vmrglw_direct_v4si by CODE_FOR_altivec_vmrglw_direct_v4si_be for BE and CODE_FOR_altivec_vmrglw_direct_v4si_le for LE. * config/rs6000/vsx.md (vsx_xxmrghw_): Adjust by calling gen_altivec_vmrghw_direct_v4si_be for BE and gen_altivec_vmrglw_direct_v4si_le for LE. (vsx_xxmrglw_): Adjust by calling gen_altivec_vmrglw_direct_v4si_be for BE and gen_altivec_vmrghw_direct_v4si_le for LE. gcc/testsuite/ChangeLog: * g++.target/powerpc/pr106069.C: New test. * gcc.target/powerpc/pr115355.c: New test. (cherry picked from commit 52c112800d9f44457c4832309a48c00945811313) diff --git a/gcc/config/rs6000/altivec.md b/gcc/config/rs6000/altivec.md index 3849db5ca3c..0c408a9e839 100644 --- a/gcc/config/rs6000/altivec.md +++ b/gcc/config/rs6000/altivec.md @@ -1212,16 +1212,18 @@ (use (match_operand:V4SI 2 "register_operand"))] "VECTOR_MEM_ALTIVEC_P (V4SImode)" { - rtx (*fun) (rtx, rtx, rtx); - fun = BYTES_BIG_ENDIAN ? gen_altivec_vmrghw_direct_v4si - : gen_altivec_vmrglw_direct_v4si; - if (!BYTES_BIG_ENDIAN) - std::swap (operands[1], operands[2]); - emit_insn (fun (operands[0], operands[1], operands[2])); + if (BYTES_BIG_ENDIAN) + emit_insn (gen_altivec_vmrghw_direct_v4si_be (operands[0], + operands[1], + operands[2])); + else + emit_insn (gen_altivec_vmrglw_direct_v4si_le (operands[0], + operands[2], + operands[1])); DONE; }) -(define_insn "altivec_vmrghw_direct_" +(define_insn "altivec_vmrghw_direct__be" [(set (match_operand:VSX_W 0 "register_operand" "=wa,v") (vec_select:VSX_W (vec_concat: @@ -1229,7 +1231,21 @@ (match_operand:VSX_W 2 "register_operand" "wa,v")) (parallel [(const_int 0) (const_int 4) (const_int 1) (const_int 5)])))] - "TARGET_ALTIVEC" + "TARGET_ALTIVEC && BYTES_BIG_ENDIAN" + "@ + xxmrghw %x0,%x1,%x2 + vmrghw %0,%1,%2" + [(set_attr "type" "vecperm")]) + +(define_insn "altivec_vmrghw_direct__le" + [(set (match_operand:VSX_W 0 "register_operand" "=wa,v") + (vec_select:VSX_W + (vec_concat: + (match_operand:VSX_W 2 "register_operand" "wa,v") + (match_operand:VSX_W 1 "register_operand" "wa,v")) + (parallel [(const_int 2) (const_int 6) + (const_int 3) (const_int 7)])))] + "TARGET_ALTIVEC && !BYTES_BIG_ENDIAN" "@ xxmrghw %x0,%x1,%x2 vmrghw %0,%1,%2" @@ -1318,16 +1334,18 @@ (use (match_operand:V4SI 2 "register_operand"))] "VECTOR_MEM_ALTIVEC_P (V4SImode)" { - rtx (*fun) (rtx, rtx, rtx); - fun = BYTES_BIG_ENDIAN ? gen_altivec_vmrglw_direct_v4si - : gen_altivec_vmrghw_direct_v4si; - if (!BYTES_BIG_ENDIAN) - std::swap (operands[1], operands[2]); - emit_insn (fun (operands[0], operands[1], operands[2])); + if (BYTES_BIG_ENDIAN) + emit_insn (gen_altivec_vmrglw_direct_v4si_be (operands[0], + operands[1], + operands[2])); + else + emit_insn (gen_altivec_vmrghw_direct_v4si_le (operands[0], + operands[2], + operands[1])); DONE; }) -(define_insn "altivec_vmrglw_direct_" +(define_insn "altivec_vmrglw_direct__be" [(set (match_operand:VSX_W 0 "register_operand" "=wa,v") (vec_select:VSX_W (vec_concat: @@ -1335,7 +1353,21 @@ (match_operand:VSX_W 2 "register_operand" "wa,v")) (parallel [(const_int 2) (const_int 6) (const_int 3) (const_int 7)])))] - "TARGET_ALTIVEC" + "TARGET_ALTIVEC && BYTES_BIG_ENDIAN" + "@ + xxmrglw %x0,%x1,%x2 + vmrglw %0,%1,%2" + [(set_attr "type" "vecperm")]) + +(define_insn "altivec_vmrglw_direct__le" + [(set (match_operand:VSX_W 0 "register_operand" "=wa,v") + (vec_select:VSX_W + (vec_concat: + (match_operand:VSX_W 2 "register_operand" "wa,v") + (match_operand:VSX_W 1 "register_operand" "wa,v")) + (parallel [(const_int 0) (const_int 4) + (const_int 1) (const_int 5)])))] + "TARGET_ALTIVEC && !BYTES_BIG_ENDIAN" "@ xxmrglw %x0,%x1,%x2 vmrglw %0,%1,%2" @@ -3807,13 +3839,13 @@ { emit_insn (gen_altivec_vmuleuh (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmulouh (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrghw_direct_v4si (operands[0], ve, vo)); + emit_insn (gen_altivec_vmrghw (operands[0], ve, vo)); } else { emit_insn (gen_altivec_vmulouh (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmuleuh (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrghw_direct_v4si (operands[0], vo, ve)); + emit_insn (gen_altivec_vmrglw (operands[0], ve, vo)); } DONE; }) @@ -3832,13 +3864,13 @@ { emit_insn (gen_altivec_vmuleuh (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmulouh (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrglw_direct_v4si (operands[0], ve, vo)); + emit_insn (gen_altivec_vmrglw (operands[0], ve, vo)); } else { emit_insn (gen_altivec_vmulouh (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmuleuh (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrglw_direct_v4si (operands[0], vo, ve)); + emit_insn (gen_altivec_vmrghw (operands[0], ve, vo)); } DONE; }) @@ -3857,13 +3889,13 @@ { emit_insn (gen_altivec_vmulesh (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmulosh (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrghw_direct_v4si (operands[0], ve, vo)); + emit_insn (gen_altivec_vmrghw (operands[0], ve, vo)); } else { emit_insn (gen_altivec_vmulosh (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmulesh (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrghw_direct_v4si (operands[0], vo, ve)); + emit_insn (gen_altivec_vmrglw (operands[0], ve, vo)); } DONE; }) @@ -3882,13 +3914,13 @@ { emit_insn (gen_altivec_vmulesh (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmulosh (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrglw_direct_v4si (operands[0], ve, vo)); + emit_insn (gen_altivec_vmrglw (operands[0], ve, vo)); } else { emit_insn (gen_altivec_vmulosh (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmulesh (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrglw_direct_v4si (operands[0], vo, ve)); + emit_insn (gen_altivec_vmrghw (operands[0], ve, vo)); } DONE; }) diff --git a/gcc/config/rs6000/rs6000.cc b/gcc/config/rs6000/rs6000.cc index f5db6436dfa..23b553131a9 100644 --- a/gcc/config/rs6000/rs6000.cc +++ b/gcc/config/rs6000/rs6000.cc @@ -22979,8 +22979,8 @@ altivec_expand_vec_perm_const (rtx target, rtx op0, rtx op1, : CODE_FOR_altivec_vmrglh_direct, {0, 1, 16, 17, 2, 3, 18, 19, 4, 5, 20, 21, 6, 7, 22, 23}}, {OPTION_MASK_ALTIVEC, - BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrghw_direct_v4si - : CODE_FOR_altivec_vmrglw_direct_v4si, + BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrghw_direct_v4si_be + : CODE_FOR_altivec_vmrglw_direct_v4si_le, {0, 1, 2, 3, 16, 17, 18, 19, 4, 5, 6, 7, 20, 21, 22, 23}}, {OPTION_MASK_ALTIVEC, BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrglb_direct @@ -22991,8 +22991,8 @@ altivec_expand_vec_perm_const (rtx target, rtx op0, rtx op1, : CODE_FOR_altivec_vmrghh_direct, {8, 9, 24, 25, 10, 11, 26, 27, 12, 13, 28, 29, 14, 15, 30, 31}}, {OPTION_MASK_ALTIVEC, - BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrglw_direct_v4si - : CODE_FOR_altivec_vmrghw_direct_v4si, + BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrglw_direct_v4si_be + : CODE_FOR_altivec_vmrghw_direct_v4si_le, {8, 9, 10, 11, 24, 25, 26, 27, 12, 13, 14, 15, 28, 29, 30, 31}}, {OPTION_MASK_P8_VECTOR, BYTES_BIG_ENDIAN ? CODE_FOR_p8_vmrgew_v4sf_direct diff --git a/gcc/config/rs6000/vsx.md b/gcc/config/rs6000/vsx.md index e16f893c073..226a1049917 100644 --- a/gcc/config/rs6000/vsx.md +++ b/gcc/config/rs6000/vsx.md @@ -4694,12 +4694,14 @@ (const_int 1) (const_int 5)])))] "VECTOR_MEM_VSX_P (mode)" { - rtx (*fun) (rtx, rtx, rtx); - fun = BYTES_BIG_ENDIAN ? gen_altivec_vmrghw_direct_ - : gen_altivec_vmrglw_direct_; - if (!BYTES_BIG_ENDIAN) - std::swap (operands[1], operands[2]); - emit_insn (fun (operands[0], operands[1], operands[2])); + if (BYTES_BIG_ENDIAN) + emit_insn (gen_altivec_vmrghw_direct_v4si_be (operands[0], + operands[1], + operands[2])); + else + emit_insn (gen_altivec_vmrglw_direct_v4si_le (operands[0], + operands[2], + operands[1])); DONE; } [(set_attr "type" "vecperm")]) @@ -4714,12 +4716,14 @@ (const_int 3) (const_int 7)])))] "VECTOR_MEM_VSX_P (mode)" { - rtx (*fun) (rtx, rtx, rtx); - fun = BYTES_BIG_ENDIAN ? gen_altivec_vmrglw_direct_ - : gen_altivec_vmrghw_direct_; - if (!BYTES_BIG_ENDIAN) - std::swap (operands[1], operands[2]); - emit_insn (fun (operands[0], operands[1], operands[2])); + if (BYTES_BIG_ENDIAN) + emit_insn (gen_altivec_vmrglw_direct_v4si_be (operands[0], + operands[1], + operands[2])); + else + emit_insn (gen_altivec_vmrghw_direct_v4si_le (operands[0], + operands[2], + operands[1])); DONE; } [(set_attr "type" "vecperm")]) diff --git a/gcc/testsuite/g++.target/powerpc/pr106069.C b/gcc/testsuite/g++.target/powerpc/pr106069.C new file mode 100644 index 00000000000..537207d2fe8 --- /dev/null +++ b/gcc/testsuite/g++.target/powerpc/pr106069.C @@ -0,0 +1,119 @@ +/* { dg-options "-O -fno-tree-forwprop -maltivec" } */ +/* { dg-require-effective-target vmx_hw } */ +/* { dg-do run } */ + +typedef __attribute__ ((altivec (vector__))) unsigned native_simd_type; + +union +{ + native_simd_type V; + int R[4]; +} store_le_vec; + +struct S +{ + S () = default; + S (unsigned B0) + { + native_simd_type val{B0}; + m_simd = val; + } + void store_le (unsigned int out[]) + { + store_le_vec.V = m_simd; + unsigned int x0 = store_le_vec.R[0]; + __builtin_memcpy (out, &x0, 4); + } + S rotl (unsigned int r) + { + native_simd_type rot{r}; + return __builtin_vec_rl (m_simd, rot); + } + void operator+= (S other) + { + m_simd = __builtin_vec_add (m_simd, other.m_simd); + } + void operator^= (S other) + { + m_simd = __builtin_vec_xor (m_simd, other.m_simd); + } + static void transpose (S &B0, S B1, S B2, S B3) + { + native_simd_type T0 = __builtin_vec_mergeh (B0.m_simd, B2.m_simd); + native_simd_type T1 = __builtin_vec_mergeh (B1.m_simd, B3.m_simd); + native_simd_type T2 = __builtin_vec_mergel (B0.m_simd, B2.m_simd); + native_simd_type T3 = __builtin_vec_mergel (B1.m_simd, B3.m_simd); + B0 = __builtin_vec_mergeh (T0, T1); + B3 = __builtin_vec_mergel (T2, T3); + } + S (native_simd_type x) : m_simd (x) {} + native_simd_type m_simd; +}; + +void +foo (unsigned int output[], unsigned state[]) +{ + S R00 = state[0]; + S R01 = state[0]; + S R02 = state[2]; + S R03 = state[0]; + S R05 = state[5]; + S R06 = state[6]; + S R07 = state[7]; + S R08 = state[8]; + S R09 = state[9]; + S R10 = state[10]; + S R11 = state[11]; + S R12 = state[12]; + S R13 = state[13]; + S R14 = state[4]; + S R15 = state[15]; + for (int r = 0; r != 10; ++r) + { + R09 += R13; + R11 += R15; + R05 ^= R09; + R06 ^= R10; + R07 ^= R11; + R07 = R07.rotl (7); + R00 += R05; + R01 += R06; + R02 += R07; + R15 ^= R00; + R12 ^= R01; + R13 ^= R02; + R00 += R05; + R01 += R06; + R02 += R07; + R15 ^= R00; + R12 = R12.rotl (8); + R13 = R13.rotl (8); + R10 += R15; + R11 += R12; + R08 += R13; + R09 += R14; + R05 ^= R10; + R06 ^= R11; + R07 ^= R08; + R05 = R05.rotl (7); + R06 = R06.rotl (7); + R07 = R07.rotl (7); + } + R00 += state[0]; + S::transpose (R00, R01, R02, R03); + R00.store_le (output); +} + +unsigned int res[1]; +unsigned main_state[]{1634760805, 60878, 2036477234, 6, + 0, 825562964, 1471091955, 1346092787, + 506976774, 4197066702, 518848283, 118491664, + 0, 0, 0, 0}; +int +main () +{ + foo (res, main_state); + if (res[0] != 0x41fcef98) + __builtin_abort (); + return 0; +} diff --git a/gcc/testsuite/gcc.target/powerpc/pr115355.c b/gcc/testsuite/gcc.target/powerpc/pr115355.c new file mode 100644 index 00000000000..8955126b808 --- /dev/null +++ b/gcc/testsuite/gcc.target/powerpc/pr115355.c @@ -0,0 +1,37 @@ +/* { dg-do run } */ +/* { dg-require-effective-target p9vector_hw } */ +/* Force vectorization with -fno-vect-cost-model to have vector unpack + which exposes the issue in PR115355. */ +/* { dg-options "-O2 -mdejagnu-cpu=power9 -fno-vect-cost-model" } */ + +/* Verify it runs successfully. */ + +__attribute__((noipa)) +void setToIdentityGOOD(unsigned long long *mVec, unsigned int mLen) +{ + #pragma GCC novector + for (unsigned int i = 0; i < mLen; i++) + mVec[i] = i; +} + +__attribute__((noipa)) +void setToIdentityBAD(unsigned long long *mVec, unsigned int mLen) +{ + for (unsigned int i = 0; i < mLen; i++) + mVec[i] = i; +} + +unsigned long long vec1[100]; +unsigned long long vec2[100]; + +int main() +{ + unsigned int l = 29; + setToIdentityGOOD (vec1, 29); + setToIdentityBAD (vec2, 29); + + if (__builtin_memcmp (vec1, vec2, l * sizeof (vec1[0])) != 0) + __builtin_abort (); + + return 0; +} commit ae5bf1a308ffc634e5104ba251886af903f983d6 Author: GCC Administrator Date: Sat Jun 29 00:20:32 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index ca82b09f87c..26851373a78 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,46 @@ +2024-06-28 Kewen Lin + + Backported from master: + 2024-06-21 Kewen Lin + Xionghu Luo + + PR target/106069 + PR target/115355 + * config/rs6000/altivec.md (altivec_vmrghw_direct_): Rename + to ... + (altivec_vmrghw_direct__be): ... this. Add the condition + BYTES_BIG_ENDIAN. + (altivec_vmrghw_direct__le): New define_insn. + (altivec_vmrglw_direct_): Rename to ... + (altivec_vmrglw_direct__be): ... this. Add the condition + BYTES_BIG_ENDIAN. + (altivec_vmrglw_direct__le): New define_insn. + (altivec_vmrghw): Adjust by calling gen_altivec_vmrghw_direct_v4si_be + for BE and gen_altivec_vmrglw_direct_v4si_le for LE. + (altivec_vmrglw): Adjust by calling gen_altivec_vmrglw_direct_v4si_be + for BE and gen_altivec_vmrghw_direct_v4si_le for LE. + (vec_widen_umult_hi_v8hi): Adjust the call to + gen_altivec_vmrghw_direct_v4si by gen_altivec_vmrghw for BE + and by gen_altivec_vmrglw for LE. + (vec_widen_smult_hi_v8hi): Likewise. + (vec_widen_umult_lo_v8hi): Adjust the call to + gen_altivec_vmrglw_direct_v4si by gen_altivec_vmrglw for BE + and by gen_altivec_vmrghw for LE + (vec_widen_smult_lo_v8hi): Likewise. + * config/rs6000/rs6000.cc (altivec_expand_vec_perm_const): Replace + CODE_FOR_altivec_vmrghw_direct_v4si by + CODE_FOR_altivec_vmrghw_direct_v4si_be for BE and + CODE_FOR_altivec_vmrghw_direct_v4si_le for LE. And replace + CODE_FOR_altivec_vmrglw_direct_v4si by + CODE_FOR_altivec_vmrglw_direct_v4si_be for BE and + CODE_FOR_altivec_vmrglw_direct_v4si_le for LE. + * config/rs6000/vsx.md (vsx_xxmrghw_): Adjust by calling + gen_altivec_vmrghw_direct_v4si_be for BE and + gen_altivec_vmrglw_direct_v4si_le for LE. + (vsx_xxmrglw_): Adjust by calling + gen_altivec_vmrglw_direct_v4si_be for BE and + gen_altivec_vmrghw_direct_v4si_le for LE. + 2024-06-27 Kyrylo Tkachov * config/aarch64/aarch64-cores.def (grace): New entry. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index b08c5cee3e2..674b36c5c7c 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240628 +20240629 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index def1a220cd2..5c60df32f88 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,14 @@ +2024-06-28 Kewen Lin + + Backported from master: + 2024-06-21 Kewen Lin + Xionghu Luo + + PR target/106069 + PR target/115355 + * g++.target/powerpc/pr106069.C: New test. + * gcc.target/powerpc/pr115355.c: New test. + 2024-06-24 Kewen Lin Backported from master: commit 1a837bc65b6010a8af252480699eb56b567f4482 Author: GCC Administrator Date: Sun Jun 30 00:20:30 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 674b36c5c7c..e88fa6b412b 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240629 +20240630 commit 1d6c409fdff3255fe316be05b3f0a1f77a998a59 Author: GCC Administrator Date: Mon Jul 1 00:21:04 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e88fa6b412b..ff13db4fe3e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240630 +20240701 commit 53305588cfbf74604bafcc27902e1eded5677ae6 Author: Georg-Johann Lay Date: Mon Jul 1 12:31:01 2024 +0200 AVR: target/88236, target/115726 - Fix __memx code generation. PR target/88236 PR target/115726 gcc/ * config/avr/avr.md (mov) [avr_mem_memx_p]: Expand in such a way that the destination does not overlap with any hard register clobbered / used by xload8qi_A resp. xload_A. * config/avr/avr.cc (avr_out_xload): Avoid early-clobber situation for Z by executing just one load when the output register overlaps with Z. gcc/testsuite/ * gcc.target/avr/torture/pr88236-pr115726.c: New test. (cherry picked from commit 3d23abd3dd9c8c226ea302203b214b346f4fe8d7) diff --git a/gcc/config/avr/avr.cc b/gcc/config/avr/avr.cc index bc15017c61c..ee033d3204d 100644 --- a/gcc/config/avr/avr.cc +++ b/gcc/config/avr/avr.cc @@ -4071,7 +4071,13 @@ avr_out_xload (rtx_insn *insn ATTRIBUTE_UNUSED, rtx *op, int *plen) xop[2] = lpm_addr_reg_rtx; xop[3] = AVR_HAVE_LPMX ? op[0] : lpm_reg_rtx; - avr_asm_len (AVR_HAVE_LPMX ? "lpm %3,%a2" : "lpm", xop, plen, -1); + if (plen) + *plen = 0; + + if (reg_overlap_mentioned_p (xop[3], lpm_addr_reg_rtx)) + avr_asm_len ("sbrs %1,7", xop, plen, 1); + + avr_asm_len (AVR_HAVE_LPMX ? "lpm %3,%a2" : "lpm", xop, plen, 1); avr_asm_len ("sbrc %1,7" CR_TAB "ld %3,%a2", xop, plen, 2); diff --git a/gcc/config/avr/avr.md b/gcc/config/avr/avr.md index de029476908..f76249340b8 100644 --- a/gcc/config/avr/avr.md +++ b/gcc/config/avr/avr.md @@ -715,12 +715,26 @@ if (!REG_P (addr)) src = replace_equiv_address (src, copy_to_mode_reg (PSImode, addr)); + rtx dest2 = reg_overlap_mentioned_p (dest, lpm_addr_reg_rtx) + ? gen_reg_rtx (mode) + : dest; + if (!avr_xload_libgcc_p (mode)) /* ; No here because gen_xload8_A only iterates over ALL1. ; insn-emit does not depend on the mode, it's all about operands. */ - emit_insn (gen_xload8qi_A (dest, src)); + emit_insn (gen_xload8qi_A (dest2, src)); else - emit_insn (gen_xload_A (dest, src)); + { + rtx reg_22 = gen_rtx_REG (mode, 22); + if (reg_overlap_mentioned_p (dest2, reg_22) + || reg_overlap_mentioned_p (dest2, all_regs_rtx[21])) + dest2 = gen_reg_rtx (mode); + + emit_insn (gen_xload_A (dest2, src)); + } + + if (dest2 != dest) + emit_move_insn (dest, dest2); DONE; } diff --git a/gcc/testsuite/gcc.target/avr/torture/pr88236-pr115726.c b/gcc/testsuite/gcc.target/avr/torture/pr88236-pr115726.c new file mode 100644 index 00000000000..9fd5fd3b5f5 --- /dev/null +++ b/gcc/testsuite/gcc.target/avr/torture/pr88236-pr115726.c @@ -0,0 +1,115 @@ +/* { dg-do run { target { ! avr_tiny } } } */ +/* { dg-additional-options "-std=gnu99" } */ + +const __flash char fvals8[] = { 1, 2, 3 }; +char rvals8[] = { 0, 2, 4 }; + +const __flash int fvals16[] = { 1, 2, 3 }; +int rvals16[] = { 0, 2, 4 }; + +__attribute__((noinline, noclone)) +char xload8_r30 (const __memx char *pc) +{ + register char c __asm ("r30"); + c = *pc; + __asm (";;" : "+r" (c)); + return c; +} + +__attribute__((noinline, noclone)) +int xload16_r30 (const __memx int *pc) +{ + register int c __asm ("r30"); + c = *pc; + __asm (";;" : "+r" (c)); + return c; +} + +__attribute__((noinline, noclone)) +char xload8_r22 (const __memx char *pc) +{ + register char c __asm ("r22"); + c = *pc; + __asm (";;" : "+r" (c)); + return c; +} + +__attribute__((noinline, noclone)) +int xload16_r22 (const __memx int *pc) +{ + register int c __asm ("r22"); + c = *pc; + __asm (";;" : "+r" (c)); + return c; +} + +__attribute__((noinline, noclone)) +int xload16_r20 (const __memx int *pc) +{ + register int c __asm ("r20"); + c = *pc; + __asm (";;" : "+r" (c)); + return c; +} + +void test8 (void) +{ + char c; + for (int i = 0; i < 3; ++i) + { + c = xload8_r30 (fvals8 + i); + if (c != 1 + i) + __builtin_exit (__LINE__); + + c = xload8_r22 (fvals8 + i); + if (c != 1 + i) + __builtin_exit (__LINE__); + + c = xload8_r30 (rvals8 + i); + if (c != 2 * i) + __builtin_exit (__LINE__); + + c = xload8_r22 (rvals8 + i); + if (c != 2 * i) + __builtin_exit (__LINE__); + } +} + +void test16 (void) +{ + int c; + for (int i = 0; i < 3; ++i) + { + c = xload16_r30 (fvals16 + i); + if (c != 1 + i) + __builtin_exit (__LINE__); + + c = xload16_r22 (fvals16 + i); + if (c != 1 + i) + __builtin_exit (__LINE__); + + c = xload16_r20 (fvals16 + i); + if (c != 1 + i) + __builtin_exit (__LINE__); + + c = xload16_r30 (rvals16 + i); + if (c != 2 * i) + __builtin_exit (__LINE__); + + c = xload16_r22 (rvals16 + i); + if (c != 2 * i) + __builtin_exit (__LINE__); + + c = xload16_r20 (rvals16 + i); + if (c != 2 * i) + __builtin_exit (__LINE__); + } +} + +int main (void) +{ + test8(); + test16(); + + return 0; +} commit 4351caffa7f9112712ce396421d95f7bad90a8f3 Author: GCC Administrator Date: Tue Jul 2 00:20:50 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 26851373a78..45bf020f832 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,17 @@ +2024-07-01 Georg-Johann Lay + + Backported from master: + 2024-07-01 Georg-Johann Lay + + PR target/88236 + PR target/115726 + * config/avr/avr.md (mov) [avr_mem_memx_p]: Expand in such a + way that the destination does not overlap with any hard register + clobbered / used by xload8qi_A resp. xload_A. + * config/avr/avr.cc (avr_out_xload): Avoid early-clobber + situation for Z by executing just one load when the output register + overlaps with Z. + 2024-06-28 Kewen Lin Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index ff13db4fe3e..7f1fc5ab8cf 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240701 +20240702 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 5c60df32f88..60449edd77b 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,12 @@ +2024-07-01 Georg-Johann Lay + + Backported from master: + 2024-07-01 Georg-Johann Lay + + PR target/88236 + PR target/115726 + * gcc.target/avr/torture/pr88236-pr115726.c: New test. + 2024-06-28 Kewen Lin Backported from master: commit c99573f7c9fafabaff5d2a51caed98005da49007 Author: GCC Administrator Date: Wed Jul 3 00:21:39 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7f1fc5ab8cf..f8fa5e4aa67 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240702 +20240703 commit 13f0528c782c3732052973a5d340769af8182c8f Author: Kewen Lin Date: Wed Jun 26 02:16:17 2024 -0500 rs6000: Fix wrong RTL patterns for vector merge high/low char on LE Commit r12-4496 changes some define_expands and define_insns for vector merge high/low char, which are altivec_vmrg[hl]b. These defines are mainly for built-in function vec_merge{h,l} and some internal gen function needs. These functions should consider endianness, taking vec_mergeh as example, as PVIPR defines, vec_mergeh "Merges the first halves (in element order) of two vectors", it does note it's in element order. So it's mapped into vmrghb on BE while vmrglb on LE respectively. Although the mapped insns are different, as the discussion in PR106069, the RTL pattern should be still the same, it is conformed before commit r12-4496, but gets changed into different patterns on BE and LE starting from commit r12-4496. Similar to 32-bit element case in commit log of r15-1504, this 8-bit element pattern on LE doesn't actually match what the underlying insn is intended to represent, once some optimization like combine does some changes basing on it, it would cause the unexpected consequence. The newly constructed test case pr106069-1.c is a typical example for this issue. So this patch is to fix the wrong RTL pattern, ensure the associated RTL patterns become the same as before which can have the same semantic as their mapped insns. With the proposed patch, the expanders like altivec_vmrghb expands into altivec_vmrghb_direct_be or altivec_vmrglb_direct_le depending on endianness, "direct" can easily show which insn would be generated, _be and _le are mainly for the different RTL patterns as endianness. Co-authored-by: Xionghu Luo PR target/106069 PR target/115355 gcc/ChangeLog: * config/rs6000/altivec.md (altivec_vmrghb_direct): Rename to ... (altivec_vmrghb_direct_be): ... this. Add condition BYTES_BIG_ENDIAN. (altivec_vmrghb_direct_le): New define_insn. (altivec_vmrglb_direct): Rename to ... (altivec_vmrglb_direct_be): ... this. Add condition BYTES_BIG_ENDIAN. (altivec_vmrglb_direct_le): New define_insn. (altivec_vmrghb): Adjust by calling gen_altivec_vmrghb_direct_be for BE and gen_altivec_vmrglb_direct_le for LE. (altivec_vmrglb): Adjust by calling gen_altivec_vmrglb_direct_be for BE and gen_altivec_vmrghb_direct_le for LE. * config/rs6000/rs6000.cc (altivec_expand_vec_perm_const): Replace CODE_FOR_altivec_vmrghb_direct by CODE_FOR_altivec_vmrghb_direct_be for BE and CODE_FOR_altivec_vmrghb_direct_le for LE. And replace CODE_FOR_altivec_vmrglb_direct by CODE_FOR_altivec_vmrglb_direct_be for BE and CODE_FOR_altivec_vmrglb_direct_le for LE. gcc/testsuite/ChangeLog: * gcc.target/powerpc/pr106069-1.c: New test. (cherry picked from commit 62520e4e9f7e2fe8a16ee57a4bd35da2e921ae22) diff --git a/gcc/config/rs6000/altivec.md b/gcc/config/rs6000/altivec.md index 0c408a9e839..b8baae679c4 100644 --- a/gcc/config/rs6000/altivec.md +++ b/gcc/config/rs6000/altivec.md @@ -1152,15 +1152,16 @@ (use (match_operand:V16QI 2 "register_operand"))] "TARGET_ALTIVEC" { - rtx (*fun) (rtx, rtx, rtx) = BYTES_BIG_ENDIAN ? gen_altivec_vmrghb_direct - : gen_altivec_vmrglb_direct; - if (!BYTES_BIG_ENDIAN) - std::swap (operands[1], operands[2]); - emit_insn (fun (operands[0], operands[1], operands[2])); + if (BYTES_BIG_ENDIAN) + emit_insn ( + gen_altivec_vmrghb_direct_be (operands[0], operands[1], operands[2])); + else + emit_insn ( + gen_altivec_vmrglb_direct_le (operands[0], operands[2], operands[1])); DONE; }) -(define_insn "altivec_vmrghb_direct" +(define_insn "altivec_vmrghb_direct_be" [(set (match_operand:V16QI 0 "register_operand" "=v") (vec_select:V16QI (vec_concat:V32QI @@ -1174,7 +1175,25 @@ (const_int 5) (const_int 21) (const_int 6) (const_int 22) (const_int 7) (const_int 23)])))] - "TARGET_ALTIVEC" + "TARGET_ALTIVEC && BYTES_BIG_ENDIAN" + "vmrghb %0,%1,%2" + [(set_attr "type" "vecperm")]) + +(define_insn "altivec_vmrghb_direct_le" + [(set (match_operand:V16QI 0 "register_operand" "=v") + (vec_select:V16QI + (vec_concat:V32QI + (match_operand:V16QI 2 "register_operand" "v") + (match_operand:V16QI 1 "register_operand" "v")) + (parallel [(const_int 8) (const_int 24) + (const_int 9) (const_int 25) + (const_int 10) (const_int 26) + (const_int 11) (const_int 27) + (const_int 12) (const_int 28) + (const_int 13) (const_int 29) + (const_int 14) (const_int 30) + (const_int 15) (const_int 31)])))] + "TARGET_ALTIVEC && !BYTES_BIG_ENDIAN" "vmrghb %0,%1,%2" [(set_attr "type" "vecperm")]) @@ -1274,15 +1293,16 @@ (use (match_operand:V16QI 2 "register_operand"))] "TARGET_ALTIVEC" { - rtx (*fun) (rtx, rtx, rtx) = BYTES_BIG_ENDIAN ? gen_altivec_vmrglb_direct - : gen_altivec_vmrghb_direct; - if (!BYTES_BIG_ENDIAN) - std::swap (operands[1], operands[2]); - emit_insn (fun (operands[0], operands[1], operands[2])); + if (BYTES_BIG_ENDIAN) + emit_insn ( + gen_altivec_vmrglb_direct_be (operands[0], operands[1], operands[2])); + else + emit_insn ( + gen_altivec_vmrghb_direct_le (operands[0], operands[2], operands[1])); DONE; }) -(define_insn "altivec_vmrglb_direct" +(define_insn "altivec_vmrglb_direct_be" [(set (match_operand:V16QI 0 "register_operand" "=v") (vec_select:V16QI (vec_concat:V32QI @@ -1296,7 +1316,25 @@ (const_int 13) (const_int 29) (const_int 14) (const_int 30) (const_int 15) (const_int 31)])))] - "TARGET_ALTIVEC" + "TARGET_ALTIVEC && BYTES_BIG_ENDIAN" + "vmrglb %0,%1,%2" + [(set_attr "type" "vecperm")]) + +(define_insn "altivec_vmrglb_direct_le" + [(set (match_operand:V16QI 0 "register_operand" "=v") + (vec_select:V16QI + (vec_concat:V32QI + (match_operand:V16QI 2 "register_operand" "v") + (match_operand:V16QI 1 "register_operand" "v")) + (parallel [(const_int 0) (const_int 16) + (const_int 1) (const_int 17) + (const_int 2) (const_int 18) + (const_int 3) (const_int 19) + (const_int 4) (const_int 20) + (const_int 5) (const_int 21) + (const_int 6) (const_int 22) + (const_int 7) (const_int 23)])))] + "TARGET_ALTIVEC && !BYTES_BIG_ENDIAN" "vmrglb %0,%1,%2" [(set_attr "type" "vecperm")]) diff --git a/gcc/config/rs6000/rs6000.cc b/gcc/config/rs6000/rs6000.cc index 23b553131a9..e8ce629182b 100644 --- a/gcc/config/rs6000/rs6000.cc +++ b/gcc/config/rs6000/rs6000.cc @@ -22971,8 +22971,8 @@ altivec_expand_vec_perm_const (rtx target, rtx op0, rtx op1, CODE_FOR_altivec_vpkuwum_direct, {2, 3, 6, 7, 10, 11, 14, 15, 18, 19, 22, 23, 26, 27, 30, 31}}, {OPTION_MASK_ALTIVEC, - BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrghb_direct - : CODE_FOR_altivec_vmrglb_direct, + BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrghb_direct_be + : CODE_FOR_altivec_vmrglb_direct_le, {0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23}}, {OPTION_MASK_ALTIVEC, BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrghh_direct @@ -22983,8 +22983,8 @@ altivec_expand_vec_perm_const (rtx target, rtx op0, rtx op1, : CODE_FOR_altivec_vmrglw_direct_v4si_le, {0, 1, 2, 3, 16, 17, 18, 19, 4, 5, 6, 7, 20, 21, 22, 23}}, {OPTION_MASK_ALTIVEC, - BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrglb_direct - : CODE_FOR_altivec_vmrghb_direct, + BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrglb_direct_be + : CODE_FOR_altivec_vmrghb_direct_le, {8, 24, 9, 25, 10, 26, 11, 27, 12, 28, 13, 29, 14, 30, 15, 31}}, {OPTION_MASK_ALTIVEC, BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrglh_direct diff --git a/gcc/testsuite/gcc.target/powerpc/pr106069-1.c b/gcc/testsuite/gcc.target/powerpc/pr106069-1.c new file mode 100644 index 00000000000..4945d8fedfb --- /dev/null +++ b/gcc/testsuite/gcc.target/powerpc/pr106069-1.c @@ -0,0 +1,39 @@ +/* { dg-do run } */ +/* { dg-options "-O2" } */ +/* { dg-require-effective-target vmx_hw } */ + +/* Test vector merge for 8-bit element size, + it will abort if the RTL pattern isn't expected. */ + +#include "altivec.h" + +__attribute__((noipa)) +signed char elem_6 (vector signed char a, vector signed char b) +{ + vector signed char c = vec_mergeh (a,b); + return vec_extract (c, 6); +} + +__attribute__((noipa)) +unsigned char elem_15 (vector unsigned char a, vector unsigned char b) +{ + vector unsigned char c = vec_mergel (a,b); + return vec_extract (c, 15); +} + +int +main () +{ + vector unsigned char v1 + = {3, 33, 22, 12, 34, 14, 5, 25, 30, 11, 0, 21, 17, 27, 38, 8}; + vector unsigned char v2 + = {81, 82, 83, 84, 68, 67, 66, 65, 99, 100, 101, 102, 250, 125, 0, 6}; + signed char x1 = elem_6 ((vector signed char) v1, (vector signed char) v2); + unsigned char x2 = elem_15 (v1, v2); + + if (x1 != 12 || x2 != 6) + __builtin_abort (); + + return 0; +} + commit ca6eea0eb33de8b2e23e0bef3466575bb14ab63f Author: Kewen Lin Date: Wed Jun 26 02:16:17 2024 -0500 rs6000: Fix wrong RTL patterns for vector merge high/low short on LE Commit r12-4496 changes some define_expands and define_insns for vector merge high/low short, which are altivec_vmrg[hl]h. These defines are mainly for built-in function vec_merge{h,l} and some internal gen function needs. These functions should consider endianness, taking vec_mergeh as example, as PVIPR defines, vec_mergeh "Merges the first halves (in element order) of two vectors", it does note it's in element order. So it's mapped into vmrghh on BE while vmrglh on LE respectively. Although the mapped insns are different, as the discussion in PR106069, the RTL pattern should be still the same, it is conformed before commit r12-4496, but gets changed into different patterns on BE and LE starting from commit r12-4496. Similar to 32-bit element case in commit log of r15-1504, this 16-bit element pattern on LE doesn't actually match what the underlying insn is intended to represent, once some optimization like combine does some changes basing on it, it would cause the unexpected consequence. The newly constructed test case pr106069-2.c is a typical example for this issue on element type short. So this patch is to fix the wrong RTL pattern, ensure the associated RTL patterns become the same as before which can have the same semantic as their mapped insns. With the proposed patch, the expanders like altivec_vmrghh expands into altivec_vmrghh_direct_be or altivec_vmrglh_direct_le depending on endianness, "direct" can easily show which insn would be generated, _be and _le are mainly for the different RTL patterns as endianness. Co-authored-by: Xionghu Luo PR target/106069 PR target/115355 gcc/ChangeLog: * config/rs6000/altivec.md (altivec_vmrghh_direct): Rename to ... (altivec_vmrghh_direct_be): ... this. Add condition BYTES_BIG_ENDIAN. (altivec_vmrghh_direct_le): New define_insn. (altivec_vmrglh_direct): Rename to ... (altivec_vmrglh_direct_be): ... this. Add condition BYTES_BIG_ENDIAN. (altivec_vmrglh_direct_le): New define_insn. (altivec_vmrghh): Adjust by calling gen_altivec_vmrghh_direct_be for BE and gen_altivec_vmrglh_direct_le for LE. (altivec_vmrglh): Adjust by calling gen_altivec_vmrglh_direct_be for BE and gen_altivec_vmrghh_direct_le for LE. (vec_widen_umult_hi_v16qi): Adjust the call to gen_altivec_vmrghh_direct by gen_altivec_vmrghh for BE and by gen_altivec_vmrglh for LE. (vec_widen_smult_hi_v16qi): Likewise. (vec_widen_umult_lo_v16qi): Adjust the call to gen_altivec_vmrglh_direct by gen_altivec_vmrglh for BE and by gen_altivec_vmrghh for LE. (vec_widen_smult_lo_v16qi): Likewise. * config/rs6000/rs6000.cc (altivec_expand_vec_perm_const): Replace CODE_FOR_altivec_vmrghh_direct by CODE_FOR_altivec_vmrghh_direct_be for BE and CODE_FOR_altivec_vmrghh_direct_le for LE. And replace CODE_FOR_altivec_vmrglh_direct by CODE_FOR_altivec_vmrglh_direct_be for BE and CODE_FOR_altivec_vmrglh_direct_le for LE. gcc/testsuite/ChangeLog: * gcc.target/powerpc/pr106069-2.c: New test. (cherry picked from commit 812c70bf4981958488331d4ea5af8709b5321da1) diff --git a/gcc/config/rs6000/altivec.md b/gcc/config/rs6000/altivec.md index b8baae679c4..50689e418ed 100644 --- a/gcc/config/rs6000/altivec.md +++ b/gcc/config/rs6000/altivec.md @@ -1203,17 +1203,18 @@ (use (match_operand:V8HI 2 "register_operand"))] "TARGET_ALTIVEC" { - rtx (*fun) (rtx, rtx, rtx) = BYTES_BIG_ENDIAN ? gen_altivec_vmrghh_direct - : gen_altivec_vmrglh_direct; - if (!BYTES_BIG_ENDIAN) - std::swap (operands[1], operands[2]); - emit_insn (fun (operands[0], operands[1], operands[2])); + if (BYTES_BIG_ENDIAN) + emit_insn ( + gen_altivec_vmrghh_direct_be (operands[0], operands[1], operands[2])); + else + emit_insn ( + gen_altivec_vmrglh_direct_le (operands[0], operands[2], operands[1])); DONE; }) -(define_insn "altivec_vmrghh_direct" +(define_insn "altivec_vmrghh_direct_be" [(set (match_operand:V8HI 0 "register_operand" "=v") - (vec_select:V8HI + (vec_select:V8HI (vec_concat:V16HI (match_operand:V8HI 1 "register_operand" "v") (match_operand:V8HI 2 "register_operand" "v")) @@ -1221,7 +1222,21 @@ (const_int 1) (const_int 9) (const_int 2) (const_int 10) (const_int 3) (const_int 11)])))] - "TARGET_ALTIVEC" + "TARGET_ALTIVEC && BYTES_BIG_ENDIAN" + "vmrghh %0,%1,%2" + [(set_attr "type" "vecperm")]) + +(define_insn "altivec_vmrghh_direct_le" + [(set (match_operand:V8HI 0 "register_operand" "=v") + (vec_select:V8HI + (vec_concat:V16HI + (match_operand:V8HI 2 "register_operand" "v") + (match_operand:V8HI 1 "register_operand" "v")) + (parallel [(const_int 4) (const_int 12) + (const_int 5) (const_int 13) + (const_int 6) (const_int 14) + (const_int 7) (const_int 15)])))] + "TARGET_ALTIVEC && !BYTES_BIG_ENDIAN" "vmrghh %0,%1,%2" [(set_attr "type" "vecperm")]) @@ -1344,15 +1359,16 @@ (use (match_operand:V8HI 2 "register_operand"))] "TARGET_ALTIVEC" { - rtx (*fun) (rtx, rtx, rtx) = BYTES_BIG_ENDIAN ? gen_altivec_vmrglh_direct - : gen_altivec_vmrghh_direct; - if (!BYTES_BIG_ENDIAN) - std::swap (operands[1], operands[2]); - emit_insn (fun (operands[0], operands[1], operands[2])); + if (BYTES_BIG_ENDIAN) + emit_insn ( + gen_altivec_vmrglh_direct_be (operands[0], operands[1], operands[2])); + else + emit_insn ( + gen_altivec_vmrghh_direct_le (operands[0], operands[2], operands[1])); DONE; }) -(define_insn "altivec_vmrglh_direct" +(define_insn "altivec_vmrglh_direct_be" [(set (match_operand:V8HI 0 "register_operand" "=v") (vec_select:V8HI (vec_concat:V16HI @@ -1362,7 +1378,21 @@ (const_int 5) (const_int 13) (const_int 6) (const_int 14) (const_int 7) (const_int 15)])))] - "TARGET_ALTIVEC" + "TARGET_ALTIVEC && BYTES_BIG_ENDIAN" + "vmrglh %0,%1,%2" + [(set_attr "type" "vecperm")]) + +(define_insn "altivec_vmrglh_direct_le" + [(set (match_operand:V8HI 0 "register_operand" "=v") + (vec_select:V8HI + (vec_concat:V16HI + (match_operand:V8HI 2 "register_operand" "v") + (match_operand:V8HI 1 "register_operand" "v")) + (parallel [(const_int 0) (const_int 8) + (const_int 1) (const_int 9) + (const_int 2) (const_int 10) + (const_int 3) (const_int 11)])))] + "TARGET_ALTIVEC && !BYTES_BIG_ENDIAN" "vmrglh %0,%1,%2" [(set_attr "type" "vecperm")]) @@ -3777,13 +3807,13 @@ { emit_insn (gen_altivec_vmuleub (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmuloub (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrghh_direct (operands[0], ve, vo)); + emit_insn (gen_altivec_vmrghh (operands[0], ve, vo)); } else { emit_insn (gen_altivec_vmuloub (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmuleub (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrghh_direct (operands[0], vo, ve)); + emit_insn (gen_altivec_vmrglh (operands[0], ve, vo)); } DONE; }) @@ -3802,13 +3832,13 @@ { emit_insn (gen_altivec_vmuleub (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmuloub (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrglh_direct (operands[0], ve, vo)); + emit_insn (gen_altivec_vmrglh (operands[0], ve, vo)); } else { emit_insn (gen_altivec_vmuloub (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmuleub (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrglh_direct (operands[0], vo, ve)); + emit_insn (gen_altivec_vmrghh (operands[0], ve, vo)); } DONE; }) @@ -3827,13 +3857,13 @@ { emit_insn (gen_altivec_vmulesb (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmulosb (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrghh_direct (operands[0], ve, vo)); + emit_insn (gen_altivec_vmrghh (operands[0], ve, vo)); } else { emit_insn (gen_altivec_vmulosb (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmulesb (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrghh_direct (operands[0], vo, ve)); + emit_insn (gen_altivec_vmrglh (operands[0], ve, vo)); } DONE; }) @@ -3852,13 +3882,13 @@ { emit_insn (gen_altivec_vmulesb (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmulosb (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrglh_direct (operands[0], ve, vo)); + emit_insn (gen_altivec_vmrglh (operands[0], ve, vo)); } else { emit_insn (gen_altivec_vmulosb (ve, operands[1], operands[2])); emit_insn (gen_altivec_vmulesb (vo, operands[1], operands[2])); - emit_insn (gen_altivec_vmrglh_direct (operands[0], vo, ve)); + emit_insn (gen_altivec_vmrghh (operands[0], ve, vo)); } DONE; }) diff --git a/gcc/config/rs6000/rs6000.cc b/gcc/config/rs6000/rs6000.cc index e8ce629182b..34be43c9f84 100644 --- a/gcc/config/rs6000/rs6000.cc +++ b/gcc/config/rs6000/rs6000.cc @@ -22975,8 +22975,8 @@ altivec_expand_vec_perm_const (rtx target, rtx op0, rtx op1, : CODE_FOR_altivec_vmrglb_direct_le, {0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23}}, {OPTION_MASK_ALTIVEC, - BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrghh_direct - : CODE_FOR_altivec_vmrglh_direct, + BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrghh_direct_be + : CODE_FOR_altivec_vmrglh_direct_le, {0, 1, 16, 17, 2, 3, 18, 19, 4, 5, 20, 21, 6, 7, 22, 23}}, {OPTION_MASK_ALTIVEC, BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrghw_direct_v4si_be @@ -22987,8 +22987,8 @@ altivec_expand_vec_perm_const (rtx target, rtx op0, rtx op1, : CODE_FOR_altivec_vmrghb_direct_le, {8, 24, 9, 25, 10, 26, 11, 27, 12, 28, 13, 29, 14, 30, 15, 31}}, {OPTION_MASK_ALTIVEC, - BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrglh_direct - : CODE_FOR_altivec_vmrghh_direct, + BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrglh_direct_be + : CODE_FOR_altivec_vmrghh_direct_le, {8, 9, 24, 25, 10, 11, 26, 27, 12, 13, 28, 29, 14, 15, 30, 31}}, {OPTION_MASK_ALTIVEC, BYTES_BIG_ENDIAN ? CODE_FOR_altivec_vmrglw_direct_v4si_be diff --git a/gcc/testsuite/gcc.target/powerpc/pr106069-2.c b/gcc/testsuite/gcc.target/powerpc/pr106069-2.c new file mode 100644 index 00000000000..283e3290fb3 --- /dev/null +++ b/gcc/testsuite/gcc.target/powerpc/pr106069-2.c @@ -0,0 +1,37 @@ +/* { dg-do run } */ +/* { dg-options "-O2" } */ +/* { dg-require-effective-target vmx_hw } */ + +/* Test vector merge for 16-bit element size, + it will abort if the RTL pattern isn't expected. */ + +#include "altivec.h" + +__attribute__((noipa)) +signed short elem_2 (vector signed short a, vector signed short b) +{ + vector signed short c = vec_mergeh (a,b); + return vec_extract (c, 2); +} + +__attribute__((noipa)) +unsigned short elem_7 (vector unsigned short a, vector unsigned short b) +{ + vector unsigned short c = vec_mergel (a,b); + return vec_extract (c, 7); +} + +int +main () +{ + vector unsigned short v1 = {3, 22, 12, 34, 5, 25, 30, 11}; + vector unsigned short v2 = {84, 168, 267, 966, 65, 399, 999, 99}; + signed short x1 = elem_2 ((vector signed short) v1, (vector signed short) v2); + unsigned short x2 = elem_7 (v1, v2); + + if (x1 != 22 || x2 != 99) + __builtin_abort (); + + return 0; +} + commit 5f699cb08eed44a903393f601009e9c6d0b59c59 Author: Georg-Johann Lay Date: Wed Jul 3 10:29:18 2024 +0200 AVR: target/98762 - Handle partial clobber in movqi output. PR target/98762 gcc/ * config/avr/avr.cc (avr_out_movqi_r_mr_reg_disp_tiny): Properly restore the base register when it is partially clobbered. gcc/testsuite/ * gcc.target/avr/torture/pr98762.c: New test. (cherry picked from commit e9fb6efa1cf542353fd44ddcbb5136344c463fd0) diff --git a/gcc/config/avr/avr.cc b/gcc/config/avr/avr.cc index ee033d3204d..f355146f992 100644 --- a/gcc/config/avr/avr.cc +++ b/gcc/config/avr/avr.cc @@ -4223,13 +4223,30 @@ avr_out_movqi_r_mr_reg_disp_tiny (rtx_insn *insn, rtx op[], int *plen) rtx dest = op[0]; rtx src = op[1]; rtx x = XEXP (src, 0); + rtx base = XEXP (x, 0); - avr_asm_len (TINY_ADIW (%I1, %J1, %o1) CR_TAB - "ld %0,%b1" , op, plen, -3); + if (plen) + *plen = 0; - if (!reg_overlap_mentioned_p (dest, XEXP (x, 0)) - && !reg_unused_after (insn, XEXP (x, 0))) - avr_asm_len (TINY_SBIW (%I1, %J1, %o1), op, plen, 2); + if (!reg_overlap_mentioned_p (dest, base)) + { + avr_asm_len (TINY_ADIW (%I1, %J1, %o1) CR_TAB + "ld %0,%b1", op, plen, 3); + if (!reg_unused_after (insn, base)) + avr_asm_len (TINY_SBIW (%I1, %J1, %o1), op, plen, 2); + } + else + { + // PR98762: The base register overlaps dest and is only partly clobbered. + rtx base2 = all_regs_rtx[1 ^ REGNO (dest)]; + + if (!reg_unused_after (insn, base2)) + avr_asm_len ("mov __tmp_reg__,%0" , &base2, plen, 1); + avr_asm_len (TINY_ADIW (%I1, %J1, %o1) CR_TAB + "ld %0,%b1", op, plen, 3); + if (!reg_unused_after (insn, base2)) + avr_asm_len ("mov %0,__tmp_reg__" , &base2, plen, 1); + } return ""; } diff --git a/gcc/testsuite/gcc.target/avr/torture/pr98762.c b/gcc/testsuite/gcc.target/avr/torture/pr98762.c new file mode 100644 index 00000000000..c3ba7da69a8 --- /dev/null +++ b/gcc/testsuite/gcc.target/avr/torture/pr98762.c @@ -0,0 +1,19 @@ +/* { dg-do run } */ +/* { dg-additional-options "-std=c99" } */ + +long long acc = 0x1122334455667788; + +__attribute__((noinline,noclone)) +void addhi (short a) +{ + acc += (long long) a << 32; +} + +int main (void) +{ + addhi (0x0304); + if (acc != 0x1122364855667788) + __builtin_abort(); + + return 0; +} commit 0c98d9479cec88148eb3be8d0098e36bce061cd6 Author: John David Anglin Date: Sun Jun 30 09:48:21 2024 -0400 hppa: Fix ICE caused by mismatched predicate and constraint in xmpyu patterns 2024-06-30 John David Anglin gcc/ChangeLog: PR target/115691 * config/pa/pa.md: Remove incorrect xmpyu patterns. diff --git a/gcc/config/pa/pa.md b/gcc/config/pa/pa.md index d82f168c8a3..43241958722 100644 --- a/gcc/config/pa/pa.md +++ b/gcc/config/pa/pa.md @@ -5493,24 +5493,6 @@ [(set_attr "type" "fpmuldbl") (set_attr "length" "4")]) -(define_insn "" - [(set (match_operand:DI 0 "register_operand" "=f") - (mult:DI (zero_extend:DI (match_operand:SI 1 "register_operand" "f")) - (match_operand:DI 2 "uint32_operand" "f")))] - "TARGET_PA_11 && ! TARGET_SOFT_FLOAT && ! TARGET_SOFT_MULT && !TARGET_64BIT" - "xmpyu %1,%R2,%0" - [(set_attr "type" "fpmuldbl") - (set_attr "length" "4")]) - -(define_insn "" - [(set (match_operand:DI 0 "register_operand" "=f") - (mult:DI (zero_extend:DI (match_operand:SI 1 "register_operand" "f")) - (match_operand:DI 2 "uint32_operand" "f")))] - "TARGET_PA_11 && ! TARGET_SOFT_FLOAT && ! TARGET_SOFT_MULT && TARGET_64BIT" - "xmpyu %1,%2R,%0" - [(set_attr "type" "fpmuldbl") - (set_attr "length" "4")]) - (define_insn "" [(set (reg:SI 29) (mult:SI (reg:SI 26) (reg:SI 25))) (clobber (match_operand:SI 0 "register_operand" "=a")) commit 843eba6f38d3debc44fe4ec4084aaaf9d5907485 Author: GCC Administrator Date: Thu Jul 4 00:21:22 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 45bf020f832..d002973771d 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,77 @@ +2024-07-03 John David Anglin + + PR target/115691 + * config/pa/pa.md: Remove incorrect xmpyu patterns. + +2024-07-03 Georg-Johann Lay + + Backported from master: + 2024-07-03 Georg-Johann Lay + + PR target/98762 + * config/avr/avr.cc (avr_out_movqi_r_mr_reg_disp_tiny): Properly + restore the base register when it is partially clobbered. + +2024-07-03 Kewen Lin + + Backported from master: + 2024-06-26 Kewen Lin + Xionghu Luo + + PR target/106069 + PR target/115355 + * config/rs6000/altivec.md (altivec_vmrghh_direct): Rename to ... + (altivec_vmrghh_direct_be): ... this. Add condition BYTES_BIG_ENDIAN. + (altivec_vmrghh_direct_le): New define_insn. + (altivec_vmrglh_direct): Rename to ... + (altivec_vmrglh_direct_be): ... this. Add condition BYTES_BIG_ENDIAN. + (altivec_vmrglh_direct_le): New define_insn. + (altivec_vmrghh): Adjust by calling gen_altivec_vmrghh_direct_be + for BE and gen_altivec_vmrglh_direct_le for LE. + (altivec_vmrglh): Adjust by calling gen_altivec_vmrglh_direct_be + for BE and gen_altivec_vmrghh_direct_le for LE. + (vec_widen_umult_hi_v16qi): Adjust the call to + gen_altivec_vmrghh_direct by gen_altivec_vmrghh for BE + and by gen_altivec_vmrglh for LE. + (vec_widen_smult_hi_v16qi): Likewise. + (vec_widen_umult_lo_v16qi): Adjust the call to + gen_altivec_vmrglh_direct by gen_altivec_vmrglh for BE + and by gen_altivec_vmrghh for LE. + (vec_widen_smult_lo_v16qi): Likewise. + * config/rs6000/rs6000.cc (altivec_expand_vec_perm_const): Replace + CODE_FOR_altivec_vmrghh_direct by + CODE_FOR_altivec_vmrghh_direct_be for BE and + CODE_FOR_altivec_vmrghh_direct_le for LE. And replace + CODE_FOR_altivec_vmrglh_direct by + CODE_FOR_altivec_vmrglh_direct_be for BE and + CODE_FOR_altivec_vmrglh_direct_le for LE. + +2024-07-03 Kewen Lin + + Backported from master: + 2024-06-26 Kewen Lin + Xionghu Luo + + PR target/106069 + PR target/115355 + * config/rs6000/altivec.md (altivec_vmrghb_direct): Rename to ... + (altivec_vmrghb_direct_be): ... this. Add condition BYTES_BIG_ENDIAN. + (altivec_vmrghb_direct_le): New define_insn. + (altivec_vmrglb_direct): Rename to ... + (altivec_vmrglb_direct_be): ... this. Add condition BYTES_BIG_ENDIAN. + (altivec_vmrglb_direct_le): New define_insn. + (altivec_vmrghb): Adjust by calling gen_altivec_vmrghb_direct_be + for BE and gen_altivec_vmrglb_direct_le for LE. + (altivec_vmrglb): Adjust by calling gen_altivec_vmrglb_direct_be + for BE and gen_altivec_vmrghb_direct_le for LE. + * config/rs6000/rs6000.cc (altivec_expand_vec_perm_const): Replace + CODE_FOR_altivec_vmrghb_direct by + CODE_FOR_altivec_vmrghb_direct_be for BE and + CODE_FOR_altivec_vmrghb_direct_le for LE. And replace + CODE_FOR_altivec_vmrglb_direct by + CODE_FOR_altivec_vmrglb_direct_be for BE and + CODE_FOR_altivec_vmrglb_direct_le for LE. + 2024-07-01 Georg-Johann Lay Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index f8fa5e4aa67..efaf824b8cd 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240703 +20240704 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 60449edd77b..ede549333cc 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,31 @@ +2024-07-03 Georg-Johann Lay + + Backported from master: + 2024-07-03 Georg-Johann Lay + + PR target/98762 + * gcc.target/avr/torture/pr98762.c: New test. + +2024-07-03 Kewen Lin + + Backported from master: + 2024-06-26 Kewen Lin + Xionghu Luo + + PR target/106069 + PR target/115355 + * gcc.target/powerpc/pr106069-2.c: New test. + +2024-07-03 Kewen Lin + + Backported from master: + 2024-06-26 Kewen Lin + Xionghu Luo + + PR target/106069 + PR target/115355 + * gcc.target/powerpc/pr106069-1.c: New test. + 2024-07-01 Georg-Johann Lay Backported from master: commit ebf561429ee4fbd125aa51ee985e32f1cfd4daed Author: Kyrylo Tkachov Date: Thu Jun 27 16:10:41 2024 +0530 aarch64: PR target/115457 Implement missing __ARM_FEATURE_BF16 macro The ACLE asks the user to test for __ARM_FEATURE_BF16 before using the header but GCC doesn't set this up. LLVM does, so this is an inconsistency between the compilers. This patch enables that macro for TARGET_BF16_FP. Bootstrapped and tested on aarch64-none-linux-gnu. gcc/ PR target/115457 * config/aarch64/aarch64-c.cc (aarch64_update_cpp_builtins): Define __ARM_FEATURE_BF16 for TARGET_BF16_FP. gcc/testsuite/ PR target/115457 * gcc.target/aarch64/acle/bf16_feature.c: New test. Signed-off-by: Kyrylo Tkachov (cherry picked from commit c10942134fa759843ac1ed1424b86fcb8e6368ba) diff --git a/gcc/config/aarch64/aarch64-c.cc b/gcc/config/aarch64/aarch64-c.cc index a4c407724a7..b31b967c140 100644 --- a/gcc/config/aarch64/aarch64-c.cc +++ b/gcc/config/aarch64/aarch64-c.cc @@ -200,6 +200,8 @@ aarch64_update_cpp_builtins (cpp_reader *pfile) "__ARM_FEATURE_BF16_VECTOR_ARITHMETIC", pfile); aarch64_def_or_undef (TARGET_BF16_FP, "__ARM_FEATURE_BF16_SCALAR_ARITHMETIC", pfile); + aarch64_def_or_undef (TARGET_BF16_FP, + "__ARM_FEATURE_BF16", pfile); aarch64_def_or_undef (TARGET_LS64, "__ARM_FEATURE_LS64", pfile); aarch64_def_or_undef (AARCH64_ISA_RCPC, "__ARM_FEATURE_RCPC", pfile); diff --git a/gcc/testsuite/gcc.target/aarch64/acle/bf16_feature.c b/gcc/testsuite/gcc.target/aarch64/acle/bf16_feature.c new file mode 100644 index 00000000000..96584b4b988 --- /dev/null +++ b/gcc/testsuite/gcc.target/aarch64/acle/bf16_feature.c @@ -0,0 +1,10 @@ +/* { dg-do compile } */ + +#pragma GCC target "+bf16" +#ifndef __ARM_FEATURE_BF16 +#error "__ARM_FEATURE_BF16 is not defined but should be!" +#endif + +void +foo (void) {} + commit cdeb7ce83f71d1527626975e70d294ef55535d03 Author: Kyrylo Tkachov Date: Fri Jun 28 13:22:37 2024 +0530 aarch64: PR target/115475 Implement missing __ARM_FEATURE_SVE_BF16 macro The ACLE requires __ARM_FEATURE_SVE_BF16 to be enabled when SVE and BF16 and the associated intrinsics are available. GCC does support the required intrinsics for TARGET_SVE_BF16 so define this macro too. Bootstrapped and tested on aarch64-none-linux-gnu. gcc/ PR target/115475 * config/aarch64/aarch64-c.cc (aarch64_update_cpp_builtins): Define __ARM_FEATURE_SVE_BF16 for TARGET_SVE_BF16. gcc/testsuite/ PR target/115475 * gcc.target/aarch64/acle/bf16_sve_feature.c: New test. Signed-off-by: Kyrylo Tkachov (cherry picked from commit 6492c7130d6ae9992298fc3d072e2589d1131376) diff --git a/gcc/config/aarch64/aarch64-c.cc b/gcc/config/aarch64/aarch64-c.cc index b31b967c140..e024d410dc7 100644 --- a/gcc/config/aarch64/aarch64-c.cc +++ b/gcc/config/aarch64/aarch64-c.cc @@ -202,6 +202,9 @@ aarch64_update_cpp_builtins (cpp_reader *pfile) "__ARM_FEATURE_BF16_SCALAR_ARITHMETIC", pfile); aarch64_def_or_undef (TARGET_BF16_FP, "__ARM_FEATURE_BF16", pfile); + aarch64_def_or_undef (TARGET_SVE_BF16, + "__ARM_FEATURE_SVE_BF16", pfile); + aarch64_def_or_undef (TARGET_LS64, "__ARM_FEATURE_LS64", pfile); aarch64_def_or_undef (AARCH64_ISA_RCPC, "__ARM_FEATURE_RCPC", pfile); diff --git a/gcc/testsuite/gcc.target/aarch64/acle/bf16_sve_feature.c b/gcc/testsuite/gcc.target/aarch64/acle/bf16_sve_feature.c new file mode 100644 index 00000000000..cb3ddac71a3 --- /dev/null +++ b/gcc/testsuite/gcc.target/aarch64/acle/bf16_sve_feature.c @@ -0,0 +1,10 @@ +/* { dg-do compile } */ + +#pragma GCC target "+sve+bf16" +#ifndef __ARM_FEATURE_SVE_BF16 +#error "__ARM_FEATURE_SVE_BF16 is not defined but should be!" +#endif + +void +foo (void) {} + commit 92a6ff14c4d725beb8b36295b6720f182e4bffbc Author: GCC Administrator Date: Fri Jul 5 00:20:24 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index d002973771d..335ff1bc253 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,21 @@ +2024-07-04 Kyrylo Tkachov + + Backported from master: + 2024-07-03 Kyrylo Tkachov + + PR target/115475 + * config/aarch64/aarch64-c.cc (aarch64_update_cpp_builtins): + Define __ARM_FEATURE_SVE_BF16 for TARGET_SVE_BF16. + +2024-07-04 Kyrylo Tkachov + + Backported from master: + 2024-07-03 Kyrylo Tkachov + + PR target/115457 + * config/aarch64/aarch64-c.cc (aarch64_update_cpp_builtins): + Define __ARM_FEATURE_BF16 for TARGET_BF16_FP. + 2024-07-03 John David Anglin PR target/115691 diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index efaf824b8cd..90cf77743f7 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240704 +20240705 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index ede549333cc..e1ef55580ce 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,19 @@ +2024-07-04 Kyrylo Tkachov + + Backported from master: + 2024-07-03 Kyrylo Tkachov + + PR target/115475 + * gcc.target/aarch64/acle/bf16_sve_feature.c: New test. + +2024-07-04 Kyrylo Tkachov + + Backported from master: + 2024-07-03 Kyrylo Tkachov + + PR target/115457 + * gcc.target/aarch64/acle/bf16_feature.c: New test. + 2024-07-03 Georg-Johann Lay Backported from master: commit 5f4a60c43d5cd805add6529b4528c35893c283ae Author: Georg-Johann Lay Date: Fri Jul 5 13:22:12 2024 +0200 AVR: target/87376 - Use nop_general_operand for DImode inputs. The avr-dimode.md expanders have code like emit_move_insn(acc_a, operands[1]) where acc_a is a hard register and operands[1] might be a non-generic address-space memory reference. Such loads may clobber hard regs since some of them are implemented as libgcc calls /and/ 64-moves are expanded as eight byte-moves, so that acc_a or acc_b might be clobbered by such a load. This patch simply denies non-generic address-space references by using nop_general_operand for all avr-dimode.md input predicates. With the patch, all memory loads that require library calls are issued before the expander codes from avr-dimode.md are run. PR target/87376 gcc/ * config/avr/avr-dimode.md: Use "nop_general_operand" instead of "general_operand" as predicate for all input operands. gcc/testsuite/ * gcc.target/avr/torture/pr87376.c: New test. (cherry picked from commit 23a0935262d6817097406578b1c70563f424804b) diff --git a/gcc/config/avr/avr-dimode.md b/gcc/config/avr/avr-dimode.md index 28e97da0516..78a7a745d34 100644 --- a/gcc/config/avr/avr-dimode.md +++ b/gcc/config/avr/avr-dimode.md @@ -62,8 +62,8 @@ ;; "addta3" "adduta3" (define_expand "add3" [(parallel [(match_operand:ALL8 0 "general_operand" "") - (match_operand:ALL8 1 "general_operand" "") - (match_operand:ALL8 2 "general_operand" "")])] + (match_operand:ALL8 1 "nop_general_operand") + (match_operand:ALL8 2 "nop_general_operand")])] "avr_have_dimode" { rtx acc_a = gen_rtx_REG (mode, ACC_A); @@ -178,8 +178,8 @@ ;; "subta3" "subuta3" (define_expand "sub3" [(parallel [(match_operand:ALL8 0 "general_operand" "") - (match_operand:ALL8 1 "general_operand" "") - (match_operand:ALL8 2 "general_operand" "")])] + (match_operand:ALL8 1 "nop_general_operand") + (match_operand:ALL8 2 "nop_general_operand")])] "avr_have_dimode" { rtx acc_a = gen_rtx_REG (mode, ACC_A); @@ -259,8 +259,8 @@ (define_expand "3" [(set (match_operand:ALL8S 0 "general_operand" "") - (ss_addsub:ALL8S (match_operand:ALL8S 1 "general_operand" "") - (match_operand:ALL8S 2 "general_operand" "")))] + (ss_addsub:ALL8S (match_operand:ALL8S 1 "nop_general_operand") + (match_operand:ALL8S 2 "nop_general_operand")))] "avr_have_dimode" { rtx acc_a = gen_rtx_REG (mode, ACC_A); @@ -332,8 +332,8 @@ (define_expand "3" [(set (match_operand:ALL8U 0 "general_operand" "") - (us_addsub:ALL8U (match_operand:ALL8U 1 "general_operand" "") - (match_operand:ALL8U 2 "general_operand" "")))] + (us_addsub:ALL8U (match_operand:ALL8U 1 "nop_general_operand") + (match_operand:ALL8U 2 "nop_general_operand")))] "avr_have_dimode" { rtx acc_a = gen_rtx_REG (mode, ACC_A); @@ -405,7 +405,7 @@ (define_expand "negdi2" [(parallel [(match_operand:DI 0 "general_operand" "") - (match_operand:DI 1 "general_operand" "")])] + (match_operand:DI 1 "nop_general_operand")])] "avr_have_dimode" { rtx acc_a = gen_rtx_REG (DImode, ACC_A); @@ -602,8 +602,8 @@ ;; "ashluta3" "ashruta3" "lshruta3" "rotluta3" (define_expand "3" [(parallel [(match_operand:ALL8 0 "general_operand" "") - (di_shifts:ALL8 (match_operand:ALL8 1 "general_operand" "") - (match_operand:QI 2 "general_operand" ""))])] + (di_shifts:ALL8 (match_operand:ALL8 1 "nop_general_operand") + (match_operand:QI 2 "nop_general_operand"))])] "avr_have_dimode" { rtx acc_a = gen_rtx_REG (mode, ACC_A); @@ -648,8 +648,8 @@ ;; "mulsidi3" (define_expand "mulsidi3" [(parallel [(match_operand:DI 0 "register_operand" "") - (match_operand:SI 1 "general_operand" "") - (match_operand:SI 2 "general_operand" "") + (match_operand:SI 1 "nop_general_operand") + (match_operand:SI 2 "nop_general_operand") ;; Just to mention the iterator (clobber (any_extend:SI (match_dup 1)))])] "avr_have_dimode diff --git a/gcc/testsuite/gcc.target/avr/torture/pr87376.c b/gcc/testsuite/gcc.target/avr/torture/pr87376.c new file mode 100644 index 00000000000..c31a4a9dda5 --- /dev/null +++ b/gcc/testsuite/gcc.target/avr/torture/pr87376.c @@ -0,0 +1,60 @@ +/* { dg-do run { target { ! avr_tiny } } } */ +/* { dg-additional-options "-std=gnu99" } */ + +typedef __UINT64_TYPE__ uint64_t; + +extern const __memx uint64_t aa __asm ("real_aa"); +extern const uint64_t bb __asm ("real_bb"); + +const __memx uint64_t real_aa = 0x1122334455667788; +const uint64_t real_bb = 0x0908070605040302; + +__attribute__((noinline,noclone)) +uint64_t add1 (void) +{ + return aa + bb; +} + +__attribute__((noinline,noclone)) +uint64_t add2 (void) +{ + return bb + aa; +} + +__attribute__((noinline,noclone)) +uint64_t sub1 (void) +{ + return aa - bb; +} + +__attribute__((noinline,noclone)) +uint64_t sub2 (void) +{ + return bb - aa; +} + +__attribute__((noinline,noclone)) +uint64_t neg1 (void) +{ + return -aa; +} + +int main (void) +{ + if (neg1() != -real_aa) + __builtin_exit (__LINE__); + + if (add1() != real_aa + real_bb) + __builtin_exit (__LINE__); + + if (add2() != real_bb + real_aa) + __builtin_exit (__LINE__); + + if (sub1() != real_aa - real_bb) + __builtin_exit (__LINE__); + + if (sub2() != real_bb - real_aa) + __builtin_exit (__LINE__); + + return 0; +} commit b9d16d8361a9e3a82a2f21e759e760d235d43322 Author: Wilco Dijkstra Date: Wed Oct 25 16:28:04 2023 +0100 AArch64: Fix strict-align cpymem/setmem [PR103100] The cpymemdi/setmemdi implementation doesn't fully support strict alignment. Block the expansion if the alignment is less than 16 with STRICT_ALIGNMENT. Clean up the condition when to use MOPS. gcc/ChangeLog/ PR target/103100 * config/aarch64/aarch64.md (cpymemdi): Remove pattern condition. (setmemdi): Likewise. * config/aarch64/aarch64.cc (aarch64_expand_cpymem): Support strict-align. Cleanup condition for using MOPS. (aarch64_expand_setmem): Likewise. (cherry picked from commit 318f5232cfb3e0c9694889565e1f5424d0354463) diff --git a/gcc/config/aarch64/aarch64.cc b/gcc/config/aarch64/aarch64.cc index f8082c4035e..cd2f4053a1a 100644 --- a/gcc/config/aarch64/aarch64.cc +++ b/gcc/config/aarch64/aarch64.cc @@ -24782,27 +24782,23 @@ aarch64_expand_cpymem (rtx *operands) int mode_bits; rtx dst = operands[0]; rtx src = operands[1]; + unsigned align = UINTVAL (operands[3]); rtx base; machine_mode cur_mode = BLKmode; + bool size_p = optimize_function_for_size_p (cfun); - /* Variable-sized memcpy can go through the MOPS expansion if available. */ - if (!CONST_INT_P (operands[2])) + /* Variable-sized or strict-align copies may use the MOPS expansion. */ + if (!CONST_INT_P (operands[2]) || (STRICT_ALIGNMENT && align < 16)) return aarch64_expand_cpymem_mops (operands); - unsigned HOST_WIDE_INT size = INTVAL (operands[2]); - - /* Try to inline up to 256 bytes or use the MOPS threshold if available. */ - unsigned HOST_WIDE_INT max_copy_size - = TARGET_MOPS ? aarch64_mops_memcpy_size_threshold : 256; + unsigned HOST_WIDE_INT size = UINTVAL (operands[2]); - bool size_p = optimize_function_for_size_p (cfun); + /* Try to inline up to 256 bytes. */ + unsigned max_copy_size = 256; + unsigned mops_threshold = aarch64_mops_memcpy_size_threshold; - /* Large constant-sized cpymem should go through MOPS when possible. - It should be a win even for size optimization in the general case. - For speed optimization the choice between MOPS and the SIMD sequence - depends on the size of the copy, rather than number of instructions, - alignment etc. */ - if (size > max_copy_size) + /* Large copies use MOPS when available or a library call. */ + if (size > max_copy_size || (TARGET_MOPS && size > mops_threshold)) return aarch64_expand_cpymem_mops (operands); int copy_bits = 256; @@ -24966,12 +24962,13 @@ aarch64_expand_setmem (rtx *operands) unsigned HOST_WIDE_INT len; rtx dst = operands[0]; rtx val = operands[2], src; + unsigned align = UINTVAL (operands[3]); rtx base; machine_mode cur_mode = BLKmode, next_mode; - /* If we don't have SIMD registers or the size is variable use the MOPS - inlined sequence if possible. */ - if (!CONST_INT_P (operands[1]) || !TARGET_SIMD) + /* Variable-sized or strict-align memset may use the MOPS expansion. */ + if (!CONST_INT_P (operands[1]) || !TARGET_SIMD + || (STRICT_ALIGNMENT && align < 16)) return aarch64_expand_setmem_mops (operands); bool size_p = optimize_function_for_size_p (cfun); @@ -24979,10 +24976,13 @@ aarch64_expand_setmem (rtx *operands) /* Default the maximum to 256-bytes when considering only libcall vs SIMD broadcast sequence. */ unsigned max_set_size = 256; + unsigned mops_threshold = aarch64_mops_memset_size_threshold; - len = INTVAL (operands[1]); - if (len > max_set_size && !TARGET_MOPS) - return false; + len = UINTVAL (operands[1]); + + /* Large memset uses MOPS when available or a library call. */ + if (len > max_set_size || (TARGET_MOPS && len > mops_threshold)) + return aarch64_expand_setmem_mops (operands); int cst_val = !!(CONST_INT_P (val) && (INTVAL (val) != 0)); /* The MOPS sequence takes: @@ -24995,12 +24995,6 @@ aarch64_expand_setmem (rtx *operands) the arguments + 1 for the call. */ unsigned libcall_cost = 4; - /* Upper bound check. For large constant-sized setmem use the MOPS sequence - when available. */ - if (TARGET_MOPS - && len >= (unsigned HOST_WIDE_INT) aarch64_mops_memset_size_threshold) - return aarch64_expand_setmem_mops (operands); - /* Attempt a sequence with a vector broadcast followed by stores. Count the number of operations involved to see if it's worth it against the alternatives. A simple counter simd_ops on the @@ -25042,10 +25036,8 @@ aarch64_expand_setmem (rtx *operands) simd_ops++; n -= mode_bits; - /* Do certain trailing copies as overlapping if it's going to be - cheaper. i.e. less instructions to do so. For instance doing a 15 - byte copy it's more efficient to do two overlapping 8 byte copies than - 8 + 4 + 2 + 1. Only do this when -mstrict-align is not supplied. */ + /* Emit trailing writes using overlapping unaligned accesses + (when !STRICT_ALIGNMENT) - this is smaller and faster. */ if (n > 0 && n < copy_limit / 2 && !STRICT_ALIGNMENT) { next_mode = smallest_mode_for_size (n, MODE_INT); diff --git a/gcc/config/aarch64/aarch64.md b/gcc/config/aarch64/aarch64.md index 99f185718c9..47b70feff02 100644 --- a/gcc/config/aarch64/aarch64.md +++ b/gcc/config/aarch64/aarch64.md @@ -1637,7 +1637,7 @@ (match_operand:BLK 1 "memory_operand") (match_operand:DI 2 "general_operand") (match_operand:DI 3 "immediate_operand")] - "!STRICT_ALIGNMENT || TARGET_MOPS" + "" { if (aarch64_expand_cpymem (operands)) DONE; @@ -1734,7 +1734,7 @@ (match_operand:QI 2 "nonmemory_operand")) ;; Value (use (match_operand:DI 1 "general_operand")) ;; Length (match_operand 3 "immediate_operand")] ;; Align - "TARGET_SIMD || TARGET_MOPS" + "" { if (aarch64_expand_setmem (operands)) DONE; commit d85d7ce0ad98e62104aef8da40a86b44a0dbe231 Author: GCC Administrator Date: Sat Jul 6 00:20:30 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 335ff1bc253..d66f16bdb5e 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,24 @@ +2024-07-05 Wilco Dijkstra + + Backported from master: + 2023-11-30 Wilco Dijkstra + + PR target/103100 + * config/aarch64/aarch64.md (cpymemdi): Remove pattern condition. + (setmemdi): Likewise. + * config/aarch64/aarch64.cc (aarch64_expand_cpymem): Support + strict-align. Cleanup condition for using MOPS. + (aarch64_expand_setmem): Likewise. + +2024-07-05 Georg-Johann Lay + + Backported from master: + 2024-07-05 Georg-Johann Lay + + PR target/87376 + * config/avr/avr-dimode.md: Use "nop_general_operand" instead + of "general_operand" as predicate for all input operands. + 2024-07-04 Kyrylo Tkachov Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 90cf77743f7..d97996cc287 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240705 +20240706 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index e1ef55580ce..97e6e8bc00d 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-07-05 Georg-Johann Lay + + Backported from master: + 2024-07-05 Georg-Johann Lay + + PR target/87376 + * gcc.target/avr/torture/pr87376.c: New test. + 2024-07-04 Kyrylo Tkachov Backported from master: commit 1855da033f9955c36e29dce526b28bd36fe8e384 Author: GCC Administrator Date: Sun Jul 7 00:19:56 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d97996cc287..391dc0be055 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240706 +20240707 commit 69520d77fae798338ee849b42e1a61b08ed27275 Author: GCC Administrator Date: Mon Jul 8 00:20:02 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 391dc0be055..d50aabd12dd 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240707 +20240708 commit fe82e2ae166a6ef3dc2efb3b2fcd030b135142da Author: GCC Administrator Date: Tue Jul 9 00:20:40 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d50aabd12dd..b01fad6bd4f 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240708 +20240709 commit 987e70f4290451abd98eda8b82e97b4ad25ee3c2 Author: Jonathan Wakely Date: Sun Jul 7 12:22:42 2024 +0100 libstdc++: Fix _Atomic(T) macro in [PR115807] The definition of the _Atomic(T) macro needs to refer to ::std::atomic, not some other std::atomic relative to the current namespace. libstdc++-v3/ChangeLog: PR libstdc++/115807 * include/c_compatibility/stdatomic.h (_Atomic): Ensure it refers to std::atomic in the global namespace. * testsuite/29_atomics/headers/stdatomic.h/115807.cc: New test. (cherry picked from commit 40d234dd6439e8c8cfbf3f375a61906aed35c80d) diff --git a/libstdc++-v3/include/c_compatibility/stdatomic.h b/libstdc++-v3/include/c_compatibility/stdatomic.h index b565a1c1ab1..2bb23decf8d 100644 --- a/libstdc++-v3/include/c_compatibility/stdatomic.h +++ b/libstdc++-v3/include/c_compatibility/stdatomic.h @@ -34,7 +34,7 @@ #define __cpp_lib_stdatomic_h 202011L -#define _Atomic(_Tp) std::atomic<_Tp> +#define _Atomic(_Tp) ::std::atomic<_Tp> using std::memory_order; using std::memory_order_relaxed; diff --git a/libstdc++-v3/testsuite/29_atomics/headers/stdatomic.h/115807.cc b/libstdc++-v3/testsuite/29_atomics/headers/stdatomic.h/115807.cc new file mode 100644 index 00000000000..14f320fe835 --- /dev/null +++ b/libstdc++-v3/testsuite/29_atomics/headers/stdatomic.h/115807.cc @@ -0,0 +1,14 @@ +// { dg-do compile { target c++23 } } +#include +namespace other { + namespace std { + int atomic = 0; + } + _Atomic(long) a{}; +} + +#include + +namespace non::std { + static_assert( ::std::is_same_v<_Atomic(int), ::std::atomic> ); +} commit a6283abe36cd4376141c680ada5f68000b8f603d Author: GCC Administrator Date: Wed Jul 10 00:21:12 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index b01fad6bd4f..1f0d44748df 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240709 +20240710 diff --git a/libstdc++-v3/ChangeLog b/libstdc++-v3/ChangeLog index 62345b0d11e..92466e991a0 100644 --- a/libstdc++-v3/ChangeLog +++ b/libstdc++-v3/ChangeLog @@ -1,3 +1,13 @@ +2024-07-09 Jonathan Wakely + + Backported from master: + 2024-07-08 Jonathan Wakely + + PR libstdc++/115807 + * include/c_compatibility/stdatomic.h (_Atomic): Ensure it + refers to std::atomic in the global namespace. + * testsuite/29_atomics/headers/stdatomic.h/115807.cc: New test. + 2024-06-27 Alexandre Oliva Backported from master: commit 10904e051f1b970cd8e030dff7dec8374c946b12 Author: Uros Bizjak Date: Wed Jul 10 09:27:27 2024 +0200 middle-end: Fix stalled swapped condition code value [PR115836] emit_store_flag_1 calculates scode (swapped condition code) at the beginning of the function from the value of code variable. However, code variable may change before scode usage site, resulting in invalid stalled scode value. Move calculation of scode value just before its only usage site to avoid stalled scode value. PR middle-end/115836 gcc/ChangeLog: * expmed.cc (emit_store_flag_1): Move calculation of scode just before its only usage site. (cherry picked from commit 44933fdeb338e00c972e42224b9a83d3f8f6a757) diff --git a/gcc/expmed.cc b/gcc/expmed.cc index 1bb4da8d094..39e53faec70 100644 --- a/gcc/expmed.cc +++ b/gcc/expmed.cc @@ -5601,11 +5601,9 @@ emit_store_flag_1 (rtx target, enum rtx_code code, rtx op0, rtx op1, enum insn_code icode; machine_mode compare_mode; enum mode_class mclass; - enum rtx_code scode; if (unsignedp) code = unsigned_condition (code); - scode = swap_condition (code); /* If one operand is constant, make it the second one. Only do this if the other operand is not constant as well. */ @@ -5773,6 +5771,8 @@ emit_store_flag_1 (rtx target, enum rtx_code code, rtx op0, rtx op1, if (GET_MODE_CLASS (mode) == MODE_FLOAT) { + enum rtx_code scode = swap_condition (code); + tem = emit_cstore (target, icode, scode, mode, compare_mode, unsignedp, op1, op0, normalizep, target_mode); if (tem) commit 8476fb6214d52a1f08fa479169f48d91d065bb87 Author: GCC Administrator Date: Thu Jul 11 00:20:20 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index d66f16bdb5e..40dde4b7cf7 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2024-07-10 Uros Bizjak + + Backported from master: + 2024-07-10 Uros Bizjak + + PR middle-end/115836 + * expmed.cc (emit_store_flag_1): Move calculation of + scode just before its only usage site. + 2024-07-05 Wilco Dijkstra Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 1f0d44748df..d8524ff6036 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240710 +20240711 commit a655c8d2098aff5235934263b065a389a9fcbbca Author: Andre Vieira Date: Thu Jul 11 15:38:45 2024 +0100 mve: Fix vsetq_lane for 64-bit elements with lane 1 [PR 115611] This patch fixes the backend pattern that was printing the wrong input scalar register pair when inserting into lane 1. Added a new test to force float-abi=hard so we can use scan-assembler to check correct codegen. gcc/ChangeLog: PR target/115611 * config/arm/mve.md (mve_vec_setv2di_internal): Fix printing of input scalar register pair when lane = 1. gcc/testsuite/ChangeLog: * gcc.target/arm/mve/intrinsics/vsetq_lane_su64.c: New test. (cherry picked from commit 7c11fdd2cc11a7058e9643b6abf27831970ad2c9) diff --git a/gcc/config/arm/mve.md b/gcc/config/arm/mve.md index 860d734b9a1..2de14013c43 100644 --- a/gcc/config/arm/mve.md +++ b/gcc/config/arm/mve.md @@ -10159,7 +10159,7 @@ if (elt == 0) return "vmov\t%e0, %Q1, %R1"; else - return "vmov\t%f0, %J1, %K1"; + return "vmov\t%f0, %Q1, %R1"; } [(set_attr "type" "mve_move")]) diff --git a/gcc/testsuite/gcc.target/arm/mve/intrinsics/vsetq_lane_su64.c b/gcc/testsuite/gcc.target/arm/mve/intrinsics/vsetq_lane_su64.c new file mode 100644 index 00000000000..5aa3bc9a76a --- /dev/null +++ b/gcc/testsuite/gcc.target/arm/mve/intrinsics/vsetq_lane_su64.c @@ -0,0 +1,63 @@ +/* { dg-require-effective-target arm_v8_1m_mve_ok } */ +/* { dg-add-options arm_v8_1m_mve } */ +/* { dg-require-effective-target arm_hard_ok } */ +/* { dg-additional-options "-mfloat-abi=hard -O2" } */ +/* { dg-final { check-function-bodies "**" "" } } */ + +#include "arm_mve.h" + +#ifdef __cplusplus +extern "C" { +#endif + +/* +**fn1: +** vmov d0, r0, r1 +** bx lr +*/ +uint64x2_t +fn1 (uint64_t a, uint64x2_t b) +{ + return vsetq_lane_u64 (a, b, 0); +} + +/* +**fn2: +** vmov d1, r0, r1 +** bx lr +*/ +uint64x2_t +fn2 (uint64_t a, uint64x2_t b) +{ + return vsetq_lane_u64 (a, b, 1); +} + +/* +**fn3: +** vmov d0, r0, r1 +** bx lr +*/ +int64x2_t +fn3 (int64_t a, int64x2_t b) +{ + return vsetq_lane_s64 (a, b, 0); +} + +/* +**fn4: +** vmov d1, r0, r1 +** bx lr +*/ +int64x2_t +fn4 (int64_t a, int64x2_t b) +{ + return vsetq_lane_s64 (a, b, 1); +} + + +#ifdef __cplusplus +} +#endif + +/* { dg-final { scan-assembler-not "__ARM_undef" } } */ + commit d6f54150239d195f68ca2454d1155969ce369a7c Author: GCC Administrator Date: Fri Jul 12 00:23:08 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 40dde4b7cf7..65eeec81b0f 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2024-07-11 Andre Vieira + + Backported from master: + 2024-07-11 Andre Vieira + + PR target/115611 + * config/arm/mve.md (mve_vec_setv2di_internal): Fix printing of input + scalar register pair when lane = 1. + 2024-07-10 Uros Bizjak Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d8524ff6036..93ea1dc1b26 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240711 +20240712 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 97e6e8bc00d..36a83377824 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,10 @@ +2024-07-11 Andre Vieira + + Backported from master: + 2024-07-11 Andre Vieira + + * gcc.target/arm/mve/intrinsics/vsetq_lane_su64.c: New test. + 2024-07-05 Georg-Johann Lay Backported from master: commit 419bbd5a4183ff13be9b126cf3dab31986191f19 Author: GCC Administrator Date: Sat Jul 13 00:21:22 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 93ea1dc1b26..c6f47ba7ab6 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240712 +20240713 commit 541ef1b454f3a7c799e994202e400c762e97f5b5 Author: GCC Administrator Date: Sun Jul 14 00:21:49 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c6f47ba7ab6..7a5180344d5 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240713 +20240714 commit b23efee77aabcd884d29c465b09510d47d989041 Author: GCC Administrator Date: Mon Jul 15 00:21:15 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7a5180344d5..5e0ee8afc37 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240714 +20240715 commit e1427b39d28f382d21e7a0ea1714b3250e0a6e5d Author: liuhongt Date: Fri Jul 12 09:39:23 2024 +0800 Fix SSA_NAME leak due to def_stmt is removed before use_stmt. - _5 = __atomic_fetch_or_8 (&set_work_pending_p, 1, 0); - # DEBUG old => (long int) _5 + _6 = .ATOMIC_BIT_TEST_AND_SET (&set_work_pending_p, 0, 1, 0, __atomic_fetch_or_8); + # DEBUG old => NULL # DEBUG BEGIN_STMT - # DEBUG D#2 => _5 & 1 + # DEBUG D#2 => NULL ... - _10 = ~_5; - _8 = (_Bool) _10; - # DEBUG ret => _8 + _8 = _6 == 0; + # DEBUG ret => (_Bool) _10 confirmed. convert_atomic_bit_not does this, it checks for single_use and removes the def, failing to release the name (which would fix this up IIRC). Note the function removes stmts in "wrong" order (before uses of LHS are removed), so it requires larger surgery. And it leaks SSA names. gcc/ChangeLog: PR target/115872 * tree-ssa-ccp.cc (convert_atomic_bit_not): Remove use_stmt after use_nop_stmt is removed. (optimize_atomic_bit_test_and): Ditto. gcc/testsuite/ChangeLog: * gcc.target/i386/pr115872.c: New test. (cherry picked from commit a8209237dc46dc4db7d9d8e3807e6c93734c64b5) diff --git a/gcc/testsuite/gcc.target/i386/pr115872.c b/gcc/testsuite/gcc.target/i386/pr115872.c new file mode 100644 index 00000000000..937004456d3 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr115872.c @@ -0,0 +1,16 @@ +/* { dg-do compile } */ +/* { dg-options "-O2 -g" } */ + +long set_work_pending_p; +_Bool set_work_pending() { + _Bool __trans_tmp_1; + long mask = 1, old = __atomic_fetch_or(&set_work_pending_p, mask, 0); + __trans_tmp_1 = old & mask; + return !__trans_tmp_1; +} +void __queue_work() { + _Bool ret = set_work_pending(); + if (ret) + __queue_work(); +} + diff --git a/gcc/tree-ssa-ccp.cc b/gcc/tree-ssa-ccp.cc index 42a02dccaeb..3c63f2dd8a3 100644 --- a/gcc/tree-ssa-ccp.cc +++ b/gcc/tree-ssa-ccp.cc @@ -3306,9 +3306,10 @@ convert_atomic_bit_not (enum internal_fn fn, gimple *use_stmt, return nullptr; gimple_stmt_iterator gsi; - gsi = gsi_for_stmt (use_stmt); - gsi_remove (&gsi, true); tree var = make_ssa_name (TREE_TYPE (lhs)); + /* use_stmt need to be removed after use_nop_stmt, + so use_lhs can be released. */ + gimple *use_stmt_removal = use_stmt; use_stmt = gimple_build_assign (var, BIT_AND_EXPR, lhs, and_mask); gsi = gsi_for_stmt (use_not_stmt); gsi_insert_before (&gsi, use_stmt, GSI_NEW_STMT); @@ -3318,6 +3319,8 @@ convert_atomic_bit_not (enum internal_fn fn, gimple *use_stmt, gsi_insert_after (&gsi, g, GSI_NEW_STMT); gsi = gsi_for_stmt (use_not_stmt); gsi_remove (&gsi, true); + gsi = gsi_for_stmt (use_stmt_removal); + gsi_remove (&gsi, true); return use_stmt; } @@ -3569,8 +3572,7 @@ optimize_atomic_bit_test_and (gimple_stmt_iterator *gsip, */ } var = make_ssa_name (TREE_TYPE (use_rhs)); - gsi = gsi_for_stmt (use_stmt); - gsi_remove (&gsi, true); + gimple* use_stmt_removal = use_stmt; g = gimple_build_assign (var, BIT_AND_EXPR, use_rhs, and_mask); gsi = gsi_for_stmt (use_nop_stmt); @@ -3584,6 +3586,8 @@ optimize_atomic_bit_test_and (gimple_stmt_iterator *gsip, gsi_insert_after (&gsi, g, GSI_NEW_STMT); gsi = gsi_for_stmt (use_nop_stmt); gsi_remove (&gsi, true); + gsi = gsi_for_stmt (use_stmt_removal); + gsi_remove (&gsi, true); } } else commit fbc5751fafe97df9e88217ec4c8eae4b25523d8f Author: GCC Administrator Date: Tue Jul 16 00:23:24 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 65eeec81b0f..3636cf488a5 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2024-07-15 liuhongt + + Backported from master: + 2024-07-15 liuhongt + + PR target/115872 + * tree-ssa-ccp.cc (convert_atomic_bit_not): Remove use_stmt after use_nop_stmt is removed. + (optimize_atomic_bit_test_and): Ditto. + 2024-07-11 Andre Vieira Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 5e0ee8afc37..e63e19096bc 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240715 +20240716 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 36a83377824..223909bb58e 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,10 @@ +2024-07-15 liuhongt + + Backported from master: + 2024-07-15 liuhongt + + * gcc.target/i386/pr115872.c: New test. + 2024-07-11 Andre Vieira Backported from master: commit 06d8257194444fd4b71c8e3d34fd9756be7f847b Author: Stefan Schulze Frielinghaus Date: Tue Jul 16 14:01:50 2024 +0200 s390: Align *cjump_64 and *icjump_64 During machine reorg we optimize backward jumps and transform insns as e.g. (jump_insn 118 117 119 (set (pc) (if_then_else (ne (reg:CCRAW 33 %cc) (const_int 8 [0x8])) (label_ref 134) (pc))) "dec_math_1.f90":204:8 discrim 1 2161 {*cjump_64} (expr_list:REG_DEAD (reg:CCRAW 33 %cc) (int_list:REG_BR_PROB 719407028 (nil))) -> 134) into (jump_insn 118 117 432 (set (pc) (if_then_else (ne (reg:CCRAW 33 %cc) (const_int 8 [0x8])) (pc) (label_ref 433))) "dec_math_1.f90":204:8 discrim 1 -1 (expr_list:REG_DEAD (reg:CCRAW 33 %cc) (int_list:REG_BR_PROB 719407028 (nil))) -> 433) The latter is not recognized anymore since *icjump_64 only matches CC_REGNUM against zero. Fixed by aligning *cjump_64 and *icjump_64. gcc/ChangeLog: * config/s390/s390.md (*icjump_64): Allow raw CC comparisons, i.e., any constant integer between 0 and 15 for CC comparisons. (cherry picked from commit 56de68aba6cb9cf3022d9e303eec6c6cdb49ad4d) diff --git a/gcc/config/s390/s390.md b/gcc/config/s390/s390.md index aaa247d7612..5b174e0d866 100644 --- a/gcc/config/s390/s390.md +++ b/gcc/config/s390/s390.md @@ -9472,7 +9472,8 @@ (define_insn "*icjump_64" [(set (pc) (if_then_else - (match_operator 1 "s390_comparison" [(reg CC_REGNUM) (const_int 0)]) + (match_operator 1 "s390_comparison" [(reg CC_REGNUM) + (match_operand 2 "const_int_operand" "")]) (pc) (label_ref (match_operand 0 "" ""))))] "" commit 9e00ae3e23eef8bff497981e00853ca092772201 Author: Stefan Schulze Frielinghaus Date: Tue Jul 16 14:01:58 2024 +0200 s390: Fix output template for movv1qi Although for instructions MVI and MVIY it does not make a difference whether the immediate is interpreted as signed or unsigned, GAS expects unsigned immediates for instruction format SI_URD. gcc/ChangeLog: * config/s390/vector.md (mov): Fix output template for movv1qi. (cherry picked from commit e6680d3f392f7f7cc2a1515276213e21e9eeab1c) diff --git a/gcc/config/s390/vector.md b/gcc/config/s390/vector.md index 624729814af..e795b4ffef7 100644 --- a/gcc/config/s390/vector.md +++ b/gcc/config/s390/vector.md @@ -357,8 +357,8 @@ lr\t%0,%1 mvi\t%0,0 mviy\t%0,0 - mvi\t%0,-1 - mviy\t%0,-1 + mvi\t%0,255 + mviy\t%0,255 lhi\t%0,0 lhi\t%0,-1 llc\t%0,%1 commit 3666b142d1a9442a4c465e035d665647cef5141f Author: GCC Administrator Date: Wed Jul 17 00:21:57 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 3636cf488a5..b49222264b2 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,19 @@ +2024-07-16 Stefan Schulze Frielinghaus + + Backported from master: + 2024-07-12 Stefan Schulze Frielinghaus + + * config/s390/vector.md (mov): Fix output template for + movv1qi. + +2024-07-16 Stefan Schulze Frielinghaus + + Backported from master: + 2024-07-12 Stefan Schulze Frielinghaus + + * config/s390/s390.md (*icjump_64): Allow raw CC comparisons, + i.e., any constant integer between 0 and 15 for CC comparisons. + 2024-07-15 liuhongt Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e63e19096bc..5961a7c72a1 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240716 +20240717 commit a98dc99196dde71c7ac3cc76e77f5bbf0d09e556 Author: GCC Administrator Date: Thu Jul 18 00:21:44 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 5961a7c72a1..1d16bb88567 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240717 +20240718 commit c5a26fc24b0af61498fae65ccad69d51d63d2a8b Author: Uros Bizjak Date: Wed Jul 17 18:11:26 2024 +0200 alpha: Fix duplicate !tlsgd!62 assemble error [PR115526] Add missing "cannot_copy" attribute to instructions that have to stay in 1-1 correspondence with another insn. PR target/115526 gcc/ChangeLog: * config/alpha/alpha.md (movdi_er_high_g): Add cannot_copy attribute. (movdi_er_tlsgd): Ditto. (movdi_er_tlsldm): Ditto. (call_value_osf_): Ditto. gcc/testsuite/ChangeLog: * gcc.target/alpha/pr115526.c: New test. (cherry picked from commit 0841fd4c42ab053be951b7418233f0478282d020) diff --git a/gcc/config/alpha/alpha.md b/gcc/config/alpha/alpha.md index 442953fe50e..b6795e1d263 100644 --- a/gcc/config/alpha/alpha.md +++ b/gcc/config/alpha/alpha.md @@ -3933,7 +3933,8 @@ else return "ldq %0,%2(%1)\t\t!literal!%3"; } - [(set_attr "type" "ldsym")]) + [(set_attr "type" "ldsym") + (set_attr "cannot_copy" "true")]) (define_split [(set (match_operand:DI 0 "register_operand") @@ -3957,7 +3958,8 @@ return "lda %0,%2(%1)\t\t!tlsgd"; else return "lda %0,%2(%1)\t\t!tlsgd!%3"; -}) +} + [(set_attr "cannot_copy" "true")]) (define_insn "movdi_er_tlsldm" [(set (match_operand:DI 0 "register_operand" "=r") @@ -3970,7 +3972,8 @@ return "lda %0,%&(%1)\t\t!tlsldm"; else return "lda %0,%&(%1)\t\t!tlsldm!%2"; -}) +} + [(set_attr "cannot_copy" "true")]) (define_insn "*movdi_er_gotdtp" [(set (match_operand:DI 0 "register_operand" "=r") @@ -5939,6 +5942,7 @@ "HAVE_AS_TLS" "ldq $27,%1($29)\t\t!literal!%2\;jsr $26,($27),%1\t\t!lituse_!%2\;ldah $29,0($26)\t\t!gpdisp!%*\;lda $29,0($29)\t\t!gpdisp!%*" [(set_attr "type" "jsr") + (set_attr "cannot_copy" "true") (set_attr "length" "16")]) ;; We must use peep2 instead of a split because we need accurate life diff --git a/gcc/testsuite/gcc.target/alpha/pr115526.c b/gcc/testsuite/gcc.target/alpha/pr115526.c new file mode 100644 index 00000000000..2f57903fec3 --- /dev/null +++ b/gcc/testsuite/gcc.target/alpha/pr115526.c @@ -0,0 +1,46 @@ +/* PR target/115526 */ +/* { dg-do assemble } */ +/* { dg-options "-O2 -Wno-attributes -fvisibility=hidden -fPIC -mcpu=ev4" } */ + +struct _ts { + struct _dtoa_state *interp; +}; +struct Bigint { + int k; +} *_Py_dg_strtod_bs; +struct _dtoa_state { + struct Bigint p5s; + struct Bigint *freelist[]; +}; +extern _Thread_local struct _ts _Py_tss_tstate; +typedef struct Bigint Bigint; +int pow5mult_k; +long _Py_dg_strtod_ndigits; +void PyMem_Free(); +void Bfree(Bigint *v) { + if (v) + { + if (v->k) + PyMem_Free(); + else { + struct _dtoa_state *interp = _Py_tss_tstate.interp; + interp->freelist[v->k] = v; + } + } +} +static Bigint *pow5mult(Bigint *b) { + for (;;) { + if (pow5mult_k & 1) { + Bfree(b); + if (b == 0) + return 0; + } + if (!(pow5mult_k >>= 1)) + break; + } + return 0; +} +void _Py_dg_strtod() { + if (_Py_dg_strtod_ndigits) + pow5mult(_Py_dg_strtod_bs); +} commit b1ec101ee38ab974782685429f00258d824258ef Author: GCC Administrator Date: Fri Jul 19 00:21:11 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index b49222264b2..3ac7e625ffb 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,14 @@ +2024-07-18 Uros Bizjak + + Backported from master: + 2024-07-17 Uros Bizjak + + PR target/115526 + * config/alpha/alpha.md (movdi_er_high_g): Add cannot_copy attribute. + (movdi_er_tlsgd): Ditto. + (movdi_er_tlsldm): Ditto. + (call_value_osf_): Ditto. + 2024-07-16 Stefan Schulze Frielinghaus Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 1d16bb88567..92d0274c3c6 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240718 +20240719 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 223909bb58e..2799a1df693 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-07-18 Uros Bizjak + + Backported from master: + 2024-07-17 Uros Bizjak + + PR target/115526 + * gcc.target/alpha/pr115526.c: New test. + 2024-07-15 liuhongt Backported from master: commit 2c5f48a43f26223cb8603b826d7c0d52cdbcfb46 Author: LIU Hao Date: Mon Jul 15 16:55:52 2024 +0800 Do not use caller-saved registers for COMDAT functions A reference to a COMDAT function may be resolved to another definition outside the current translation unit, so it's not eligible for `-fipa-ra`. In `decl_binds_to_current_def_p()` there is already a check for weak symbols. This commit checks for COMDAT functions that are not implemented as weak symbols, for example, on *-*-mingw32. gcc/ChangeLog: PR rtl-optimization/115049 * varasm.cc (decl_binds_to_current_def_p): Add a check for COMDAT declarations too, like weak ones. (cherry picked from commit 5080840d8fbf25a321dd27543a1462d393d338bc) diff --git a/gcc/varasm.cc b/gcc/varasm.cc index dc9f733791a..9fa3a42bf45 100644 --- a/gcc/varasm.cc +++ b/gcc/varasm.cc @@ -7621,6 +7621,8 @@ decl_binds_to_current_def_p (const_tree decl) for all other declaration types. */ if (DECL_WEAK (decl)) return false; + if (DECL_COMDAT_GROUP (decl)) + return false; if (DECL_COMMON (decl) && (DECL_INITIAL (decl) == NULL || (!in_lto_p && DECL_INITIAL (decl) == error_mark_node))) commit 958c386ac93b4e7ad01421c76e1e0de909c6cd0d Author: GCC Administrator Date: Sat Jul 20 00:19:45 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 3ac7e625ffb..cc8a3db8764 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2024-07-19 LIU Hao + + Backported from master: + 2024-07-18 LIU Hao + + PR rtl-optimization/115049 + * varasm.cc (decl_binds_to_current_def_p): Add a check for COMDAT + declarations too, like weak ones. + 2024-07-18 Uros Bizjak Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 92d0274c3c6..6c93b259fd6 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240719 +20240720 commit 7ad764fe3c3ad0e1167b58cf3785629d788491f4 Author: Stefan Schulze Frielinghaus Date: Sat Jul 20 17:13:03 2024 +0200 s390: Fix unresolved iterators bhfgq and xdee Code attribute bhfgq is missing a mapping for TF. This results in unresolved iterators in assembler templates for *bswaptf. With the TF mapping added the base mnemonics vlbr and vstbr are not "used" anymore but only the extended mnemonics (vlbr was interpreted as vlbr; likewise for vstbr). Therefore, remove the base mnemonics from the scheduling description, otherwise, genattrtab would error about unknown mnemonics. Likewise, for movtf_vr only the extended mnemonics for vrepi are used, now, which means the base mnemonic is "unused" and has to be removed from the scheduling description. Similarly, we end up with unresolved iterators in assembler templates for mulfprx23 since code attribute xdee is missing a mapping for FPRX2. Note, this is basically a cherry pick of commit r15-2060-ga4abda934aa426 with the addition that vrepi is removed from the scheduling description, too. gcc/ChangeLog: * config/s390/3931.md (vlbr, vstbr, vrepi): Remove. * config/s390/s390.md (xdee): Add FPRX2 mapping. * config/s390/vector.md (bhfgq): Add TF mapping. diff --git a/gcc/config/s390/3931.md b/gcc/config/s390/3931.md index bc97bc50b78..ecd290abc08 100644 --- a/gcc/config/s390/3931.md +++ b/gcc/config/s390/3931.md @@ -404,7 +404,6 @@ vlvgg, vlvgh, vlvgp, vst, -vstbr, vstbrf, vstbrg, vstbrh, @@ -627,7 +626,6 @@ tm, tmy, vl, vlbb, -vlbr, vlbrf, vlbrg, vlbrh, @@ -661,7 +659,6 @@ vlreph, vlrl, vlrlr, vst, -vstbr, vstbrf, vstbrg, vstbrh, @@ -1077,7 +1074,6 @@ vrepb, vrepf, vrepg, vreph, -vrepi, vrepib, vrepif, vrepig, @@ -1930,7 +1926,6 @@ vrepb, vrepf, vrepg, vreph, -vrepi, vrepib, vrepif, vrepig, @@ -2156,7 +2151,6 @@ vistrfs, vistrhs, vl, vlbb, -vlbr, vlbrf, vlbrg, vlbrh, @@ -2248,7 +2242,6 @@ tbegin, tbeginc, tend, vst, -vstbr, vstbrf, vstbrg, vstbrh, diff --git a/gcc/config/s390/s390.md b/gcc/config/s390/s390.md index 5b174e0d866..b8dcf2bb58a 100644 --- a/gcc/config/s390/s390.md +++ b/gcc/config/s390/s390.md @@ -745,7 +745,7 @@ ;; In FP templates, a in "mr" will expand to "mxr" in ;; TF/TDmode, "mdr" in DF/DDmode, "meer" in SFmode and "mer in ;; SDmode. -(define_mode_attr xdee [(TF "x") (DF "d") (SF "ee") (TD "x") (DD "d") (SD "e")]) +(define_mode_attr xdee [(TF "x") (FPRX2 "x") (DF "d") (SF "ee") (TD "x") (DD "d") (SD "e")]) ;; The decimal floating point variants of add, sub, div and mul support 3 ;; fp register operands. The following attributes allow to merge the bfp and diff --git a/gcc/config/s390/vector.md b/gcc/config/s390/vector.md index e795b4ffef7..75912280c23 100644 --- a/gcc/config/s390/vector.md +++ b/gcc/config/s390/vector.md @@ -134,7 +134,7 @@ (V1TI "q") (TI "q") (V1SF "f") (V2SF "f") (V4SF "f") (V1DF "g") (V2DF "g") - (V1TF "q")]) + (V1TF "q") (TF "q")]) ; This is for vmalhw. It gets an 'w' attached to avoid confusion with ; multiply and add logical high vmalh. commit 319b57fb02b52ba9036c00dda36ff28d8274e13d Author: René Rebe Date: Fri Jul 12 21:17:08 2024 +0000 rs6000: Fix .machine cpu selection w/ altivec [PR97367] There are various non-IBM CPUs with altivec, so we cannot use that flag to determine which .machine cpu to use, so ignore it. Emit an additional ".machine altivec" if Altivec is enabled so that the assembler doesn't require an explicit -maltivec option to assemble any Altivec instructions for those targets where the ".machine cpu" is insufficient to enable Altivec. For example, -mcpu=G5 emits a ".machine power4". 2024-07-18 René Rebe Peter Bergner gcc/ PR target/97367 * config/rs6000/rs6000.cc (rs6000_machine_from_flags): Do not consider OPTION_MASK_ALTIVEC. (emit_asm_machine): For Altivec compiles, emit a ".machine altivec". gcc/testsuite/ PR target/97367 * gcc.target/powerpc/pr97367.c: New test. Signed-off-by: René Rebe (cherry picked from commit 6962835bca3e6bef0f6ceae84a7814138b08b8a5) diff --git a/gcc/config/rs6000/rs6000.cc b/gcc/config/rs6000/rs6000.cc index 34be43c9f84..03893b2cf8b 100644 --- a/gcc/config/rs6000/rs6000.cc +++ b/gcc/config/rs6000/rs6000.cc @@ -5813,7 +5813,8 @@ rs6000_machine_from_flags (void) HOST_WIDE_INT flags = rs6000_isa_flags; /* Disable the flags that should never influence the .machine selection. */ - flags &= ~(OPTION_MASK_PPC_GFXOPT | OPTION_MASK_PPC_GPOPT | OPTION_MASK_ISEL); + flags &= ~(OPTION_MASK_PPC_GFXOPT | OPTION_MASK_PPC_GPOPT | OPTION_MASK_ISEL + | OPTION_MASK_ALTIVEC); if ((flags & (ISA_3_1_MASKS_SERVER & ~ISA_3_0_MASKS_SERVER)) != 0) return "power10"; @@ -5838,6 +5839,8 @@ void emit_asm_machine (void) { fprintf (asm_out_file, "\t.machine %s\n", rs6000_machine); + if (TARGET_ALTIVEC) + fprintf (asm_out_file, "\t.machine altivec\n"); } #endif diff --git a/gcc/testsuite/gcc.target/powerpc/pr97367.c b/gcc/testsuite/gcc.target/powerpc/pr97367.c new file mode 100644 index 00000000000..ef269a5f913 --- /dev/null +++ b/gcc/testsuite/gcc.target/powerpc/pr97367.c @@ -0,0 +1,13 @@ +/* PR target/97367 */ +/* { dg-options "-mdejagnu-cpu=G5" } */ + +/* Verify we emit a ".machine power4" and ".machine altivec" rather + than a ".machine power7". */ + +int dummy (void) +{ + return 0; +} + +/* { dg-final { scan-assembler {\.machine power4\M} } } */ +/* { dg-final { scan-assembler {\.machine altivec\M} } } */ commit a551d0330d62d8b5d28c4c9af63bae17afb3bbc4 Author: Siddhesh Poyarekar Date: Fri Jul 19 12:44:32 2024 -0400 Avoid undefined behaviour in build_option_suggestions The inner loop in build_option_suggestions uses OPTION to take the address of OPTB and use it across iterations, which is undefined behaviour since OPTB is defined within the loop. Pull it outside the loop to make this defined. gcc/ChangeLog: * opt-suggestions.cc (option_proposer::build_option_suggestions): Pull OPTB definition out of the innermost loop. (cherry picked from commit e0d997e913f811ecf4b3e10891e6a4aab5b38a31) diff --git a/gcc/opt-suggestions.cc b/gcc/opt-suggestions.cc index 33f298560a1..92969e74d40 100644 --- a/gcc/opt-suggestions.cc +++ b/gcc/opt-suggestions.cc @@ -167,9 +167,9 @@ option_proposer::build_option_suggestions (const char *prefix) add_misspelling_candidates (m_option_suggestions, option, opt_text); + struct cl_option optb; for (int j = 0; sanitizer_opts[j].name != NULL; ++j) { - struct cl_option optb; /* -fsanitize=all is not valid, only -fno-sanitize=all. So don't register the positive misspelling candidates for it. */ commit cf4b402cec876be889ac3f96992ef2bdcff5b7f5 Author: GCC Administrator Date: Sun Jul 21 00:19:41 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index cc8a3db8764..96fc52ebd1f 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,29 @@ +2024-07-20 Siddhesh Poyarekar + + Backported from master: + 2024-07-20 Siddhesh Poyarekar + + * opt-suggestions.cc + (option_proposer::build_option_suggestions): Pull OPTB + definition out of the innermost loop. + +2024-07-20 René Rebe + + Backported from master: + 2024-07-18 René Rebe + Peter Bergner + + PR target/97367 + * config/rs6000/rs6000.cc (rs6000_machine_from_flags): Do not consider + OPTION_MASK_ALTIVEC. + (emit_asm_machine): For Altivec compiles, emit a ".machine altivec". + +2024-07-20 Stefan Schulze Frielinghaus + + * config/s390/3931.md (vlbr, vstbr, vrepi): Remove. + * config/s390/s390.md (xdee): Add FPRX2 mapping. + * config/s390/vector.md (bhfgq): Add TF mapping. + 2024-07-19 LIU Hao Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 6c93b259fd6..d1ac33f77fb 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240720 +20240721 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 2799a1df693..3a9e4ff55e2 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,12 @@ +2024-07-20 René Rebe + + Backported from master: + 2024-07-18 René Rebe + Peter Bergner + + PR target/97367 + * gcc.target/powerpc/pr97367.c: New test. + 2024-07-18 Uros Bizjak Backported from master: commit ecc80e18f05b77a773c6d894871572029d4fc579 Author: Harald Anlauf Date: Thu Jul 18 21:15:48 2024 +0200 Fortran: character array constructor with >= 4 constant elements [PR103115] gcc/fortran/ChangeLog: PR fortran/103115 * trans-array.cc (gfc_trans_array_constructor_value): If the first element of an array constructor is deferred-length character and therefore does not have an element size known at compile time, do not try to collect subsequent constant elements into a constructor for optimization. gcc/testsuite/ChangeLog: PR fortran/103115 * gfortran.dg/string_array_constructor_4.f90: New test. (cherry picked from commit c93be1606ecf8e0f65b96b67aa023fb456ceb3a3) diff --git a/gcc/fortran/trans-array.cc b/gcc/fortran/trans-array.cc index c82b1fa47e5..85c641b55c5 100644 --- a/gcc/fortran/trans-array.cc +++ b/gcc/fortran/trans-array.cc @@ -2114,7 +2114,9 @@ gfc_trans_array_constructor_value (stmtblock_t * pblock, tree type, p = gfc_constructor_next (p); n++; } - if (n < 4) + /* Constructor with few constant elements, or element size not + known at compile time (e.g. deferred-length character). */ + if (n < 4 || !INTEGER_CST_P (TYPE_SIZE_UNIT (type))) { /* Scalar values. */ gfc_init_se (&se, NULL); diff --git a/gcc/testsuite/gfortran.dg/string_array_constructor_4.f90 b/gcc/testsuite/gfortran.dg/string_array_constructor_4.f90 new file mode 100644 index 00000000000..b5b81f07395 --- /dev/null +++ b/gcc/testsuite/gfortran.dg/string_array_constructor_4.f90 @@ -0,0 +1,59 @@ +! { dg-do run } +! PR fortran/103115 - character array constructor with >= 4 constant elements +! +! This used to ICE when the first element is deferred-length character +! or could lead to wrong results. + +program pr103115 + implicit none + integer :: i + character(*), parameter :: expect(*) = [ "1","2","3","4","5" ] + character(5) :: abc = "12345" + character(5), parameter :: def = "12345" + character(:), dimension(:), allocatable :: list + character(:), dimension(:), allocatable :: titles + titles = ["1"] + titles = [ titles& + ,"2"& + ,"3"& + ,"4"& + ,"5"& ! used to ICE + ] + if (len (titles) /= 1 .or. size (titles) /= 5) stop 1 + if (any (titles /= expect)) stop 2 + titles = ["1"] + titles = [ titles, (char(48+i),i=2,5) ] + if (len (titles) /= 1 .or. size (titles) /= 5) stop 3 + if (any (titles /= expect)) stop 4 + titles = ["1"] + titles = [ titles, ("2345"(i:i),i=1,4) ] + if (len (titles) /= 1 .or. size (titles) /= 5) stop 5 + if (any (titles /= expect)) stop 6 + titles = ["1"] + titles = [ titles, (def(i:i),i=2,5) ] + if (len (titles) /= 1 .or. size (titles) /= 5) stop 7 + if (any (titles /= expect)) stop 8 + list = [ (char(48+i),i=1,5) ] + titles = [ list(1), (char(48+i),i=2,5) ] + if (len (titles) /= 1 .or. size (titles) /= 5) stop 9 + if (any (titles /= expect)) stop 10 + titles = ["1"] + titles = [ titles, (abc(i:i),i=2,5) ] + if (len (titles) /= 1 .or. size (titles) /= 5) stop 11 + if (any (titles /= expect)) stop 12 + + ! with typespec: + list = [ (char(48+i),i=1,5) ] + titles = [ character(2) :: list(1), (char(48+i),i=2,5) ] + if (len (titles) /= 2 .or. size (titles) /= 5) stop 13 + if (any (titles /= expect)) stop 14 + titles = ["1"] + titles = [ character(2) :: titles, (char(48+i),i=2,5) ] + if (len (titles) /= 2 .or. size (titles) /= 5) stop 15 + if (any (titles /= expect)) stop 16 + titles = ["1"] + titles = [ character(2) :: titles, (def(i:i),i=2,5) ] + if (len (titles) /= 2 .or. size (titles) /= 5) stop 17 + if (any (titles /= expect)) stop 18 + deallocate (titles, list) +end commit 5f5c379607c36f177fe46de1222a11bf1b1e7f4c Author: GCC Administrator Date: Mon Jul 22 00:20:00 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d1ac33f77fb..01bbf6ed5e1 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240721 +20240722 diff --git a/gcc/fortran/ChangeLog b/gcc/fortran/ChangeLog index 211d8ffc1a4..089fbce18c4 100644 --- a/gcc/fortran/ChangeLog +++ b/gcc/fortran/ChangeLog @@ -1,3 +1,15 @@ +2024-07-21 Harald Anlauf + + Backported from master: + 2024-07-19 Harald Anlauf + + PR fortran/103115 + * trans-array.cc (gfc_trans_array_constructor_value): If the first + element of an array constructor is deferred-length character and + therefore does not have an element size known at compile time, do + not try to collect subsequent constant elements into a constructor + for optimization. + 2024-06-20 Release Manager * GCC 12.4.0 released. diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 3a9e4ff55e2..a7b00d7882c 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-07-21 Harald Anlauf + + Backported from master: + 2024-07-19 Harald Anlauf + + PR fortran/103115 + * gfortran.dg/string_array_constructor_4.f90: New test. + 2024-07-20 René Rebe Backported from master: commit 8d8f804b18e4a38671957b3e4c239ef625506317 Author: Maciej W. Rozycki Date: Sat Jun 29 23:26:55 2024 +0100 [PR115565] cse: Don't use a valid regno for non-register in comparison_qty Use INT_MIN rather than -1 in `comparison_qty' where a comparison is not with a register, because the value of -1 is actually a valid reference to register 0 in the case where it has not been assigned a quantity. Using -1 makes `REG_QTY (REGNO (folded_arg1)) == ent->comparison_qty' comparison in `fold_rtx' to incorrectly trigger in rare circumstances and return true for a memory reference, making CSE consider a comparison operation to evaluate to a constant expression and consequently make the resulting code incorrectly execute or fail to execute conditional blocks. This has caused a miscompilation of rwlock.c from LinuxThreads for the `alpha-linux-gnu' target, where `rwlock->__rw_writer != thread_self ()' expression (where `thread_self' returns the thread pointer via a PALcode call) has been decided to be always true (with `ent->comparison_qty' using -1 for a reference to to `rwlock->__rw_writer', while register 0 holding the thread pointer retrieved by `thread_self') and code for the false case has been optimized away where it mustn't have, causing program lockups. The issue has been observed as a regression from commit 08a692679fb8 ("Undefined cse.c behaviour causes 3.4 regression on HPUX"), , and up to commit 932ad4d9b550 ("Make CSE path following use the CFG"), , where CSE has been restructured sufficiently for the issue not to trigger with the original reproducer anymore. However the original bug remains and can trigger, because `comparison_qty' will still be assigned -1 for a memory reference and the `reg_qty' member of a `cse_reg_info_table' entry will still be assigned -1 for register 0 where the entry has not been assigned a quantity, e.g. at initialization. Use INT_MIN then as noted above, so that the value remains negative, for consistency with the REGNO_QTY_VALID_P macro (even though not used on `comparison_qty'), and then so that it should not ever match a valid negated register number, fixing the regression with commit 08a692679fb8. gcc/ PR rtl-optimization/115565 * cse.cc (record_jump_cond): Use INT_MIN rather than -1 for `comparison_qty' if !REG_P. (cherry picked from commit 69bc5fb97dc3fada81869e00fa65d39f7def6acf) diff --git a/gcc/cse.cc b/gcc/cse.cc index ca53974810e..32e8ea79980 100644 --- a/gcc/cse.cc +++ b/gcc/cse.cc @@ -239,7 +239,7 @@ static int next_qty; the constant being compared against, or zero if the comparison is not against a constant. `comparison_qty' holds the quantity being compared against when the result is known. If the comparison - is not with a register, `comparison_qty' is -1. */ + is not with a register, `comparison_qty' is INT_MIN. */ struct qty_table_elem { @@ -4068,7 +4068,7 @@ record_jump_cond (enum rtx_code code, machine_mode mode, rtx op0, else { ent->comparison_const = op1; - ent->comparison_qty = -1; + ent->comparison_qty = INT_MIN; } return; commit 42016b6ceeef4d77f91974de1d80769c072bc854 Author: GCC Administrator Date: Tue Jul 23 00:20:10 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 96fc52ebd1f..912a1b4f6ec 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2024-07-22 Maciej W. Rozycki + + Backported from master: + 2024-06-29 Maciej W. Rozycki + + PR rtl-optimization/115565 + * cse.cc (record_jump_cond): Use INT_MIN rather than -1 for + `comparison_qty' if !REG_P. + 2024-07-20 Siddhesh Poyarekar Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 01bbf6ed5e1..833d506febb 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240722 +20240723 commit e142b6607267100537fc7abe6f60a52fc0d8535c Author: Alexandre Oliva Date: Tue Jul 23 02:19:55 2024 -0300 [powerpc] [testsuite] reorder dg directives [PR106069] The dg-do directive appears after dg-require-effective-target in g++.target/powerpc/pr106069.C. That doesn't work the way that was presumably intended. Both of these directives set dg-do-what, but dg-do does so fully and unconditionally, overriding any decisions recorded there by earlier directives. Reorder the directives more canonically, so that both take effect. for gcc/testsuite/ChangeLog PR target/106069 * g++.target/powerpc/pr106069.C: Reorder dg directives. (cherry picked from commit ad65caa332bc7600caff6b9b5b29175b40d91e67) diff --git a/gcc/testsuite/g++.target/powerpc/pr106069.C b/gcc/testsuite/g++.target/powerpc/pr106069.C index 537207d2fe8..826379a4479 100644 --- a/gcc/testsuite/g++.target/powerpc/pr106069.C +++ b/gcc/testsuite/g++.target/powerpc/pr106069.C @@ -1,6 +1,6 @@ +/* { dg-do run } */ /* { dg-options "-O -fno-tree-forwprop -maltivec" } */ /* { dg-require-effective-target vmx_hw } */ -/* { dg-do run } */ typedef __attribute__ ((altivec (vector__))) unsigned native_simd_type; commit f78eb9524bd97679c8baa47a62e82147272719ae Author: Richard Biener Date: Mon Jul 15 13:01:24 2024 +0200 Fixup unaligned load/store cost for znver4 Currently unaligned YMM and ZMM load and store costs are cheaper than aligned which causes the vectorizer to purposely mis-align accesses by adding an alignment prologue. It looks like the unaligned costs were simply left untouched from znver3 where they equate the aligned costs when tweaking aligned costs for znver4. The following makes the unaligned costs equal to the aligned costs. This avoids the miscompile seen in PR115843 but it's of course not a real fix for the issue uncovered there. But it makes it qualify as a regression fix. PR tree-optimization/115843 * config/i386/x86-tune-costs.h (znver4_cost): Update unaligned load and store cost from the aligned costs. (cherry picked from commit 1e3aa9c9278db69d4bdb661a750a7268789188d6) diff --git a/gcc/config/i386/x86-tune-costs.h b/gcc/config/i386/x86-tune-costs.h index f105d57cae7..d5882788899 100644 --- a/gcc/config/i386/x86-tune-costs.h +++ b/gcc/config/i386/x86-tune-costs.h @@ -1894,8 +1894,8 @@ struct processor_costs znver4_cost = { in 32bit, 64bit, 128bit, 256bit and 512bit */ {8, 8, 8, 12, 12}, /* cost of storing SSE register in 32bit, 64bit, 128bit, 256bit and 512bit */ - {6, 6, 6, 6, 6}, /* cost of unaligned loads. */ - {8, 8, 8, 8, 8}, /* cost of unaligned stores. */ + {6, 6, 10, 10, 12}, /* cost of unaligned loads. */ + {8, 8, 8, 12, 12}, /* cost of unaligned stores. */ 2, 2, 2, /* cost of moving XMM,YMM,ZMM register. */ 6, /* cost of moving SSE register to integer. */ commit aaa8af5e7caa07b6c213010e68a91fa8cea9ef38 Author: GCC Administrator Date: Wed Jul 24 00:20:56 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 912a1b4f6ec..645d8748256 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2024-07-23 Richard Biener + + Backported from master: + 2024-07-16 Richard Biener + + PR tree-optimization/115843 + * config/i386/x86-tune-costs.h (znver4_cost): Update unaligned + load and store cost from the aligned costs. + 2024-07-22 Maciej W. Rozycki Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 833d506febb..d050dd67721 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240723 +20240724 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index a7b00d7882c..6357df9e57d 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-07-23 Alexandre Oliva + + Backported from master: + 2024-07-23 Alexandre Oliva + + PR target/106069 + * g++.target/powerpc/pr106069.C: Reorder dg directives. + 2024-07-21 Harald Anlauf Backported from master: commit 2647b9e052eafbbec1094558167be9a24e2d8221 Author: Peter Bergner Date: Fri Jun 7 16:03:08 2024 -0500 rs6000: Update ELFv2 stack frame comment showing the correct ROP save location The ELFv2 stack frame layout comment in rs6000-logue.cc shows the ROP hash save slot in the wrong location. Update the comment to show the correct ROP hash save location in the frame. 2024-06-07 Peter Bergner gcc/ * config/rs6000/rs6000-logue.cc (rs6000_stack_info): Update comment. (cherry picked from commit e91cf26a954a5c1bf431e36f3a1e69f94e9fa4fe) diff --git a/gcc/config/rs6000/rs6000-logue.cc b/gcc/config/rs6000/rs6000-logue.cc index 33077b72611..270f4157375 100644 --- a/gcc/config/rs6000/rs6000-logue.cc +++ b/gcc/config/rs6000/rs6000-logue.cc @@ -595,21 +595,21 @@ rs6000_savres_strategy (rs6000_stack_t *info, +---------------------------------------+ | Parameter save area (+padding*) (P) | 32 +---------------------------------------+ - | Optional ROP hash slot (R) | 32+P + | Alloca space (A) | 32+P +---------------------------------------+ - | Alloca space (A) | 32+P+R + | Local variable space (L) | 32+P+A +---------------------------------------+ - | Local variable space (L) | 32+P+R+A + | Optional ROP hash slot (R) | 32+P+A+L +---------------------------------------+ - | Save area for AltiVec registers (W) | 32+P+R+A+L + | Save area for AltiVec registers (W) | 32+P+A+L+R +---------------------------------------+ - | AltiVec alignment padding (Y) | 32+P+R+A+L+W + | AltiVec alignment padding (Y) | 32+P+A+L+R+W +---------------------------------------+ - | Save area for GP registers (G) | 32+P+R+A+L+W+Y + | Save area for GP registers (G) | 32+P+A+L+R+W+Y +---------------------------------------+ - | Save area for FP registers (F) | 32+P+R+A+L+W+Y+G + | Save area for FP registers (F) | 32+P+A+L+R+W+Y+G +---------------------------------------+ - old SP->| back chain to caller's caller | 32+P+R+A+L+W+Y+G+F + old SP->| back chain to caller's caller | 32+P+A+L+R+W+Y+G+F +---------------------------------------+ * If the alloca area is present, the parameter save area is commit 60e513cd47aadd8f139079f8388b14930e6e0913 Author: Peter Bergner Date: Fri Jun 14 14:36:20 2024 -0500 rs6000: Compute rop_hash_save_offset for non-Altivec compiles [PR115389] We currently only compute the offset for the ROP hash save location in the stack frame for Altivec compiles. For non-Altivec compiles when we emit ROP mitigation instructions, we use a default offset of zero which corresponds to the backchain save location which will get clobbered on any call. The fix is to compute the ROP hash save location for all compiles. 2024-06-14 Peter Bergner gcc/ PR target/115389 * config/rs6000/rs6000-logue.cc (rs6000_stack_info): Compute rop_hash_save_offset for non-Altivec compiles. gcc/testsuite PR target/115389 * gcc.target/powerpc/pr115389.c: New test. (cherry picked from commit c70eea0dba5f223d49c80cfb3e80e87b74330aac) diff --git a/gcc/config/rs6000/rs6000-logue.cc b/gcc/config/rs6000/rs6000-logue.cc index 270f4157375..3894bd23d17 100644 --- a/gcc/config/rs6000/rs6000-logue.cc +++ b/gcc/config/rs6000/rs6000-logue.cc @@ -821,17 +821,16 @@ rs6000_stack_info (void) gcc_assert (info->altivec_size == 0 || info->altivec_save_offset % 16 == 0); - /* Adjust for AltiVec case. */ - info->ehrd_offset = info->altivec_save_offset - ehrd_size; - /* Adjust for ROP protection. */ info->rop_hash_save_offset = info->altivec_save_offset - info->rop_hash_size; - info->ehrd_offset -= info->rop_hash_size; } else - info->ehrd_offset = info->gp_save_offset - ehrd_size; + /* Adjust for ROP protection. */ + info->rop_hash_save_offset + = info->gp_save_offset - info->rop_hash_size; + info->ehrd_offset = info->rop_hash_save_offset - ehrd_size; info->ehcr_offset = info->ehrd_offset - ehcr_size; info->cr_save_offset = reg_size; /* first word when 64-bit. */ info->lr_save_offset = 2*reg_size; diff --git a/gcc/testsuite/gcc.target/powerpc/pr115389.c b/gcc/testsuite/gcc.target/powerpc/pr115389.c new file mode 100644 index 00000000000..a091ee8a1be --- /dev/null +++ b/gcc/testsuite/gcc.target/powerpc/pr115389.c @@ -0,0 +1,17 @@ +/* PR target/115389 */ +/* { dg-do assemble } */ +/* { dg-options "-O2 -mdejagnu-cpu=power10 -mrop-protect -mno-vsx -mno-altivec -mabi=no-altivec -save-temps" } */ +/* { dg-require-effective-target rop_ok } */ + +/* Verify we do not emit invalid offsets for our ROP insns. */ + +extern void foo (void); +long +bar (void) +{ + foo (); + return 0; +} + +/* { dg-final { scan-assembler-times {\mhashst\M} 1 } } */ +/* { dg-final { scan-assembler-times {\mhashchk\M} 1 } } */ commit aa293f40770bae5e94f33d4700f2f0ce9eff712b Author: Peter Bergner Date: Wed Jun 19 16:07:29 2024 -0500 rs6000: ROP - Emit hashst and hashchk insns on Power8 and later [PR114759] We currently only emit the ROP-protect hash* insns for Power10, where the insns were added to the architecture. We want to emit them for earlier cpus (where they operate as NOPs), so that if those older binaries are ever executed on a Power10, then they'll be protected from ROP attacks. Binutils accepts hashst and hashchk back to Power8, so change GCC to emit them for Power8 and later. This matches clang's behavior. 2024-06-19 Peter Bergner gcc/ PR target/114759 * config/rs6000/rs6000-logue.cc (rs6000_stack_info): Use TARGET_POWER8. (rs6000_emit_prologue): Likewise. * config/rs6000/rs6000.md (hashchk): Likewise. (hashst): Likewise. Fix whitespace. gcc/testsuite/ PR target/114759 * gcc.target/powerpc/pr114759-2.c: New test. * lib/target-supports.exp (rop_ok): Use check_effective_target_has_arch_pwr8. (cherry picked from commit a05c3d23d1e1c8d2971b123804fc7a61a3561adb) diff --git a/gcc/config/rs6000/rs6000-logue.cc b/gcc/config/rs6000/rs6000-logue.cc index 3894bd23d17..9817ce78639 100644 --- a/gcc/config/rs6000/rs6000-logue.cc +++ b/gcc/config/rs6000/rs6000-logue.cc @@ -720,7 +720,7 @@ rs6000_stack_info (void) info->calls_p = (!crtl->is_leaf || cfun->machine->ra_needs_full_frame); info->rop_hash_size = 0; - if (TARGET_POWER10 + if (TARGET_POWER8 && info->calls_p && DEFAULT_ABI == ABI_ELFv2 && rs6000_rop_protect) @@ -3279,7 +3279,7 @@ rs6000_emit_prologue (void) /* NOTE: The hashst isn't needed if we're going to do a sibcall, but there's no way to know that here. Harmless except for performance, of course. */ - if (TARGET_POWER10 && rs6000_rop_protect && info->rop_hash_size != 0) + if (TARGET_POWER8 && rs6000_rop_protect && info->rop_hash_size != 0) { gcc_assert (DEFAULT_ABI == ABI_ELFv2); rtx stack_ptr = gen_rtx_REG (Pmode, STACK_POINTER_REGNUM); @@ -5026,7 +5026,7 @@ rs6000_emit_epilogue (enum epilogue_type epilogue_type) /* The ROP hash check must occur after the stack pointer is restored (since the hash involves r1), and is not performed for a sibcall. */ - if (TARGET_POWER10 + if (TARGET_POWER8 && rs6000_rop_protect && info->rop_hash_size != 0 && epilogue_type != EPILOGUE_TYPE_SIBCALL) diff --git a/gcc/config/rs6000/rs6000.md b/gcc/config/rs6000/rs6000.md index c38bebde185..d1ae5be96d4 100644 --- a/gcc/config/rs6000/rs6000.md +++ b/gcc/config/rs6000/rs6000.md @@ -15557,9 +15557,9 @@ (define_insn "hashst" [(set (match_operand:DI 0 "simple_offsettable_mem_operand" "=m") - (unspec_volatile:DI [(match_operand:DI 1 "int_reg_operand" "r")] + (unspec_volatile:DI [(match_operand:DI 1 "int_reg_operand" "r")] UNSPEC_HASHST))] - "TARGET_POWER10 && rs6000_rop_protect" + "TARGET_POWER8 && rs6000_rop_protect" { static char templ[32]; const char *p = rs6000_privileged ? "p" : ""; @@ -15572,7 +15572,7 @@ [(unspec_volatile [(match_operand:DI 0 "int_reg_operand" "r") (match_operand:DI 1 "simple_offsettable_mem_operand" "m")] UNSPEC_HASHCHK)] - "TARGET_POWER10 && rs6000_rop_protect" + "TARGET_POWER8 && rs6000_rop_protect" { static char templ[32]; const char *p = rs6000_privileged ? "p" : ""; diff --git a/gcc/testsuite/gcc.target/powerpc/pr114759-2.c b/gcc/testsuite/gcc.target/powerpc/pr114759-2.c new file mode 100644 index 00000000000..3881ebd416e --- /dev/null +++ b/gcc/testsuite/gcc.target/powerpc/pr114759-2.c @@ -0,0 +1,17 @@ +/* { dg-do compile } */ +/* { dg-options "-O2 -mdejagnu-cpu=power8 -mrop-protect" } */ +/* { dg-require-effective-target rop_ok } Only enable on supported ABIs. */ + +/* Verify we generate ROP-protect hash insns when compiling for Power8. */ + +extern void foo (void); + +int +bar (void) +{ + foo (); + return 5; +} + +/* { dg-final { scan-assembler-times {\mhashst\M} 1 } } */ +/* { dg-final { scan-assembler-times {\mhashchk\M} 1 } } */ diff --git a/gcc/testsuite/lib/target-supports.exp b/gcc/testsuite/lib/target-supports.exp index 64216dfbdb2..ee2fa5ed8a6 100644 --- a/gcc/testsuite/lib/target-supports.exp +++ b/gcc/testsuite/lib/target-supports.exp @@ -6752,7 +6752,7 @@ proc check_effective_target_powerpc_elfv2 { } { # Return 1 if this is a PowerPC target supporting -mrop-protect proc check_effective_target_rop_ok { } { - return [check_effective_target_power10_ok] && [check_effective_target_powerpc_elfv2] + return [check_effective_target_has_arch_pwr8] && [check_effective_target_powerpc_elfv2] } # The VxWorks SPARC simulator accepts only EM_SPARC executables and commit 25cf4d2a2200903fe868f8cbd9d24f35768041c1 Author: Peter Bergner Date: Mon Jul 15 16:57:32 2024 -0500 rs6000: Error on CPUs and ABIs that don't support the ROP protection insns [PR114759] We currently silently ignore the -mrop-protect option for old CPUs we don't support with the ROP hash insns, but we throw an error for unsupported ABIs. This patch treats unsupported CPUs and ABIs similarly by throwing an error both both. This matches clang behavior and allows us to simplify our tests in the code that generates our prologue and epilogue code. 2024-06-26 Peter Bergner gcc/ PR target/114759 * config/rs6000/rs6000.cc (rs6000_option_override_internal): Disallow CPUs and ABIs that do no support the ROP protection insns. * config/rs6000/rs6000-logue.cc (rs6000_stack_info): Remove now unneeded tests. (rs6000_emit_prologue): Likewise. Remove unneeded gcc_assert. (rs6000_emit_epilogue): Likewise. * config/rs6000/rs6000.md: Likewise. gcc/testsuite/ PR target/114759 * gcc.target/powerpc/pr114759-3.c: New test. (cherry picked from commit 6f2bab9b5d1ce1914c748b7dcd8638dafaa98df7) diff --git a/gcc/config/rs6000/rs6000-logue.cc b/gcc/config/rs6000/rs6000-logue.cc index 9817ce78639..d891d43c074 100644 --- a/gcc/config/rs6000/rs6000-logue.cc +++ b/gcc/config/rs6000/rs6000-logue.cc @@ -720,17 +720,11 @@ rs6000_stack_info (void) info->calls_p = (!crtl->is_leaf || cfun->machine->ra_needs_full_frame); info->rop_hash_size = 0; - if (TARGET_POWER8 - && info->calls_p - && DEFAULT_ABI == ABI_ELFv2 - && rs6000_rop_protect) + /* If we want ROP protection and this function makes a call, indicate + we need to create a stack slot to save the hashed return address in. */ + if (rs6000_rop_protect + && info->calls_p) info->rop_hash_size = 8; - else if (rs6000_rop_protect && DEFAULT_ABI != ABI_ELFv2) - { - /* We can't check this in rs6000_option_override_internal since - DEFAULT_ABI isn't established yet. */ - error ("%qs requires the ELFv2 ABI", "-mrop-protect"); - } /* Determine if we need to save the condition code registers. */ if (save_reg_p (CR2_REGNO) @@ -3279,9 +3273,8 @@ rs6000_emit_prologue (void) /* NOTE: The hashst isn't needed if we're going to do a sibcall, but there's no way to know that here. Harmless except for performance, of course. */ - if (TARGET_POWER8 && rs6000_rop_protect && info->rop_hash_size != 0) + if (info->rop_hash_size) { - gcc_assert (DEFAULT_ABI == ABI_ELFv2); rtx stack_ptr = gen_rtx_REG (Pmode, STACK_POINTER_REGNUM); rtx addr = gen_rtx_PLUS (Pmode, stack_ptr, GEN_INT (info->rop_hash_save_offset)); @@ -5026,12 +5019,9 @@ rs6000_emit_epilogue (enum epilogue_type epilogue_type) /* The ROP hash check must occur after the stack pointer is restored (since the hash involves r1), and is not performed for a sibcall. */ - if (TARGET_POWER8 - && rs6000_rop_protect - && info->rop_hash_size != 0 + if (info->rop_hash_size && epilogue_type != EPILOGUE_TYPE_SIBCALL) { - gcc_assert (DEFAULT_ABI == ABI_ELFv2); rtx stack_ptr = gen_rtx_REG (Pmode, STACK_POINTER_REGNUM); rtx addr = gen_rtx_PLUS (Pmode, stack_ptr, GEN_INT (info->rop_hash_save_offset)); diff --git a/gcc/config/rs6000/rs6000.cc b/gcc/config/rs6000/rs6000.cc index 03893b2cf8b..cf0d089d06b 100644 --- a/gcc/config/rs6000/rs6000.cc +++ b/gcc/config/rs6000/rs6000.cc @@ -4888,6 +4888,18 @@ rs6000_option_override_internal (bool global_init_p) rs6000_print_builtin_options (stderr, 0, "builtin mask", rs6000_builtin_mask); + /* We only support ROP protection on certain targets. */ + if (rs6000_rop_protect) + { + /* Disallow CPU targets we don't support. */ + if (!TARGET_POWER8) + error ("%<-mrop-protect%> requires %<-mcpu=power8%> or later"); + + /* Disallow ABI targets we don't support. */ + if (DEFAULT_ABI != ABI_ELFv2) + error ("%<-mrop-protect%> requires the ELFv2 ABI"); + } + /* Initialize all of the registers. */ rs6000_init_hard_regno_mode_ok (global_init_p); diff --git a/gcc/config/rs6000/rs6000.md b/gcc/config/rs6000/rs6000.md index d1ae5be96d4..b0614868f94 100644 --- a/gcc/config/rs6000/rs6000.md +++ b/gcc/config/rs6000/rs6000.md @@ -15559,7 +15559,7 @@ [(set (match_operand:DI 0 "simple_offsettable_mem_operand" "=m") (unspec_volatile:DI [(match_operand:DI 1 "int_reg_operand" "r")] UNSPEC_HASHST))] - "TARGET_POWER8 && rs6000_rop_protect" + "rs6000_rop_protect" { static char templ[32]; const char *p = rs6000_privileged ? "p" : ""; @@ -15572,7 +15572,7 @@ [(unspec_volatile [(match_operand:DI 0 "int_reg_operand" "r") (match_operand:DI 1 "simple_offsettable_mem_operand" "m")] UNSPEC_HASHCHK)] - "TARGET_POWER8 && rs6000_rop_protect" + "rs6000_rop_protect" { static char templ[32]; const char *p = rs6000_privileged ? "p" : ""; diff --git a/gcc/testsuite/gcc.target/powerpc/pr114759-3.c b/gcc/testsuite/gcc.target/powerpc/pr114759-3.c new file mode 100644 index 00000000000..6770a9aec3b --- /dev/null +++ b/gcc/testsuite/gcc.target/powerpc/pr114759-3.c @@ -0,0 +1,19 @@ +/* PR target/114759 */ +/* { dg-do compile } */ +/* { dg-options "-O2 -mdejagnu-cpu=power7 -mrop-protect" } */ + +/* Verify we emit an error if we use -mrop-protect with an unsupported cpu. */ + +extern void foo (void); + +int +bar (void) +{ + foo (); + return 5; +} + +/* The correct line number is in the preamble to the error message, not + in the final line (which is all that dg-error inspects). Hence, we have + to tell dg-error to ignore the line number. */ +/* { dg-error "'-mrop-protect' requires '-mcpu=power8'" "PR114759" { target *-*-* } 0 } */ commit f7bebf4c07dffaa75c77152e8004aa0ccbf6eeac Author: Peter Bergner Date: Thu Jul 18 18:01:46 2024 -0500 rs6000: Catch unsupported ABI errors when using -mrop-protect [PR114759,PR115988] 2024-07-18 Peter Bergner gcc/testsuite/ PR target/114759 PR target/115988 * gcc.target/powerpc/pr114759-3.c: Catch unsupported ABI errors. (cherry picked from commit b2f47a5c1d5204131660ea0372a08e692df8844e) diff --git a/gcc/testsuite/gcc.target/powerpc/pr114759-3.c b/gcc/testsuite/gcc.target/powerpc/pr114759-3.c index 6770a9aec3b..e2f1d42e111 100644 --- a/gcc/testsuite/gcc.target/powerpc/pr114759-3.c +++ b/gcc/testsuite/gcc.target/powerpc/pr114759-3.c @@ -2,7 +2,8 @@ /* { dg-do compile } */ /* { dg-options "-O2 -mdejagnu-cpu=power7 -mrop-protect" } */ -/* Verify we emit an error if we use -mrop-protect with an unsupported cpu. */ +/* Verify we emit an error if we use -mrop-protect with an unsupported cpu + or ABI. */ extern void foo (void); @@ -17,3 +18,4 @@ bar (void) in the final line (which is all that dg-error inspects). Hence, we have to tell dg-error to ignore the line number. */ /* { dg-error "'-mrop-protect' requires '-mcpu=power8'" "PR114759" { target *-*-* } 0 } */ +/* { dg-error "'-mrop-protect' requires the ELFv2 ABI" "PR114759" { target { ! rop_ok } } 0 } */ commit 7149e22fe92581f54bf89855c996ea83bdf773ef Author: GCC Administrator Date: Thu Jul 25 00:21:26 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 645d8748256..d54c47b047f 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,46 @@ +2024-07-24 Peter Bergner + + Backported from master: + 2024-07-17 Peter Bergner + + PR target/114759 + * config/rs6000/rs6000.cc (rs6000_option_override_internal): Disallow + CPUs and ABIs that do no support the ROP protection insns. + * config/rs6000/rs6000-logue.cc (rs6000_stack_info): Remove now + unneeded tests. + (rs6000_emit_prologue): Likewise. + Remove unneeded gcc_assert. + (rs6000_emit_epilogue): Likewise. + * config/rs6000/rs6000.md: Likewise. + +2024-07-24 Peter Bergner + + Backported from master: + 2024-07-17 Peter Bergner + + PR target/114759 + * config/rs6000/rs6000-logue.cc (rs6000_stack_info): Use TARGET_POWER8. + (rs6000_emit_prologue): Likewise. + * config/rs6000/rs6000.md (hashchk): Likewise. + (hashst): Likewise. + Fix whitespace. + +2024-07-24 Peter Bergner + + Backported from master: + 2024-06-17 Peter Bergner + + PR target/115389 + * config/rs6000/rs6000-logue.cc (rs6000_stack_info): Compute + rop_hash_save_offset for non-Altivec compiles. + +2024-07-24 Peter Bergner + + Backported from master: + 2024-06-08 Peter Bergner + + * config/rs6000/rs6000-logue.cc (rs6000_stack_info): Update comment. + 2024-07-23 Richard Biener Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d050dd67721..503038db09c 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240724 +20240725 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 6357df9e57d..aea572a67be 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,38 @@ +2024-07-24 Peter Bergner + + Backported from master: + 2024-07-19 Peter Bergner + + PR target/114759 + PR target/115988 + * gcc.target/powerpc/pr114759-3.c: Catch unsupported ABI errors. + +2024-07-24 Peter Bergner + + Backported from master: + 2024-07-17 Peter Bergner + + PR target/114759 + * gcc.target/powerpc/pr114759-3.c: New test. + +2024-07-24 Peter Bergner + + Backported from master: + 2024-07-17 Peter Bergner + + PR target/114759 + * gcc.target/powerpc/pr114759-2.c: New test. + * lib/target-supports.exp (rop_ok): Use + check_effective_target_has_arch_pwr8. + +2024-07-24 Peter Bergner + + Backported from master: + 2024-06-17 Peter Bergner + + PR target/115389 + * gcc.target/powerpc/pr115389.c: New test. + 2024-07-23 Alexandre Oliva Backported from master: commit e8beb05bc9e6c84ea2250754cce8951ebda58e5c Author: GCC Administrator Date: Fri Jul 26 00:19:49 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 503038db09c..c8f8a5f000a 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240725 +20240726 commit 85e3fd8d6efdbc78831f70392a363386cba71952 Author: GCC Administrator Date: Sat Jul 27 00:19:32 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c8f8a5f000a..3eff065eb0c 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240726 +20240727 commit b110b667c14b7c933a45e5f7291881c46074f4fb Author: GCC Administrator Date: Sun Jul 28 00:20:21 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 3eff065eb0c..65b8ef17038 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240727 +20240728 commit fc5b4e94ee804936fe9a6834d0de1b99f7cd3f12 Author: GCC Administrator Date: Mon Jul 29 00:19:56 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 65b8ef17038..d1b40fa5c77 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240728 +20240729 commit bd0fbdc30d831f8c84223f583bcb5e5f6d7d72fc Author: Haochen Jiang Date: Thu Jul 25 16:12:20 2024 +0800 i386: Fix AVX512 intrin macro typo There are several typo in AVX512 intrins macro define. Correct them to solve errors when compiled with -O0. gcc/ChangeLog: * config/i386/avx512dqintrin.h (_mm_mask_fpclass_ss_mask): Correct operand order. (_mm_mask_fpclass_sd_mask): Ditto. (_mm256_maskz_reduce_round_ss): Use __builtin_ia32_reducess_mask_round instead of __builtin_ia32_reducesd_mask_round. (_mm_reduce_round_sd): Use -1 as mask since it is non-mask. (_mm_reduce_round_ss): Ditto. * config/i386/avx512vlbwintrin.h (_mm256_mask_alignr_epi8): Correct operand usage. (_mm_mask_alignr_epi8): Ditto. * config/i386/avx512vlintrin.h (_mm_mask_alignr_epi64): Ditto. gcc/testsuite/ChangeLog: * gcc.target/i386/avx512bw-vpalignr-1b.c: New test. * gcc.target/i386/avx512dq-vfpclasssd-1b.c: Ditto. * gcc.target/i386/avx512dq-vfpclassss-1b.c: Ditto. * gcc.target/i386/avx512dq-vreducesd-1b.c: Ditto. * gcc.target/i386/avx512dq-vreducess-1b.c: Ditto. * gcc.target/i386/avx512vl-valignq-1b.c: Ditto. diff --git a/gcc/config/i386/avx512dqintrin.h b/gcc/config/i386/avx512dqintrin.h index e924250a4ad..4f9451e949b 100644 --- a/gcc/config/i386/avx512dqintrin.h +++ b/gcc/config/i386/avx512dqintrin.h @@ -2800,11 +2800,11 @@ _mm512_fpclass_ps_mask (__m512 __A, const int __imm) ((__mmask8) __builtin_ia32_fpclasssd_mask ((__v2df) (__m128d) (X), \ (int) (C), (__mmask8) (-1))) \ -#define _mm_mask_fpclass_ss_mask(X, C, U) \ +#define _mm_mask_fpclass_ss_mask(U, X, C) \ ((__mmask8) __builtin_ia32_fpclassss_mask ((__v4sf) (__m128) (X), \ (int) (C), (__mmask8) (U))) -#define _mm_mask_fpclass_sd_mask(X, C, U) \ +#define _mm_mask_fpclass_sd_mask(U, X, C) \ ((__mmask8) __builtin_ia32_fpclasssd_mask ((__v2df) (__m128d) (X), \ (int) (C), (__mmask8) (U))) @@ -2839,8 +2839,9 @@ _mm512_fpclass_ps_mask (__m512 __A, const int __imm) (__mmask8)(U))) #define _mm_reduce_round_sd(A, B, C, R) \ - ((__m128d) __builtin_ia32_reducesd_round ((__v2df)(__m128d)(A), \ - (__v2df)(__m128d)(B), (int)(C), (__mmask8)(U), (int)(R))) + ((__m128d) __builtin_ia32_reducesd_mask_round ((__v2df)(__m128d)(A), \ + (__v2df)(__m128d)(B), (int)(C), (__v2df) _mm_avx512_setzero_pd (), \ + (__mmask8)(-1), (int)(R))) #define _mm_mask_reduce_round_sd(W, U, A, B, C, R) \ ((__m128d) __builtin_ia32_reducesd_mask_round ((__v2df)(__m128d)(A), \ @@ -2867,8 +2868,9 @@ _mm512_fpclass_ps_mask (__m512 __A, const int __imm) (__mmask8)(U))) #define _mm_reduce_round_ss(A, B, C, R) \ - ((__m128) __builtin_ia32_reducess_round ((__v4sf)(__m128)(A), \ - (__v4sf)(__m128)(B), (int)(C), (__mmask8)(U), (int)(R))) + ((__m128) __builtin_ia32_reducess_mask_round ((__v4sf)(__m128)(A), \ + (__v4sf)(__m128)(B), (int)(C), (__v4sf) _mm_avx512_setzero_ps (), \ + (__mmask8)(-1), (int)(R))) #define _mm_mask_reduce_round_ss(W, U, A, B, C, R) \ ((__m128) __builtin_ia32_reducess_mask_round ((__v4sf)(__m128)(A), \ @@ -2876,7 +2878,7 @@ _mm512_fpclass_ps_mask (__m512 __A, const int __imm) (__mmask8)(U), (int)(R))) #define _mm_maskz_reduce_round_ss(U, A, B, C, R) \ - ((__m128) __builtin_ia32_reducesd_mask_round ((__v4sf)(__m128)(A), \ + ((__m128) __builtin_ia32_reducess_mask_round ((__v4sf)(__m128)(A), \ (__v4sf)(__m128)(B), (int)(C), (__v4sf) _mm_setzero_ps (), \ (__mmask8)(U), (int)(R))) diff --git a/gcc/config/i386/avx512vlbwintrin.h b/gcc/config/i386/avx512vlbwintrin.h index 192d54e743f..c918ed520c5 100644 --- a/gcc/config/i386/avx512vlbwintrin.h +++ b/gcc/config/i386/avx512vlbwintrin.h @@ -1839,7 +1839,7 @@ _mm_maskz_slli_epi16 (__mmask8 __U, __m128i __A, int __B) #define _mm256_mask_alignr_epi8(W, U, X, Y, N) \ ((__m256i) __builtin_ia32_palignr256_mask ((__v4di)(__m256i)(X), \ (__v4di)(__m256i)(Y), (int)((N) * 8), \ - (__v4di)(__m256i)(X), (__mmask32)(U))) + (__v4di)(__m256i)(W), (__mmask32)(U))) #define _mm256_mask_srli_epi16(W, U, A, B) \ ((__m256i) __builtin_ia32_psrlwi256_mask ((__v16hi)(__m256i)(A), \ @@ -1922,7 +1922,7 @@ _mm_maskz_slli_epi16 (__mmask8 __U, __m128i __A, int __B) #define _mm_mask_alignr_epi8(W, U, X, Y, N) \ ((__m128i) __builtin_ia32_palignr128_mask ((__v2di)(__m128i)(X), \ (__v2di)(__m128i)(Y), (int)((N) * 8), \ - (__v2di)(__m128i)(X), (__mmask16)(U))) + (__v2di)(__m128i)(W), (__mmask16)(U))) #define _mm_maskz_alignr_epi8(U, X, Y, N) \ ((__m128i) __builtin_ia32_palignr128_mask ((__v2di)(__m128i)(X), \ diff --git a/gcc/config/i386/avx512vlintrin.h b/gcc/config/i386/avx512vlintrin.h index 26b286eae6b..c6f3f35a009 100644 --- a/gcc/config/i386/avx512vlintrin.h +++ b/gcc/config/i386/avx512vlintrin.h @@ -13609,7 +13609,7 @@ _mm256_permutex_pd (__m256d __X, const int __M) #define _mm_mask_alignr_epi64(W, U, X, Y, C) \ ((__m128i)__builtin_ia32_alignq128_mask ((__v2di)(__m128i)(X), \ - (__v2di)(__m128i)(Y), (int)(C), (__v2di)(__m128i)(X), (__mmask8)-1)) + (__v2di)(__m128i)(Y), (int)(C), (__v2di)(__m128i)(W), (__mmask8)(U))) #define _mm_maskz_alignr_epi64(U, X, Y, C) \ ((__m128i)__builtin_ia32_alignq128_mask ((__v2di)(__m128i)(X), \ diff --git a/gcc/testsuite/gcc.target/i386/avx512bw-vpalignr-1b.c b/gcc/testsuite/gcc.target/i386/avx512bw-vpalignr-1b.c new file mode 100644 index 00000000000..2b42aa90b91 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/avx512bw-vpalignr-1b.c @@ -0,0 +1,18 @@ +/* { dg-do compile } */ +/* { dg-options "-O0 -mavx512bw -mavx512vl" } */ +/* { dg-final { scan-assembler-times "vpalignr\[ \\t\]+\[^\{\n\]*%ymm\[0-9\]+\{%k\[1-7\]\}(?:\n|\[ \\t\]+#)" 1 } } */ +/* { dg-final { scan-assembler-times "vpalignr\[ \\t\]+\[^\{\n\]*%xmm\[0-9\]+\{%k\[1-7\]\}(?:\n|\[ \\t\]+#)" 1 } } */ + +#include + +volatile __m256i y; +volatile __m128i x; +volatile __mmask32 m2; +volatile __mmask16 m3; + +void extern +avx512bw_test (void) +{ + y = _mm256_mask_alignr_epi8 (y, m2, y, y, 10); + x = _mm_mask_alignr_epi8 (x, m3, x, x, 10); +} diff --git a/gcc/testsuite/gcc.target/i386/avx512dq-vfpclasssd-1b.c b/gcc/testsuite/gcc.target/i386/avx512dq-vfpclasssd-1b.c new file mode 100644 index 00000000000..8c7f96fb7a7 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/avx512dq-vfpclasssd-1b.c @@ -0,0 +1,14 @@ +/* { dg-do compile } */ +/* { dg-options "-mavx512dq -O0" } */ +/* { dg-final { scan-assembler-times "vfpclasssd\[ \\t\]+\[^\{\n\]*%xmm\[0-9\]+\[^\n^k\]*%k\[0-7\]\{%k\[0-7\]\}(?:\n|\[ \\t\]+#)" 1 } } */ + +#include + +volatile __m128d x128; +volatile __mmask8 m8; + +void extern +avx512dq_test (void) +{ + m8 = _mm_mask_fpclass_sd_mask (m8, x128, 13); +} diff --git a/gcc/testsuite/gcc.target/i386/avx512dq-vfpclassss-1b.c b/gcc/testsuite/gcc.target/i386/avx512dq-vfpclassss-1b.c new file mode 100644 index 00000000000..3196fd60d64 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/avx512dq-vfpclassss-1b.c @@ -0,0 +1,14 @@ +/* { dg-do compile } */ +/* { dg-options "-mavx512dq -O0" } */ +/* { dg-final { scan-assembler-times "vfpclassss\[ \\t\]+\[^\{\n\]*%xmm\[0-9\]+\[^\n^k\]*%k\[0-7\]\{%k\[0-7\]\}(?:\n|\[ \\t\]+#)" 1 } } */ + +#include + +volatile __m128 x128; +volatile __mmask8 m8; + +void extern +avx512dq_test (void) +{ + m8 = _mm_mask_fpclass_ss_mask (m8, x128, 13); +} diff --git a/gcc/testsuite/gcc.target/i386/avx512dq-vreducesd-1b.c b/gcc/testsuite/gcc.target/i386/avx512dq-vreducesd-1b.c new file mode 100644 index 00000000000..9ae8259d373 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/avx512dq-vreducesd-1b.c @@ -0,0 +1,16 @@ +/* { dg-do compile } */ +/* { dg-options "-mavx512dq -O0" } */ +/* { dg-final { scan-assembler-times "vreducesd\[ \\t\]+\[^\{\n\]*\{sae\}\[^\n\]*%xmm\[0-9\]+\[^\n\]*%xmm\[0-9\]+\[^\n\]*%xmm\[0-9\]+\{%k\[1-7\]\}(?:\n|\[ \\t\]+#)" 1 } } */ + +#include + +#define IMM 123 + +volatile __m128d x1, x2, xx1, xx2; +volatile __mmask8 m; + +void extern +avx512dq_test (void) +{ + xx1 = _mm_reduce_round_sd (xx1, xx2, IMM, _MM_FROUND_NO_EXC); +} diff --git a/gcc/testsuite/gcc.target/i386/avx512dq-vreducess-1b.c b/gcc/testsuite/gcc.target/i386/avx512dq-vreducess-1b.c new file mode 100644 index 00000000000..47bf48fb617 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/avx512dq-vreducess-1b.c @@ -0,0 +1,16 @@ +/* { dg-do compile } */ +/* { dg-options "-mavx512dq -O0" } */ +/* { dg-final { scan-assembler-times "vreducess\[ \\t\]+\[^\{\n\]*\{sae\}\[^\n\]*%xmm\[0-9\]+\[^\n\]*%xmm\[0-9\]+\[^\n\]*%xmm\[0-9\]+\{%k\[1-7\]\}(?:\n|\[ \\t\]+#)" 1 } } */ + +#include + +#define IMM 123 + +volatile __m128 x1, x2, xx1, xx2; +volatile __mmask8 m; + +void extern +avx512dq_test (void) +{ + xx1 = _mm_reduce_round_ss (xx1, xx2, IMM, _MM_FROUND_NO_EXC); +} diff --git a/gcc/testsuite/gcc.target/i386/avx512vl-valignq-1b.c b/gcc/testsuite/gcc.target/i386/avx512vl-valignq-1b.c new file mode 100644 index 00000000000..0ab16b27733 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/avx512vl-valignq-1b.c @@ -0,0 +1,15 @@ +/* { dg-do compile } */ +/* { dg-options "-O0 -mavx512vl" } */ +/* { dg-final { scan-assembler-times "valignq\[ \\t\]+\[^\{\n\]*%xmm\[0-9\]+\{%k\[1-7\]\}(?:\n|\[ \\t\]+#)" 1 } } */ + +#include + +volatile __m256i y; +volatile __m128i x; +volatile __mmask8 m; + +void extern +avx512vl_test (void) +{ + x = _mm_mask_alignr_epi64 (x, m, x, x, 1); +} commit 77ad22e4eaa97bb10068c6170f53caca77c99392 Author: Haochen Jiang Date: Mon Jul 29 14:10:49 2024 +0800 i386: Use _mm_setzero_ps/d instead of _mm_avx512_setzero_ps/d for GCC13/12 In GCC13/12, there is no _mm_avx512_setzero_ps/d since it is introduced in GCC14. gcc/ChangeLog: * config/i386/avx512dqintrin.h (_mm_reduce_round_sd): Use _mm_setzero_pd instead of _mm_avx512_setzero_pd. (_mm_reduce_round_ss): Use _mm_setzero_ps instead of _mm_avx512_setzero_ps. diff --git a/gcc/config/i386/avx512dqintrin.h b/gcc/config/i386/avx512dqintrin.h index 4f9451e949b..e8f8efe3be8 100644 --- a/gcc/config/i386/avx512dqintrin.h +++ b/gcc/config/i386/avx512dqintrin.h @@ -2840,7 +2840,7 @@ _mm512_fpclass_ps_mask (__m512 __A, const int __imm) #define _mm_reduce_round_sd(A, B, C, R) \ ((__m128d) __builtin_ia32_reducesd_mask_round ((__v2df)(__m128d)(A), \ - (__v2df)(__m128d)(B), (int)(C), (__v2df) _mm_avx512_setzero_pd (), \ + (__v2df)(__m128d)(B), (int)(C), (__v2df) _mm_setzero_pd (), \ (__mmask8)(-1), (int)(R))) #define _mm_mask_reduce_round_sd(W, U, A, B, C, R) \ @@ -2869,7 +2869,7 @@ _mm512_fpclass_ps_mask (__m512 __A, const int __imm) #define _mm_reduce_round_ss(A, B, C, R) \ ((__m128) __builtin_ia32_reducess_mask_round ((__v4sf)(__m128)(A), \ - (__v4sf)(__m128)(B), (int)(C), (__v4sf) _mm_avx512_setzero_ps (), \ + (__v4sf)(__m128)(B), (int)(C), (__v4sf) _mm_setzero_ps (), \ (__mmask8)(-1), (int)(R))) #define _mm_mask_reduce_round_ss(W, U, A, B, C, R) \ commit 16ea079209ea9d699e8dbecf8a2d264775e383eb Author: GCC Administrator Date: Tue Jul 30 00:20:00 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index d54c47b047f..6d5c1ccf547 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,24 @@ +2024-07-29 Haochen Jiang + + * config/i386/avx512dqintrin.h (_mm_reduce_round_sd): Use + _mm_setzero_pd instead of _mm_avx512_setzero_pd. + (_mm_reduce_round_ss): Use _mm_setzero_ps instead of + _mm_avx512_setzero_ps. + +2024-07-29 Haochen Jiang + + * config/i386/avx512dqintrin.h + (_mm_mask_fpclass_ss_mask): Correct operand order. + (_mm_mask_fpclass_sd_mask): Ditto. + (_mm256_maskz_reduce_round_ss): Use __builtin_ia32_reducess_mask_round + instead of __builtin_ia32_reducesd_mask_round. + (_mm_reduce_round_sd): Use -1 as mask since it is non-mask. + (_mm_reduce_round_ss): Ditto. + * config/i386/avx512vlbwintrin.h + (_mm256_mask_alignr_epi8): Correct operand usage. + (_mm_mask_alignr_epi8): Ditto. + * config/i386/avx512vlintrin.h (_mm_mask_alignr_epi64): Ditto. + 2024-07-24 Peter Bergner Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d1b40fa5c77..d42ae94bcd3 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240729 +20240730 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index aea572a67be..68cde584121 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,12 @@ +2024-07-29 Haochen Jiang + + * gcc.target/i386/avx512bw-vpalignr-1b.c: New test. + * gcc.target/i386/avx512dq-vfpclasssd-1b.c: Ditto. + * gcc.target/i386/avx512dq-vfpclassss-1b.c: Ditto. + * gcc.target/i386/avx512dq-vreducesd-1b.c: Ditto. + * gcc.target/i386/avx512dq-vreducess-1b.c: Ditto. + * gcc.target/i386/avx512vl-valignq-1b.c: Ditto. + 2024-07-24 Peter Bergner Backported from master: commit 3e6b076631390dfeb9211babbc942772b52dc8e9 Author: GCC Administrator Date: Wed Jul 31 00:20:45 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d42ae94bcd3..e2dab5541a0 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240730 +20240731 commit b0137fe4af2f997d1567c5131c597fa13a2625f5 Author: GCC Administrator Date: Thu Aug 1 00:21:27 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e2dab5541a0..5b64322fc60 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240731 +20240801 commit c7d2a418a44f3d1cdc27c78b31aebec7df1f6d69 Author: GCC Administrator Date: Fri Aug 2 00:20:36 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 5b64322fc60..ba6415cbd75 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240801 +20240802 commit 77c8a37ba4db6352bced3b0cde2f4aff04884241 Author: GCC Administrator Date: Sat Aug 3 00:19:48 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index ba6415cbd75..613753ee8b5 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240802 +20240803 commit 2c6bfcd3d96c47602748af9595401b11bbe6b93e Author: GCC Administrator Date: Sun Aug 4 00:18:24 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 613753ee8b5..444c7cbae43 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240803 +20240804 commit 13e6f13286e105d9e2898767ad1f4125e13f9514 Author: GCC Administrator Date: Mon Aug 5 00:18:15 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 444c7cbae43..25c46364131 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240804 +20240805 commit 0e945f6e8849ae4722ea7ac70d713f7b35d3fade Author: Paul Thomas Date: Thu Jul 18 08:51:35 2024 +0100 Fortran: Suppress bogus used uninitialized warnings [PR108889]. 2024-07-18 Paul Thomas gcc/fortran PR fortran/108889 * gfortran.h: Add bit field 'allocated_in_scope' to gfc_symbol. * trans-array.cc (gfc_array_allocate): Set 'allocated_in_scope' after allocation if not a component reference. (gfc_alloc_allocatable_for_assignment): If 'allocated_in_scope' not set, not a component ref and not allocated, set the array bounds and offset to give zero length in all dimensions. Then set allocated_in_scope. gcc/testsuite/ PR fortran/108889 * gfortran.dg/pr108889.f90: New test. (cherry picked from commit c3aa339ea50f050caf7ed2e497f5499ec2d7b9cc) diff --git a/gcc/fortran/gfortran.h b/gcc/fortran/gfortran.h index 0b0a8fe7118..7162f39f39c 100644 --- a/gcc/fortran/gfortran.h +++ b/gcc/fortran/gfortran.h @@ -1887,6 +1887,10 @@ typedef struct gfc_symbol /* Set if this should be passed by value, but is not a VALUE argument according to the Fortran standard. */ unsigned pass_as_value:1; + /* Set if an allocatable array variable has been allocated in the current + scope. Used in the suppression of uninitialized warnings in reallocation + on assignment. */ + unsigned allocated_in_scope:1; int refs; struct gfc_namespace *ns; /* namespace containing this symbol */ diff --git a/gcc/fortran/trans-array.cc b/gcc/fortran/trans-array.cc index 85c641b55c5..59668177bbf 100644 --- a/gcc/fortran/trans-array.cc +++ b/gcc/fortran/trans-array.cc @@ -6285,6 +6285,8 @@ gfc_array_allocate (gfc_se * se, gfc_expr * expr, tree status, tree errmsg, else gfc_add_expr_to_block (&se->pre, set_descriptor); + expr->symtree->n.sym->allocated_in_scope = 1; + return true; } @@ -10509,6 +10511,8 @@ gfc_alloc_allocatable_for_assignment (gfc_loopinfo *loop, stmtblock_t realloc_block; stmtblock_t alloc_block; stmtblock_t fblock; + stmtblock_t loop_pre_block; + gfc_ref *ref; gfc_ss *rss; gfc_ss *lss; gfc_array_info *linfo; @@ -10717,6 +10721,45 @@ gfc_alloc_allocatable_for_assignment (gfc_loopinfo *loop, else cond_null= gfc_evaluate_now (cond_null, &fblock); + /* If the data is null, set the descriptor bounds and offset. This suppresses + the maybe used uninitialized warning and forces the use of malloc because + the size is zero in all dimensions. Note that this block is only executed + if the lhs is unallocated and is only applied once in any namespace. + Component references are not subject to the warnings. */ + for (ref = expr1->ref; ref; ref = ref->next) + if (ref->type == REF_COMPONENT) + break; + + if (!expr1->symtree->n.sym->allocated_in_scope && !ref) + { + gfc_start_block (&loop_pre_block); + for (n = 0; n < expr1->rank; n++) + { + gfc_conv_descriptor_lbound_set (&loop_pre_block, desc, + gfc_rank_cst[n], + gfc_index_one_node); + gfc_conv_descriptor_ubound_set (&loop_pre_block, desc, + gfc_rank_cst[n], + gfc_index_zero_node); + gfc_conv_descriptor_stride_set (&loop_pre_block, desc, + gfc_rank_cst[n], + gfc_index_zero_node); + } + + tmp = gfc_conv_descriptor_offset (desc); + gfc_add_modify (&loop_pre_block, tmp, gfc_index_zero_node); + + tmp = fold_build2_loc (input_location, EQ_EXPR, + logical_type_node, array1, + build_int_cst (TREE_TYPE (array1), 0)); + tmp = build3_v (COND_EXPR, tmp, + gfc_finish_block (&loop_pre_block), + build_empty_stmt (input_location)); + gfc_prepend_expr_to_block (&loop->pre, tmp); + + expr1->symtree->n.sym->allocated_in_scope = 1; + } + tmp = build3_v (COND_EXPR, cond_null, build1_v (GOTO_EXPR, jump_label1), build_empty_stmt (input_location)); diff --git a/gcc/testsuite/gfortran.dg/pr108889.f90 b/gcc/testsuite/gfortran.dg/pr108889.f90 new file mode 100644 index 00000000000..7fd4e3882a4 --- /dev/null +++ b/gcc/testsuite/gfortran.dg/pr108889.f90 @@ -0,0 +1,43 @@ +! { dg-do compile } +! { dg-options "-Wall -fdump-tree-original" } +! +! Contributed by Tobias Burnus +! +program main + implicit none + + type :: struct + real, allocatable :: var(:) + end type struct + + type(struct) :: single + real, allocatable :: ref1(:), ref2(:), ref3(:), ref4(:) + + ref2 = [1,2,3,4,5] ! Warnings here + + single%var = ref2 ! No warnings for components + ref1 = single%var ! Warnings here + ref1 = [1,2,3,4,5] ! Should not add to tree dump count + + allocate (ref3(5)) + ref3 = single%var ! No warnings following allocation + + call set_ref4 + + call test (ref1) + call test (ref2) + call test (ref3) + call test (ref4) + +contains + subroutine test (arg) + real, allocatable :: arg(:) + if (size(arg) /= size(single%var)) stop 1 + if (lbound(arg, 1) /= 1) stop 2 + if (any (arg /= single%var)) stop 3 + end + subroutine set_ref4 + ref4 = single%var ! Warnings in contained scope + end +end +! { df-final { scan-tree-dump-times "ubound = 0" 3 "original" } } \ No newline at end of file commit 3f356a88d6c15d0ea93a5191c23b668744254f72 Author: Paul Thomas Date: Fri Jul 19 16:58:33 2024 +0100 libgomp: Remove bogus warnings from privatized-ref-2.f90. 2024-07-19 Paul Thomas libgomp/ChangeLog * testsuite/libgomp.oacc-fortran/privatized-ref-2.f90: Cut dg-note about 'a' and remove bogus warnings about its array descriptor components being used uninitialized. (cherry picked from commit 8d6994f33a98a168151a57a3d21395b19196cd9d) diff --git a/libgomp/testsuite/libgomp.oacc-fortran/privatized-ref-2.f90 b/libgomp/testsuite/libgomp.oacc-fortran/privatized-ref-2.f90 index 498ef70b63a..8cf79a10e8d 100644 --- a/libgomp/testsuite/libgomp.oacc-fortran/privatized-ref-2.f90 +++ b/libgomp/testsuite/libgomp.oacc-fortran/privatized-ref-2.f90 @@ -29,16 +29,10 @@ program main implicit none (type, external) integer :: j integer, allocatable :: A(:) - ! { dg-note {'a' declared here} {} { target *-*-* } .-1 } character(len=:), allocatable :: my_str character(len=15), allocatable :: my_str15 A = [(3*j, j=1, 10)] - ! { dg-bogus {'a\.offset' is used uninitialized} {PR77504 etc.} { xfail *-*-* } .-1 } - ! { dg-bogus {'a\.dim\[0\]\.lbound' is used uninitialized} {PR77504 etc.} { xfail *-*-* } .-2 } - ! { dg-bogus {'a\.dim\[0\]\.ubound' is used uninitialized} {PR77504 etc.} { xfail *-*-* } .-3 } - ! { dg-bogus {'a\.dim\[0\]\.lbound' may be used uninitialized} {PR77504 etc.} { xfail { ! __OPTIMIZE__ } } .-4 } - ! { dg-bogus {'a\.dim\[0\]\.ubound' may be used uninitialized} {PR77504 etc.} { xfail { ! __OPTIMIZE__ } } .-5 } call foo (A, size(A)) call bar (A) my_str = "1234567890" commit 0e2a2b93494d522775b853f4ebcb29ff84dd685d Author: GCC Administrator Date: Tue Aug 6 00:18:50 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 25c46364131..f2c439e6ce5 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240805 +20240806 diff --git a/gcc/fortran/ChangeLog b/gcc/fortran/ChangeLog index 089fbce18c4..1f360364d33 100644 --- a/gcc/fortran/ChangeLog +++ b/gcc/fortran/ChangeLog @@ -1,3 +1,17 @@ +2024-08-05 Paul Thomas + + Backported from master: + 2024-07-18 Paul Thomas + + PR fortran/108889 + * gfortran.h: Add bit field 'allocated_in_scope' to gfc_symbol. + * trans-array.cc (gfc_array_allocate): Set 'allocated_in_scope' + after allocation if not a component reference. + (gfc_alloc_allocatable_for_assignment): If 'allocated_in_scope' + not set, not a component ref and not allocated, set the array + bounds and offset to give zero length in all dimensions. Then + set allocated_in_scope. + 2024-07-21 Harald Anlauf Backported from master: diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 68cde584121..14be547c984 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-08-05 Paul Thomas + + Backported from master: + 2024-07-18 Paul Thomas + + PR fortran/108889 + * gfortran.dg/pr108889.f90: New test. + 2024-07-29 Haochen Jiang * gcc.target/i386/avx512bw-vpalignr-1b.c: New test. diff --git a/libgomp/ChangeLog b/libgomp/ChangeLog index e59af6335b9..37c1c85f3a0 100644 --- a/libgomp/ChangeLog +++ b/libgomp/ChangeLog @@ -1,3 +1,12 @@ +2024-08-05 Paul Thomas + + Backported from master: + 2024-07-19 Paul Thomas + + * testsuite/libgomp.oacc-fortran/privatized-ref-2.f90: Cut + dg-note about 'a' and remove bogus warnings about its array + descriptor components being used uninitialized. + 2024-06-20 Release Manager * GCC 12.4.0 released. commit dfacc021c9775b1563c717cf3f8114d0f874b030 Author: Andrew Pinski Date: Sat Aug 3 09:30:57 2024 -0700 sh: Don't call make_insn_raw in sh_recog_treg_set_expr [PR116189] This was an interesting compare debug failure to debug. The first symptom was in gcse which would produce different order of creating psedu-registers. This was caused by a different order of a hashtable walk, due to the hash table having different number of entries. Which in turn was due to the number of max insn being different between the 2 runs. The place max insn uid comes from was in sh_recog_treg_set_expr which is called via rtx_costs and fwprop would cause rtx_costs in some cases for debug insn related stuff. Build and tested for sh4-linux-gnu. PR target/116189 gcc/ChangeLog: * config/sh/sh.cc (sh_recog_treg_set_expr): Don't call make_insn_raw, make the insn with a fake uid. gcc/testsuite/ChangeLog: * c-c++-common/torture/pr116189-1.c: New test. Signed-off-by: Andrew Pinski (cherry picked from commit 0355c943b9e954e8f59068971d934f1b91ecb729) diff --git a/gcc/config/sh/sh.cc b/gcc/config/sh/sh.cc index 494e4536251..5455c3fddaa 100644 --- a/gcc/config/sh/sh.cc +++ b/gcc/config/sh/sh.cc @@ -12278,7 +12278,17 @@ sh_recog_treg_set_expr (rtx op, machine_mode mode) have to capture its current state and restore it afterwards. */ recog_data_d prev_recog_data = recog_data; - rtx_insn* i = make_insn_raw (gen_rtx_SET (get_t_reg_rtx (), op)); + /* Note we can't use insn_raw here since that increases the uid + and could cause debug compare differences; this insn never leaves + this function so create a dummy one. */ + rtx_insn* i = as_a (rtx_alloc (INSN)); + + INSN_UID (i) = 1; + PATTERN (i) = gen_rtx_SET (get_t_reg_rtx (), op); + INSN_CODE (i) = -1; + REG_NOTES (i) = NULL; + INSN_LOCATION (i) = curr_insn_location (); + BLOCK_FOR_INSN (i) = NULL; SET_PREV_INSN (i) = NULL; SET_NEXT_INSN (i) = NULL; diff --git a/gcc/testsuite/c-c++-common/torture/pr116189-1.c b/gcc/testsuite/c-c++-common/torture/pr116189-1.c new file mode 100644 index 00000000000..055c563f43e --- /dev/null +++ b/gcc/testsuite/c-c++-common/torture/pr116189-1.c @@ -0,0 +1,30 @@ +/* { dg-additional-options "-fcompare-debug" } */ + +/* PR target/116189 */ + +/* In the sh backend, we used to create insn in the path of rtx_costs. + This means sometimes the max uid for insns would be different between + debugging and non debugging which then would cause gcse's hashtable + to have different number of slots which would cause a different walk + for that hash table. */ + +extern void ff(void); +extern short nn[8][4]; +typedef unsigned short move_table[4]; +extern signed long long ira_overall_cost; +extern signed long long ira_load_cost; +extern move_table *x_ira_register_move_cost[1]; +struct move { struct move *next; }; +unsigned short t; +void emit_move_list(struct move * list, int freq, unsigned char mode, int regno) { + int cost; + for (; list != 0; list = list->next) + { + ff(); + unsigned short aclass = t; + cost = (nn)[mode][aclass] ; + ira_load_cost = cost; + cost = x_ira_register_move_cost[mode][aclass][aclass] * freq ; + ira_overall_cost = cost; + } +} commit 960e42bba8d276ea31cb0b37acfa8f3739d55f1d Author: John David Anglin Date: Tue Aug 6 13:40:26 2024 -0400 hppa: Fix (plus (plus (mult (a) (mem_shadd_constant)) (b)) (c)) optimization The constant C must be an integral multiple of the shift value in the above optimization. Non integral values can occur evaluating IMAGPART_EXPR when the shadd constant is 8 and we have SFmode. 2024-08-06 John David Anglin gcc/ChangeLog: PR target/113384 * config/pa/pa.cc (hppa_legitimize_address): Add check to ensure constant is an integral multiple of shift the value. diff --git a/gcc/config/pa/pa.cc b/gcc/config/pa/pa.cc index ec81f403a01..bd4dcc4e2b3 100644 --- a/gcc/config/pa/pa.cc +++ b/gcc/config/pa/pa.cc @@ -1263,6 +1263,7 @@ hppa_legitimize_address (rtx x, rtx oldx ATTRIBUTE_UNUSED, /* If the index adds a large constant, try to scale the constant so that it can be loaded with only one insn. */ if (GET_CODE (XEXP (idx, 1)) == CONST_INT + && INTVAL (XEXP (idx, 1)) % (1 << shift_val) == 0 && VAL_14_BITS_P (INTVAL (XEXP (idx, 1)) / INTVAL (XEXP (XEXP (idx, 0), 1))) && INTVAL (XEXP (idx, 1)) % INTVAL (XEXP (XEXP (idx, 0), 1)) == 0) commit df772cc4eaf4a5b2b848ab6fe909ab570dab115b Author: GCC Administrator Date: Wed Aug 7 00:18:20 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 6d5c1ccf547..c6bde7bd3e7 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,18 @@ +2024-08-06 John David Anglin + + PR target/113384 + * config/pa/pa.cc (hppa_legitimize_address): Add check to + ensure constant is an integral multiple of shift the value. + +2024-08-06 Andrew Pinski + + Backported from master: + 2024-08-06 Andrew Pinski + + PR target/116189 + * config/sh/sh.cc (sh_recog_treg_set_expr): Don't call make_insn_raw, + make the insn with a fake uid. + 2024-07-29 Haochen Jiang * config/i386/avx512dqintrin.h (_mm_reduce_round_sd): Use diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index f2c439e6ce5..0795b809c24 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240806 +20240807 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 14be547c984..fde67e61e70 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-08-06 Andrew Pinski + + Backported from master: + 2024-08-06 Andrew Pinski + + PR target/116189 + * c-c++-common/torture/pr116189-1.c: New test. + 2024-08-05 Paul Thomas Backported from master: commit 2ad59ea584f3f8f8f6714e65d68fdcdd6ed9e55f Author: GCC Administrator Date: Thu Aug 8 00:20:19 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 0795b809c24..e2d907af701 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240807 +20240808 commit 07d74871aa1384b12ce1e674e230e38e43619342 Author: GCC Administrator Date: Fri Aug 9 00:20:31 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e2d907af701..17534e79088 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240808 +20240809 commit 3a3c11c6cf99ee4e821f592f9de8f934b0ad33eb Author: GCC Administrator Date: Sat Aug 10 00:20:01 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 17534e79088..5d474ddc6cc 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240809 +20240810 commit 4b81821d685443149d62e065284065c9d81489fb Author: GCC Administrator Date: Sun Aug 11 00:19:37 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 5d474ddc6cc..c3de54be1f1 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240810 +20240811 commit 0a593976d7ff5bab36b058024c0f1bfd72913247 Author: GCC Administrator Date: Mon Aug 12 00:20:04 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c3de54be1f1..2454f4b3e28 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240811 +20240812 commit c94738e2462ff46f3013f6270f6a955b749d82b2 Author: liuhongt Date: Wed Jul 24 11:29:23 2024 +0800 Refine constraint "Bk" to define_special_memory_constraint. For below pattern, RA may still allocate r162 as v/k register, try to reload for address with leaq __libc_tsd_CTYPE_B@gottpoff(%rip), %rsi which result a linker error. (set (reg:DI 162) (mem/u/c:DI (const:DI (unspec:DI [(symbol_ref:DI ("a") [flags 0x60] )] UNSPEC_GOTNTPOFF)) Quote from H.J for why linker issue an error. >What do these do: > > leaq __libc_tsd_CTYPE_B@gottpoff(%rip), %rax > vmovq (%rax), %xmm0 > >From x86-64 TLS psABI: > >The assembler generates for the x@gottpoff(%rip) expressions a R X86 >64 GOTTPOFF relocation for the symbol x which requests the linker to >generate a GOT entry with a R X86 64 TPOFF64 relocation. The offset of >the GOT entry relative to the end of the instruction is then used in >the instruction. The R X86 64 TPOFF64 relocation is pro- cessed at >program startup time by the dynamic linker by looking up the symbol x >in the modules loaded at that point. The offset is written in the GOT >entry and later loaded by the addq instruction. > >The above code sequence looks wrong to me. gcc/ChangeLog: PR target/116043 * config/i386/constraints.md (Bk): Refine to define_special_memory_constraint. gcc/testsuite/ChangeLog: * gcc.target/i386/pr116043.c: New test. (cherry picked from commit bc1fda00d5f20e2f3e77a50b2822562b6e0040b2) diff --git a/gcc/config/i386/constraints.md b/gcc/config/i386/constraints.md index 7361687632f..e4b66340589 100644 --- a/gcc/config/i386/constraints.md +++ b/gcc/config/i386/constraints.md @@ -187,7 +187,7 @@ (and (match_operand 0 "memory_operand") (match_test "constant_address_p (XEXP (op, 0))"))) -(define_memory_constraint "Bk" +(define_special_memory_constraint "Bk" "@internal TLS address that allows insn using non-integer registers." (and (match_operand 0 "memory_operand") (not (match_test "ix86_gpr_tls_address_pattern_p (op)")))) diff --git a/gcc/testsuite/gcc.target/i386/pr116043.c b/gcc/testsuite/gcc.target/i386/pr116043.c new file mode 100644 index 00000000000..76553496c10 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr116043.c @@ -0,0 +1,33 @@ +/* { dg-do compile } */ +/* { dg-options "-mavx512bf16 -O3" } */ +/* { dg-final { scan-assembler-not {(?n)lea.*@gottpoff} } } */ + +extern __thread int a, c, i, j, k, l; +int *b; +struct d { + int e; +} f, g; +char *h; + +void m(struct d *n) { + b = &k; + for (; n->e; b++, n--) { + i = b && a; + if (i) + j = c; + } +} + +char *o(struct d *n) { + for (; n->e;) + return h; +} + +int q() { + if (l) + return 1; + int p = *o(&g); + m(&f); + m(&g); + l = p; +} commit 1d64d01e4b80629eb3c1538166a26a918c1e07cf Author: GCC Administrator Date: Tue Aug 13 00:20:00 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index c6bde7bd3e7..4397d29da19 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2024-08-12 liuhongt + + Backported from master: + 2024-07-30 liuhongt + + PR target/116043 + * config/i386/constraints.md (Bk): Refine to + define_special_memory_constraint. + 2024-08-06 John David Anglin PR target/113384 diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 2454f4b3e28..e2b84f6ebd8 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240812 +20240813 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index fde67e61e70..c1a987586bd 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,10 @@ +2024-08-12 liuhongt + + Backported from master: + 2024-07-30 liuhongt + + * gcc.target/i386/pr116043.c: New test. + 2024-08-06 Andrew Pinski Backported from master: commit 399acac731fe8d7de94e51c91601a91543fb986e Author: GCC Administrator Date: Wed Aug 14 00:19:54 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e2b84f6ebd8..94834f4f423 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240813 +20240814 commit 0906e9bc4cd2ed3938eb615a40a32cd14c5497d8 Author: GCC Administrator Date: Thu Aug 15 00:22:33 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 94834f4f423..2ea444e25fe 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240814 +20240815 commit 62b4f084037bdc1d5758c1ff48f12304539c148a Author: GCC Administrator Date: Fri Aug 16 00:19:08 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 2ea444e25fe..c11f769cc2c 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240815 +20240816 commit 2d1b1f404f3361a0e3d9d2a2bee5cf68c1290fe5 Author: Richard Sandiford Date: Fri Aug 16 15:37:50 2024 +0100 aarch64: Fix expansion of svsudot [PR114607] Not sure how this happend, but: svsudot is supposed to be expanded as USDOT with the operands swapped. However, a thinko in the expansion of svsudot meant that the arguments weren't in fact swapped; the attempted swap was just a no-op. And the testcases blithely accepted that. gcc/ PR target/114607 * config/aarch64/aarch64-sve-builtins-base.cc (svusdot_impl::expand): Fix botched attempt to swap the operands for svsudot. gcc/testsuite/ PR target/114607 * gcc.target/aarch64/sve/acle/asm/sudot_s32.c: New test. (cherry picked from commit 2c1c2485a4b1aca746ac693041e51ea6da5c64ca) diff --git a/gcc/config/aarch64/aarch64-sve-builtins-base.cc b/gcc/config/aarch64/aarch64-sve-builtins-base.cc index c24c0548724..e5e0d6ed5c9 100644 --- a/gcc/config/aarch64/aarch64-sve-builtins-base.cc +++ b/gcc/config/aarch64/aarch64-sve-builtins-base.cc @@ -2359,7 +2359,7 @@ public: version) is through the USDOT instruction but with the second and third inputs swapped. */ if (m_su) - e.rotate_inputs_left (1, 2); + e.rotate_inputs_left (1, 3); /* The ACLE function has the same order requirements as for svdot. While there's no requirement for the RTL pattern to have the same sort of order as that for dot_prod, it's easier to read. diff --git a/gcc/testsuite/gcc.target/aarch64/sve/acle/asm/sudot_s32.c b/gcc/testsuite/gcc.target/aarch64/sve/acle/asm/sudot_s32.c index 4b452619eee..e06b69affab 100644 --- a/gcc/testsuite/gcc.target/aarch64/sve/acle/asm/sudot_s32.c +++ b/gcc/testsuite/gcc.target/aarch64/sve/acle/asm/sudot_s32.c @@ -6,7 +6,7 @@ /* ** sudot_s32_tied1: -** usdot z0\.s, z2\.b, z4\.b +** usdot z0\.s, z4\.b, z2\.b ** ret */ TEST_TRIPLE_Z (sudot_s32_tied1, svint32_t, svint8_t, svuint8_t, @@ -17,7 +17,7 @@ TEST_TRIPLE_Z (sudot_s32_tied1, svint32_t, svint8_t, svuint8_t, ** sudot_s32_tied2: ** mov (z[0-9]+)\.d, z0\.d ** movprfx z0, z4 -** usdot z0\.s, z2\.b, \1\.b +** usdot z0\.s, \1\.b, z2\.b ** ret */ TEST_TRIPLE_Z_REV (sudot_s32_tied2, svint32_t, svint8_t, svuint8_t, @@ -27,7 +27,7 @@ TEST_TRIPLE_Z_REV (sudot_s32_tied2, svint32_t, svint8_t, svuint8_t, /* ** sudot_w0_s32_tied: ** mov (z[0-9]+\.b), w0 -** usdot z0\.s, z2\.b, \1 +** usdot z0\.s, \1, z2\.b ** ret */ TEST_TRIPLE_ZX (sudot_w0_s32_tied, svint32_t, svint8_t, uint8_t, @@ -37,7 +37,7 @@ TEST_TRIPLE_ZX (sudot_w0_s32_tied, svint32_t, svint8_t, uint8_t, /* ** sudot_9_s32_tied: ** mov (z[0-9]+\.b), #9 -** usdot z0\.s, z2\.b, \1 +** usdot z0\.s, \1, z2\.b ** ret */ TEST_TRIPLE_Z (sudot_9_s32_tied, svint32_t, svint8_t, uint8_t, commit 33b11c6d9a600fac25b7cc714e9905aac049685b Author: Richard Sandiford Date: Fri Aug 16 15:37:50 2024 +0100 aarch64: Fix bogus cnot optimisation [PR114603] aarch64-sve.md had a pattern that combined: cmpeq pb.T, pa/z, zc.T, #0 mov zd.T, pb/z, #1 into: cnot zd.T, pa/m, zc.T But this is only valid if pa.T is a ptrue. In other cases, the original would set inactive elements of zd.T to 0, whereas the combined form would copy elements from zc.T. gcc/ PR target/114603 * config/aarch64/aarch64-sve.md (@aarch64_pred_cnot): Replace with... (@aarch64_ptrue_cnot): ...this, requiring operand 1 to be a ptrue. (*cnot): Require operand 1 to be a ptrue. * config/aarch64/aarch64-sve-builtins-base.cc (svcnot_impl::expand): Use aarch64_ptrue_cnot for _x operations that are predicated with a ptrue. Represent other _x operations as fully-defined _m operations. gcc/testsuite/ PR target/114603 * gcc.target/aarch64/sve/acle/general/cnot_1.c: New test. (cherry picked from commit 67cbb1c638d6ab3a9cb77e674541e2b291fb67df) diff --git a/gcc/config/aarch64/aarch64-sve-builtins-base.cc b/gcc/config/aarch64/aarch64-sve-builtins-base.cc index e5e0d6ed5c9..f96cb3ccc7b 100644 --- a/gcc/config/aarch64/aarch64-sve-builtins-base.cc +++ b/gcc/config/aarch64/aarch64-sve-builtins-base.cc @@ -494,15 +494,22 @@ public: expand (function_expander &e) const OVERRIDE { machine_mode mode = e.vector_mode (0); - if (e.pred == PRED_x) - { - /* The pattern for CNOT includes an UNSPEC_PRED_Z, so needs - a ptrue hint. */ - e.add_ptrue_hint (0, e.gp_mode (0)); - return e.use_pred_x_insn (code_for_aarch64_pred_cnot (mode)); - } - - return e.use_cond_insn (code_for_cond_cnot (mode), 0); + machine_mode pred_mode = e.gp_mode (0); + /* The underlying _x pattern is effectively: + + dst = src == 0 ? 1 : 0 + + rather than an UNSPEC_PRED_X. Using this form allows autovec + constructs to be matched by combine, but it means that the + predicate on the src == 0 comparison must be all-true. + + For simplicity, represent other _x operations as fully-defined _m + operations rather than using a separate bespoke pattern. */ + if (e.pred == PRED_x + && gen_lowpart (pred_mode, e.args[0]) == CONSTM1_RTX (pred_mode)) + return e.use_pred_x_insn (code_for_aarch64_ptrue_cnot (mode)); + return e.use_cond_insn (code_for_cond_cnot (mode), + e.pred == PRED_x ? 1 : 0); } }; diff --git a/gcc/config/aarch64/aarch64-sve.md b/gcc/config/aarch64/aarch64-sve.md index b8cc47ef5fc..c68a3598423 100644 --- a/gcc/config/aarch64/aarch64-sve.md +++ b/gcc/config/aarch64/aarch64-sve.md @@ -3205,24 +3205,24 @@ ;; - CNOT ;; ------------------------------------------------------------------------- -;; Predicated logical inverse. -(define_expand "@aarch64_pred_cnot" +;; Logical inverse, predicated with a ptrue. +(define_expand "@aarch64_ptrue_cnot" [(set (match_operand:SVE_FULL_I 0 "register_operand") (unspec:SVE_FULL_I [(unspec: [(match_operand: 1 "register_operand") - (match_operand:SI 2 "aarch64_sve_ptrue_flag") + (const_int SVE_KNOWN_PTRUE) (eq: - (match_operand:SVE_FULL_I 3 "register_operand") - (match_dup 4))] + (match_operand:SVE_FULL_I 2 "register_operand") + (match_dup 3))] UNSPEC_PRED_Z) - (match_dup 5) - (match_dup 4)] + (match_dup 4) + (match_dup 3)] UNSPEC_SEL))] "TARGET_SVE" { - operands[4] = CONST0_RTX (mode); - operands[5] = CONST1_RTX (mode); + operands[3] = CONST0_RTX (mode); + operands[4] = CONST1_RTX (mode); } ) @@ -3231,7 +3231,7 @@ (unspec:SVE_I [(unspec: [(match_operand: 1 "register_operand" "Upl, Upl") - (match_operand:SI 5 "aarch64_sve_ptrue_flag") + (const_int SVE_KNOWN_PTRUE) (eq: (match_operand:SVE_I 2 "register_operand" "0, w") (match_operand:SVE_I 3 "aarch64_simd_imm_zero"))] diff --git a/gcc/testsuite/gcc.target/aarch64/sve/acle/general/cnot_1.c b/gcc/testsuite/gcc.target/aarch64/sve/acle/general/cnot_1.c new file mode 100644 index 00000000000..b1a489f0cf0 --- /dev/null +++ b/gcc/testsuite/gcc.target/aarch64/sve/acle/general/cnot_1.c @@ -0,0 +1,23 @@ +/* { dg-options "-O2" } */ +/* { dg-final { check-function-bodies "**" "" } } */ + +#include + +#ifdef __cplusplus +extern "C" { +#endif + +/* +** foo: +** cmpeq (p[0-7])\.s, p0/z, z0\.s, #0 +** mov z0\.s, \1/z, #1 +** ret +*/ +svint32_t foo(svbool_t pg, svint32_t y) +{ + return svsel(svcmpeq(pg, y, 0), svdup_s32(1), svdup_s32(0)); +} + +#ifdef __cplusplus +} +#endif commit fed75775150c53937d0728bb5b91628325715864 Author: GCC Administrator Date: Sat Aug 17 00:18:53 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 4397d29da19..7bfc4c19558 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,29 @@ +2024-08-16 Richard Sandiford + + Backported from master: + 2024-04-05 Richard Sandiford + + PR target/114603 + * config/aarch64/aarch64-sve.md (@aarch64_pred_cnot): Replace + with... + (@aarch64_ptrue_cnot): ...this, requiring operand 1 to be + a ptrue. + (*cnot): Require operand 1 to be a ptrue. + * config/aarch64/aarch64-sve-builtins-base.cc (svcnot_impl::expand): + Use aarch64_ptrue_cnot for _x operations that are predicated + with a ptrue. Represent other _x operations as fully-defined _m + operations. + +2024-08-16 Richard Sandiford + + Backported from master: + 2024-04-08 Richard Sandiford + + PR target/114607 + * config/aarch64/aarch64-sve-builtins-base.cc + (svusdot_impl::expand): Fix botched attempt to swap the operands + for svsudot. + 2024-08-12 liuhongt Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c11f769cc2c..720f32939ba 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240816 +20240817 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index c1a987586bd..227e0580cbf 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,19 @@ +2024-08-16 Richard Sandiford + + Backported from master: + 2024-04-05 Richard Sandiford + + PR target/114603 + * gcc.target/aarch64/sve/acle/general/cnot_1.c: New test. + +2024-08-16 Richard Sandiford + + Backported from master: + 2024-04-08 Richard Sandiford + + PR target/114607 + * gcc.target/aarch64/sve/acle/asm/sudot_s32.c: New test. + 2024-08-12 liuhongt Backported from master: commit 4bba0d55d569b6bd00a78ce476f7b2913447c51c Author: GCC Administrator Date: Sun Aug 18 00:18:49 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 720f32939ba..5e34934c797 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240817 +20240818 commit 7b7608d4a9f58049e5df6baed9ced7ceb15af2ea Author: GCC Administrator Date: Mon Aug 19 00:19:05 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 5e34934c797..6197155f323 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240818 +20240819 commit e861f0476aeb82acee61b9ef275be68ff8e33084 Author: GCC Administrator Date: Tue Aug 20 00:19:54 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 6197155f323..0ed256ba1fe 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240819 +20240820 commit 30a5a4ed8c8b0df99d4301daa268e75fe4b8074c Author: GCC Administrator Date: Wed Aug 21 00:20:08 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 0ed256ba1fe..2927c941b57 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240820 +20240821 commit 881b54f56e77dc470d27e4746b90dc7819c2be81 Author: Alexandre Oliva Date: Wed Jun 26 02:08:18 2024 -0300 [testsuite] [arm] [vect] adjust mve-vshr test [PR113281] The test was too optimistic, alas. We used to vectorize shifts by clamping the shift counts below the bit width of the types (e.g. at 15 for 16-bit vector elements), but (uint16_t)32768 >> (uint16_t)16 is well defined (because of promotion to 32-bit int) and must yield 0, not 1 (as before the fix). Unfortunately, in the gimple model of vector units, such large shift counts wouldn't be well-defined, so we won't vectorize such shifts any more, unless we can tell they're in range or undefined. So the test that expected the vectorization we no longer performed needs to be adjusted. Instead of nobbling the test, Richard Earnshaw suggested annotating the test with the expected ranges so as to enable the optimization, and Christophe Lyon suggested a further simplification. Co-Authored-By: Richard Earnshaw for gcc/testsuite/ChangeLog PR tree-optimization/113281 * gcc.target/arm/simd/mve-vshr.c: Add expected ranges. (cherry picked from commit 54d2339c9f87f702e02e571a5460e11c19e1c02f) diff --git a/gcc/testsuite/gcc.target/arm/simd/mve-vshr.c b/gcc/testsuite/gcc.target/arm/simd/mve-vshr.c index 8c7adef9ed8..03078de49c6 100644 --- a/gcc/testsuite/gcc.target/arm/simd/mve-vshr.c +++ b/gcc/testsuite/gcc.target/arm/simd/mve-vshr.c @@ -9,6 +9,8 @@ void test_ ## NAME ##_ ## SIGN ## BITS ## x ## NB (TYPE##BITS##_t * __restrict__ dest, TYPE##BITS##_t *a, TYPE##BITS##_t *b) { \ int i; \ for (i=0; i= (unsigned)(BITS)) \ + __builtin_unreachable(); \ dest[i] = a[i] OP b[i]; \ } \ } commit 236a83285970907170838438ee5229449a297c69 Author: GCC Administrator Date: Thu Aug 22 00:19:44 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 2927c941b57..7e53ea86edc 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240821 +20240822 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 227e0580cbf..f2a4ecee2f3 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,12 @@ +2024-08-21 Alexandre Oliva + + Backported from master: + 2024-06-26 Alexandre Oliva + Richard Earnshaw + + PR tree-optimization/113281 + * gcc.target/arm/simd/mve-vshr.c: Add expected ranges. + 2024-08-16 Richard Sandiford Backported from master: commit b4bc34db3f2948e37ad55a09870635e88c54c7d3 Author: liuhongt Date: Thu Aug 15 12:54:07 2024 +0800 Align ix86_{move_max,store_max} with vectorizer. When none of mprefer-vector-width, avx256_optimal/avx128_optimal, avx256_store_by_pieces/avx512_store_by_pieces is specified, GCC will set ix86_{move_max,store_max} as max available vector length except for AVX part. if (TARGET_AVX512F_P (opts->x_ix86_isa_flags) && TARGET_EVEX512_P (opts->x_ix86_isa_flags2)) opts->x_ix86_move_max = PVW_AVX512; else opts->x_ix86_move_max = PVW_AVX128; So for -mavx2, vectorizer will choose 256-bit for vectorization, but 128-bit is used for struct copy, there could be a potential STLF issue due to this "misalign". The patch fixes that. gcc/ChangeLog: * config/i386/i386-options.cc (ix86_option_override_internal): set ix86_{move_max,store_max} to PVW_AVX256 when TARGET_AVX instead of PVW_AVX128. gcc/testsuite/ChangeLog: * gcc.target/i386/pieces-memcpy-10.c: Add -mprefer-vector-width=128. * gcc.target/i386/pieces-memcpy-6.c: Ditto. * gcc.target/i386/pieces-memset-38.c: Ditto. * gcc.target/i386/pieces-memset-40.c: Ditto. * gcc.target/i386/pieces-memset-41.c: Ditto. * gcc.target/i386/pieces-memset-42.c: Ditto. * gcc.target/i386/pieces-memset-43.c: Ditto. * gcc.target/i386/pieces-strcpy-2.c: Ditto. * gcc.target/i386/pieces-memcpy-22.c: New test. * gcc.target/i386/pieces-memset-51.c: New test. * gcc.target/i386/pieces-strcpy-3.c: New test. (cherry picked from commit aea374238cec1a1e53fb79575d2f998e16926999) diff --git a/gcc/config/i386/i386-options.cc b/gcc/config/i386/i386-options.cc index 318f6c61455..ad496ea5a8e 100644 --- a/gcc/config/i386/i386-options.cc +++ b/gcc/config/i386/i386-options.cc @@ -2766,6 +2766,9 @@ ix86_option_override_internal (bool main_args_p, { if (TARGET_AVX512F_P (opts->x_ix86_isa_flags)) opts->x_ix86_move_max = PVW_AVX512; + /* Align with vectorizer to avoid potential STLF issue. */ + else if (TARGET_AVX_P (opts->x_ix86_isa_flags)) + opts->x_ix86_move_max = PVW_AVX256; else opts->x_ix86_move_max = PVW_AVX128; } @@ -2787,6 +2790,9 @@ ix86_option_override_internal (bool main_args_p, { if (TARGET_AVX512F_P (opts->x_ix86_isa_flags)) opts->x_ix86_store_max = PVW_AVX512; + /* Align with vectorizer to avoid potential STLF issue. */ + else if (TARGET_AVX_P (opts->x_ix86_isa_flags)) + opts->x_ix86_store_max = PVW_AVX256; else opts->x_ix86_store_max = PVW_AVX128; } diff --git a/gcc/testsuite/gcc.target/i386/pieces-memcpy-10.c b/gcc/testsuite/gcc.target/i386/pieces-memcpy-10.c index 5faee21f9b9..53ad0b3be44 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-memcpy-10.c +++ b/gcc/testsuite/gcc.target/i386/pieces-memcpy-10.c @@ -1,5 +1,5 @@ /* { dg-do compile } */ -/* { dg-options "-O2 -mno-avx2 -mavx -mtune=sandybridge" } */ +/* { dg-options "-O2 -mno-avx2 -mavx -mprefer-vector-width=128 -mtune=sandybridge" } */ extern char *dst, *src; diff --git a/gcc/testsuite/gcc.target/i386/pieces-memcpy-22.c b/gcc/testsuite/gcc.target/i386/pieces-memcpy-22.c new file mode 100644 index 00000000000..605b3623ffc --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pieces-memcpy-22.c @@ -0,0 +1,12 @@ +/* { dg-do compile { target { ! ia32 } } } */ +/* { dg-options "-O2 -mno-avx2 -mavx -mtune=generic" } */ + +extern char *dst, *src; + +void +foo (void) +{ + __builtin_memcpy (dst, src, 33); +} + +/* { dg-final { scan-assembler-times "vmovdqu\[ \\t\]+\[^\n\]*%ymm" 2 } } */ diff --git a/gcc/testsuite/gcc.target/i386/pieces-memcpy-6.c b/gcc/testsuite/gcc.target/i386/pieces-memcpy-6.c index 5f99cc98c47..cfd2a86cf33 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-memcpy-6.c +++ b/gcc/testsuite/gcc.target/i386/pieces-memcpy-6.c @@ -1,5 +1,5 @@ /* { dg-do compile { target { ! ia32 } } } */ -/* { dg-options "-O2 -mno-avx2 -mavx -mtune=sandybridge" } */ +/* { dg-options "-O2 -mno-avx2 -mavx -mprefer-vector-width=128 -mtune=sandybridge" } */ extern char *dst, *src; diff --git a/gcc/testsuite/gcc.target/i386/pieces-memset-38.c b/gcc/testsuite/gcc.target/i386/pieces-memset-38.c index ed4a24a54fd..ddd194debd5 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-memset-38.c +++ b/gcc/testsuite/gcc.target/i386/pieces-memset-38.c @@ -1,5 +1,5 @@ /* { dg-do compile } */ -/* { dg-options "-O2 -mno-avx512f -mavx2 -mtune=sandybridge" } */ +/* { dg-options "-O2 -mno-avx512f -mavx2 -mprefer-vector-width=128 -mtune=sandybridge" } */ extern char *dst; diff --git a/gcc/testsuite/gcc.target/i386/pieces-memset-40.c b/gcc/testsuite/gcc.target/i386/pieces-memset-40.c index 4eda73ead59..9c206465d46 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-memset-40.c +++ b/gcc/testsuite/gcc.target/i386/pieces-memset-40.c @@ -1,5 +1,5 @@ /* { dg-do compile } */ -/* { dg-options "-O2 -mno-avx512f -mavx2 -mtune=sandybridge" } */ +/* { dg-options "-O2 -mno-avx512f -mavx2 -mprefer-vector-width=128 -mtune=sandybridge" } */ extern char *dst; diff --git a/gcc/testsuite/gcc.target/i386/pieces-memset-41.c b/gcc/testsuite/gcc.target/i386/pieces-memset-41.c index 93df8101e4d..b0756182e35 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-memset-41.c +++ b/gcc/testsuite/gcc.target/i386/pieces-memset-41.c @@ -1,5 +1,5 @@ /* { dg-do compile } */ -/* { dg-options "-O2 -mno-avx2 -mavx -mtune=sandybridge -mno-stackrealign" } */ +/* { dg-options "-O2 -mno-avx2 -mavx -mprefer-vector-width=128 -mtune=sandybridge -mno-stackrealign" } */ extern char *dst; diff --git a/gcc/testsuite/gcc.target/i386/pieces-memset-42.c b/gcc/testsuite/gcc.target/i386/pieces-memset-42.c index df0c122aae7..103da699ae5 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-memset-42.c +++ b/gcc/testsuite/gcc.target/i386/pieces-memset-42.c @@ -1,5 +1,5 @@ /* { dg-do compile } */ -/* { dg-options "-O2 -mno-avx2 -mavx -mtune=sandybridge" } */ +/* { dg-options "-O2 -mno-avx2 -mavx -mprefer-vector-width=128 -mtune=sandybridge" } */ extern char *dst; diff --git a/gcc/testsuite/gcc.target/i386/pieces-memset-43.c b/gcc/testsuite/gcc.target/i386/pieces-memset-43.c index 2f2179c2df9..f1494e17610 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-memset-43.c +++ b/gcc/testsuite/gcc.target/i386/pieces-memset-43.c @@ -1,5 +1,5 @@ /* { dg-do compile } */ -/* { dg-options "-O2 -mno-avx2 -mavx -mtune=sandybridge" } */ +/* { dg-options "-O2 -mno-avx2 -mavx -mprefer-vector-width=128 -mtune=sandybridge" } */ extern char *dst; diff --git a/gcc/testsuite/gcc.target/i386/pieces-memset-51.c b/gcc/testsuite/gcc.target/i386/pieces-memset-51.c new file mode 100644 index 00000000000..192ec0d1647 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pieces-memset-51.c @@ -0,0 +1,12 @@ +/* { dg-do compile } */ +/* { dg-options "-O2 -mno-avx2 -mavx -mtune=generic" } */ + +extern char *dst; + +void +foo (int x) +{ + __builtin_memset (dst, x, 64); +} + +/* { dg-final { scan-assembler-times "vmovdqu\[ \\t\]+\[^\n\]*%ymm" 2 } } */ diff --git a/gcc/testsuite/gcc.target/i386/pieces-strcpy-2.c b/gcc/testsuite/gcc.target/i386/pieces-strcpy-2.c index 90446edb4f3..9bb94b7419b 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-strcpy-2.c +++ b/gcc/testsuite/gcc.target/i386/pieces-strcpy-2.c @@ -1,5 +1,5 @@ /* { dg-do compile { target { ! ia32 } } } */ -/* { dg-options "-O2 -mno-avx2 -mavx -mtune=sandybridge" } */ +/* { dg-options "-O2 -mno-avx2 -mavx -mprefer-vector-width=128 -mtune=sandybridge" } */ extern char *strcpy (char *, const char *); diff --git a/gcc/testsuite/gcc.target/i386/pieces-strcpy-3.c b/gcc/testsuite/gcc.target/i386/pieces-strcpy-3.c new file mode 100644 index 00000000000..df7571b547f --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pieces-strcpy-3.c @@ -0,0 +1,15 @@ +/* { dg-do compile { target { ! ia32 } } } */ +/* { dg-options "-O2 -mno-avx2 -mavx -mtune=generic" } */ + +extern char *strcpy (char *, const char *); + +void +foo (char *s) +{ + strcpy (s, + "1234567890abcdef123456abcdef5678123456abcdef567abcdef678" + "1234567"); +} + +/* { dg-final { scan-assembler-times "vmovdqa\[ \\t\]+\[^\n\]*%ymm" 2 } } */ +/* { dg-final { scan-assembler-times "vmovdqu\[ \\t\]+\[^\n\]*%ymm" 2 } } */ commit 141d8aa375ea32c05f0d437828e6a76f1a3ea4af Author: liuhongt Date: Thu Aug 22 14:31:40 2024 +0800 Fix testcase failure. gcc/testsuite/ChangeLog: * gcc.target/i386/pieces-memcpy-10.c: Use -mmove-max=256 and -mstore-max=256. * gcc.target/i386/pieces-memcpy-6.c: Ditto. * gcc.target/i386/pieces-memset-38.c: Ditto. * gcc.target/i386/pieces-memset-40.c: Ditto. * gcc.target/i386/pieces-memset-41.c: Ditto. * gcc.target/i386/pieces-memset-42.c: Ditto. * gcc.target/i386/pieces-memset-43.c: Ditto. * gcc.target/i386/pieces-strcpy-2.c: Ditto. (cherry picked from commit ea9c508927ec032c6d67a24df59ffa429e4d3d95) diff --git a/gcc/testsuite/gcc.target/i386/pieces-memcpy-10.c b/gcc/testsuite/gcc.target/i386/pieces-memcpy-10.c index 53ad0b3be44..78f92ac5197 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-memcpy-10.c +++ b/gcc/testsuite/gcc.target/i386/pieces-memcpy-10.c @@ -1,5 +1,5 @@ /* { dg-do compile } */ -/* { dg-options "-O2 -mno-avx2 -mavx -mprefer-vector-width=128 -mtune=sandybridge" } */ +/* { dg-options "-O2 -mno-avx2 -mavx -mmove-max=128 -mstore-max=128 -mtune=sandybridge" } */ extern char *dst, *src; diff --git a/gcc/testsuite/gcc.target/i386/pieces-memcpy-6.c b/gcc/testsuite/gcc.target/i386/pieces-memcpy-6.c index cfd2a86cf33..57b74ae4b23 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-memcpy-6.c +++ b/gcc/testsuite/gcc.target/i386/pieces-memcpy-6.c @@ -1,5 +1,5 @@ /* { dg-do compile { target { ! ia32 } } } */ -/* { dg-options "-O2 -mno-avx2 -mavx -mprefer-vector-width=128 -mtune=sandybridge" } */ +/* { dg-options "-O2 -mno-avx2 -mavx -mmove-max=128 -mstore-max=128 -mtune=sandybridge" } */ extern char *dst, *src; diff --git a/gcc/testsuite/gcc.target/i386/pieces-memset-38.c b/gcc/testsuite/gcc.target/i386/pieces-memset-38.c index ddd194debd5..d9443678735 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-memset-38.c +++ b/gcc/testsuite/gcc.target/i386/pieces-memset-38.c @@ -1,5 +1,5 @@ /* { dg-do compile } */ -/* { dg-options "-O2 -mno-avx512f -mavx2 -mprefer-vector-width=128 -mtune=sandybridge" } */ +/* { dg-options "-O2 -mno-avx512f -mavx2 -mmove-max=128 -mstore-max=128 -mtune=sandybridge" } */ extern char *dst; diff --git a/gcc/testsuite/gcc.target/i386/pieces-memset-40.c b/gcc/testsuite/gcc.target/i386/pieces-memset-40.c index 9c206465d46..8ad6ad7e494 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-memset-40.c +++ b/gcc/testsuite/gcc.target/i386/pieces-memset-40.c @@ -1,5 +1,5 @@ /* { dg-do compile } */ -/* { dg-options "-O2 -mno-avx512f -mavx2 -mprefer-vector-width=128 -mtune=sandybridge" } */ +/* { dg-options "-O2 -mno-avx512f -mavx2 -mmove-max=128 -mstore-max=128 -mtune=sandybridge" } */ extern char *dst; diff --git a/gcc/testsuite/gcc.target/i386/pieces-memset-41.c b/gcc/testsuite/gcc.target/i386/pieces-memset-41.c index b0756182e35..08fd6e9a927 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-memset-41.c +++ b/gcc/testsuite/gcc.target/i386/pieces-memset-41.c @@ -1,5 +1,5 @@ /* { dg-do compile } */ -/* { dg-options "-O2 -mno-avx2 -mavx -mprefer-vector-width=128 -mtune=sandybridge -mno-stackrealign" } */ +/* { dg-options "-O2 -mno-avx2 -mavx -mmove-max=128 -mstore-max=128 -mtune=sandybridge -mno-stackrealign" } */ extern char *dst; diff --git a/gcc/testsuite/gcc.target/i386/pieces-memset-42.c b/gcc/testsuite/gcc.target/i386/pieces-memset-42.c index 103da699ae5..6b73bb256af 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-memset-42.c +++ b/gcc/testsuite/gcc.target/i386/pieces-memset-42.c @@ -1,5 +1,5 @@ /* { dg-do compile } */ -/* { dg-options "-O2 -mno-avx2 -mavx -mprefer-vector-width=128 -mtune=sandybridge" } */ +/* { dg-options "-O2 -mno-avx2 -mavx -mmove-max=128 -mstore-max=128 -mtune=sandybridge" } */ extern char *dst; diff --git a/gcc/testsuite/gcc.target/i386/pieces-memset-43.c b/gcc/testsuite/gcc.target/i386/pieces-memset-43.c index f1494e17610..c6c7ff234da 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-memset-43.c +++ b/gcc/testsuite/gcc.target/i386/pieces-memset-43.c @@ -1,5 +1,5 @@ /* { dg-do compile } */ -/* { dg-options "-O2 -mno-avx2 -mavx -mprefer-vector-width=128 -mtune=sandybridge" } */ +/* { dg-options "-O2 -mno-avx2 -mavx -mmove-max=128 -mstore-max=128 -mtune=sandybridge" } */ extern char *dst; diff --git a/gcc/testsuite/gcc.target/i386/pieces-strcpy-2.c b/gcc/testsuite/gcc.target/i386/pieces-strcpy-2.c index 9bb94b7419b..40ada119625 100644 --- a/gcc/testsuite/gcc.target/i386/pieces-strcpy-2.c +++ b/gcc/testsuite/gcc.target/i386/pieces-strcpy-2.c @@ -1,5 +1,5 @@ /* { dg-do compile { target { ! ia32 } } } */ -/* { dg-options "-O2 -mno-avx2 -mavx -mprefer-vector-width=128 -mtune=sandybridge" } */ +/* { dg-options "-O2 -mno-avx2 -mavx -mmove-max=128 -mstore-max=128 -mtune=sandybridge" } */ extern char *strcpy (char *, const char *); commit 61d63da66ef0efd4e327e4489ebf596d0bd63e44 Author: GCC Administrator Date: Fri Aug 23 00:18:43 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 7bfc4c19558..d7150b5bf5f 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2024-08-22 liuhongt + + Backported from master: + 2024-08-22 liuhongt + + * config/i386/i386-options.cc (ix86_option_override_internal): + set ix86_{move_max,store_max} to PVW_AVX256 when TARGET_AVX + instead of PVW_AVX128. + 2024-08-16 Richard Sandiford Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7e53ea86edc..cd765ecede6 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240822 +20240823 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index f2a4ecee2f3..739b2a060a7 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,35 @@ +2024-08-22 liuhongt + + Backported from master: + 2024-08-22 liuhongt + + * gcc.target/i386/pieces-memcpy-10.c: Use -mmove-max=256 and + -mstore-max=256. + * gcc.target/i386/pieces-memcpy-6.c: Ditto. + * gcc.target/i386/pieces-memset-38.c: Ditto. + * gcc.target/i386/pieces-memset-40.c: Ditto. + * gcc.target/i386/pieces-memset-41.c: Ditto. + * gcc.target/i386/pieces-memset-42.c: Ditto. + * gcc.target/i386/pieces-memset-43.c: Ditto. + * gcc.target/i386/pieces-strcpy-2.c: Ditto. + +2024-08-22 liuhongt + + Backported from master: + 2024-08-22 liuhongt + + * gcc.target/i386/pieces-memcpy-10.c: Add -mprefer-vector-width=128. + * gcc.target/i386/pieces-memcpy-6.c: Ditto. + * gcc.target/i386/pieces-memset-38.c: Ditto. + * gcc.target/i386/pieces-memset-40.c: Ditto. + * gcc.target/i386/pieces-memset-41.c: Ditto. + * gcc.target/i386/pieces-memset-42.c: Ditto. + * gcc.target/i386/pieces-memset-43.c: Ditto. + * gcc.target/i386/pieces-strcpy-2.c: Ditto. + * gcc.target/i386/pieces-memcpy-22.c: New test. + * gcc.target/i386/pieces-memset-51.c: New test. + * gcc.target/i386/pieces-strcpy-3.c: New test. + 2024-08-21 Alexandre Oliva Backported from master: commit 15176abb931bf2ce253d27a7e3e2c697b8489d5b Author: GCC Administrator Date: Sat Aug 24 00:19:20 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index cd765ecede6..8df515083dd 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240823 +20240824 commit 19fedf7aa7810075644bfaa50c3dd4a2d5ab047c Author: GCC Administrator Date: Sun Aug 25 00:20:22 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 8df515083dd..928d43e6afd 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240824 +20240825 commit 84fc228288330eff58091dd219c175b84b0768ba Author: GCC Administrator Date: Mon Aug 26 00:20:27 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 928d43e6afd..0f44912c686 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240825 +20240826 commit c2305c8285e8f1b8baf596247b7203112076647e Author: GCC Administrator Date: Wed Aug 28 00:21:21 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 0f44912c686..00e43a06f4d 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240826 +20240828 commit 9742dbd709ff316aebc72b35b8c63c51d763d42b Author: GCC Administrator Date: Thu Aug 29 00:21:13 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 00e43a06f4d..0b8bcefbecd 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240828 +20240829 commit 2875f9fd29351c03cfe7eac014dc6e336544e9c1 Author: GCC Administrator Date: Fri Aug 30 00:24:32 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 0b8bcefbecd..8021f79746d 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240829 +20240830 commit a9284c5d4e471c19a09c4601d96bc06d6ad56846 Author: GCC Administrator Date: Sat Aug 31 00:20:04 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 8021f79746d..7036fab4770 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240830 +20240831 commit bb95e779009dd2886cba4174628d240443d6a4f0 Author: GCC Administrator Date: Sun Sep 1 00:29:35 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7036fab4770..42eedaf32c0 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240831 +20240901 commit 4dc921bcf2f97a48813fcb1773712b5ecda953cd Author: GCC Administrator Date: Mon Sep 2 00:20:58 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 42eedaf32c0..18af8d7e288 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240901 +20240902 commit 6585b06303d8fd9da907f443fc0da9faed303712 Author: liuhongt Date: Thu Aug 29 11:39:20 2024 +0800 Check avx upper register for parallel. For function arguments/return, when it's BLK mode, it's put in a parallel with an expr_list, and the expr_list contains the real mode and registers. Current ix86_check_avx_upper_register only checked for SSE_REG_P, and failed to handle that. The patch extend the handle to each subrtx. gcc/ChangeLog: PR target/116512 * config/i386/i386.cc (ix86_check_avx_upper_register): Iterate subrtx to scan for avx upper register. (ix86_check_avx_upper_stores): Inline old ix86_check_avx_upper_register. (ix86_avx_u128_mode_needed): Ditto, and replace FOR_EACH_SUBRTX with call to new ix86_check_avx_upper_register. gcc/testsuite/ChangeLog: * gcc.target/i386/pr116512.c: New test. (cherry picked from commit ab214ef734bfc3dcffcf79ff9e1dd651c2b40566) diff --git a/gcc/config/i386/i386.cc b/gcc/config/i386/i386.cc index af42e4b9739..2d272bdaf1a 100644 --- a/gcc/config/i386/i386.cc +++ b/gcc/config/i386/i386.cc @@ -14360,9 +14360,19 @@ ix86_dirflag_mode_needed (rtx_insn *insn) static bool ix86_check_avx_upper_register (const_rtx exp) { - return (SSE_REG_P (exp) - && !EXT_REX_SSE_REG_P (exp) - && GET_MODE_BITSIZE (GET_MODE (exp)) > 128); + /* construct_container may return a parallel with expr_list + which contains the real reg and mode */ + subrtx_iterator::array_type array; + FOR_EACH_SUBRTX (iter, array, exp, NONCONST) + { + const_rtx x = *iter; + if (SSE_REG_P (x) + && !EXT_REX_SSE_REG_P (x) + && GET_MODE_BITSIZE (GET_MODE (x)) > 128) + return true; + } + + return false; } /* Check if a 256bit or 512bit AVX register is referenced in stores. */ @@ -14370,7 +14380,9 @@ ix86_check_avx_upper_register (const_rtx exp) static void ix86_check_avx_upper_stores (rtx dest, const_rtx, void *data) { - if (ix86_check_avx_upper_register (dest)) + if (SSE_REG_P (dest) + && !EXT_REX_SSE_REG_P (dest) + && GET_MODE_BITSIZE (GET_MODE (dest)) > 128) { bool *used = (bool *) data; *used = true; @@ -14428,14 +14440,14 @@ ix86_avx_u128_mode_needed (rtx_insn *insn) return AVX_U128_CLEAN; } - subrtx_iterator::array_type array; - rtx set = single_set (insn); if (set) { rtx dest = SET_DEST (set); rtx src = SET_SRC (set); - if (ix86_check_avx_upper_register (dest)) + if (SSE_REG_P (dest) + && !EXT_REX_SSE_REG_P (dest) + && GET_MODE_BITSIZE (GET_MODE (dest)) > 128) { /* This is an YMM/ZMM load. Return AVX_U128_DIRTY if the source isn't zero. */ @@ -14446,9 +14458,8 @@ ix86_avx_u128_mode_needed (rtx_insn *insn) } else { - FOR_EACH_SUBRTX (iter, array, src, NONCONST) - if (ix86_check_avx_upper_register (*iter)) - return AVX_U128_DIRTY; + if (ix86_check_avx_upper_register (src)) + return AVX_U128_DIRTY; } /* This isn't YMM/ZMM load/store. */ @@ -14459,9 +14470,8 @@ ix86_avx_u128_mode_needed (rtx_insn *insn) Hardware changes state only when a 256bit register is written to, but we need to prevent the compiler from moving optimal insertion point above eventual read from 256bit or 512 bit register. */ - FOR_EACH_SUBRTX (iter, array, PATTERN (insn), NONCONST) - if (ix86_check_avx_upper_register (*iter)) - return AVX_U128_DIRTY; + if (ix86_check_avx_upper_register (PATTERN (insn))) + return AVX_U128_DIRTY; return AVX_U128_ANY; } diff --git a/gcc/testsuite/gcc.target/i386/pr116512.c b/gcc/testsuite/gcc.target/i386/pr116512.c new file mode 100644 index 00000000000..c2bc6c91b64 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr116512.c @@ -0,0 +1,26 @@ +/* { dg-do compile } */ +/* { dg-options "-march=x86-64-v4 -O2" } */ +/* { dg-final { scan-assembler-not "vzeroupper" { target { ! ia32 } } } } */ + +#include + +struct B { + union { + __m512 f; + __m512i s; + }; +}; + +struct B foo(int n) { + struct B res; + res.s = _mm512_set1_epi32(n); + + return res; +} + +__m512i bar(int n) { + struct B res; + res.s = _mm512_set1_epi32(n); + + return res.s; +} commit 911eadd4903f290f7c8bf6a490aa62742e699869 Author: GCC Administrator Date: Tue Sep 3 00:23:54 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index d7150b5bf5f..2d281e6a4b5 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,17 @@ +2024-09-02 liuhongt + + Backported from master: + 2024-08-30 liuhongt + + PR target/116512 + * config/i386/i386.cc (ix86_check_avx_upper_register): Iterate + subrtx to scan for avx upper register. + (ix86_check_avx_upper_stores): Inline old + ix86_check_avx_upper_register. + (ix86_avx_u128_mode_needed): Ditto, and replace + FOR_EACH_SUBRTX with call to new + ix86_check_avx_upper_register. + 2024-08-22 liuhongt Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 18af8d7e288..a1e4e7fb875 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240902 +20240903 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 739b2a060a7..1a9f58bd3d0 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,10 @@ +2024-09-02 liuhongt + + Backported from master: + 2024-08-30 liuhongt + + * gcc.target/i386/pr116512.c: New test. + 2024-08-22 liuhongt Backported from master: commit 6e59b188c4a051d4f2de5220d30681e6963d96c0 Author: Haochen Jiang Date: Mon Sep 2 15:00:22 2024 +0800 i386: Fix vfpclassph non-optimizied intrin The intrin for non-optimized got a typo in mask type, which will cause the high bits of __mmask32 being unexpectedly zeroed. The test does not fail under O0 with current 1b since the testcase is wrong. We need to include avx512-mask-type.h after SIZE is defined, or it will always be __mmask8. That problem also happened in AVX10.2 testcases. I will write a seperate patch to fix that. gcc/ChangeLog: * config/i386/avx512fp16intrin.h (_mm512_mask_fpclass_ph_mask): Correct mask type to __mmask32. (_mm512_fpclass_ph_mask): Ditto. gcc/testsuite/ChangeLog: * gcc.target/i386/avx512fp16-vfpclassph-1c.c: New test. diff --git a/gcc/config/i386/avx512fp16intrin.h b/gcc/config/i386/avx512fp16intrin.h index b16ccfcb7f1..6330e57ebb8 100644 --- a/gcc/config/i386/avx512fp16intrin.h +++ b/gcc/config/i386/avx512fp16intrin.h @@ -2321,11 +2321,11 @@ _mm512_fpclass_ph_mask (__m512h __A, const int __imm) #else #define _mm512_mask_fpclass_ph_mask(u, x, c) \ ((__mmask32) __builtin_ia32_fpclassph512_mask ((__v32hf) (__m512h) (x), \ - (int) (c),(__mmask8)(u))) + (int) (c),(__mmask32)(u))) #define _mm512_fpclass_ph_mask(x, c) \ ((__mmask32) __builtin_ia32_fpclassph512_mask ((__v32hf) (__m512h) (x), \ - (int) (c),(__mmask8)-1)) + (int) (c),(__mmask32)-1)) #endif /* __OPIMTIZE__ */ /* Intrinsics vgetexpph, vgetexpsh. */ diff --git a/gcc/testsuite/gcc.target/i386/avx512fp16-vfpclassph-1c.c b/gcc/testsuite/gcc.target/i386/avx512fp16-vfpclassph-1c.c new file mode 100644 index 00000000000..4739f1228e3 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/avx512fp16-vfpclassph-1c.c @@ -0,0 +1,77 @@ +/* { dg-do run } */ +/* { dg-options "-O0 -mavx512fp16" } */ +/* { dg-require-effective-target avx512fp16 } */ + +#define AVX512FP16 +#include "avx512f-helper.h" + +#include +#include +#include +#define SIZE (AVX512F_LEN / 16) +#include "avx512f-mask-type.h" + +#ifndef __FPCLASSPH__ +#define __FPCLASSPH__ +int check_fp_class_hp (_Float16 src, int imm) +{ + int qNaN_res = isnan (src); + int sNaN_res = isnan (src); + int Pzero_res = (src == 0.0); + int Nzero_res = (src == -0.0); + int PInf_res = (isinf (src) == 1); + int NInf_res = (isinf (src) == -1); + int Denorm_res = (fpclassify (src) == FP_SUBNORMAL); + int FinNeg_res = __builtin_finite (src) && (src < 0); + + int result = (((imm & 1) && qNaN_res) + || (((imm >> 1) & 1) && Pzero_res) + || (((imm >> 2) & 1) && Nzero_res) + || (((imm >> 3) & 1) && PInf_res) + || (((imm >> 4) & 1) && NInf_res) + || (((imm >> 5) & 1) && Denorm_res) + || (((imm >> 6) & 1) && FinNeg_res) + || (((imm >> 7) & 1) && sNaN_res)); + return result; +} +#endif + +MASK_TYPE +CALC (_Float16 *s1, int imm) +{ + int i; + MASK_TYPE res = 0; + + for (i = 0; i < SIZE; i++) + if (check_fp_class_hp(s1[i], imm)) + res = res | (1 << i); + + return res; +} + +void +TEST (void) +{ + int i; + UNION_TYPE (AVX512F_LEN, h) src; + MASK_TYPE res1, res2, res_ref = 0; + MASK_TYPE mask = MASK_VALUE; + + src.a[SIZE - 1] = NAN; + src.a[SIZE - 2] = 1.0 / 0.0; + for (i = 0; i < SIZE - 2; i++) + { + src.a[i] = -24.43 + 0.6 * i; + } + + res1 = INTRINSIC (_fpclass_ph_mask) (src.x, 0xFF); + res2 = INTRINSIC (_mask_fpclass_ph_mask) (mask, src.x, 0xFF); + + res_ref = CALC (src.a, 0xFF); + + if (res_ref != res1) + abort (); + + if ((mask & res_ref) != res2) + abort (); +} commit 93e66cab194d6385b6aa383242fcbdd490dae6d1 Author: GCC Administrator Date: Wed Sep 4 00:25:58 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 2d281e6a4b5..37ed9375ce5 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,9 @@ +2024-09-03 Haochen Jiang + + * config/i386/avx512fp16intrin.h + (_mm512_mask_fpclass_ph_mask): Correct mask type to __mmask32. + (_mm512_fpclass_ph_mask): Ditto. + 2024-09-02 liuhongt Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index a1e4e7fb875..301a37d563d 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240903 +20240904 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 1a9f58bd3d0..0fb37aaf192 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,7 @@ +2024-09-03 Haochen Jiang + + * gcc.target/i386/avx512fp16-vfpclassph-1c.c: New test. + 2024-09-02 liuhongt Backported from master: commit 87a5641b65da50e0f5a12e0bc23be183d18c041c Author: GCC Administrator Date: Thu Sep 5 00:20:00 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 301a37d563d..c8bedcac256 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240904 +20240905 commit 42d4aa02c6016cc8efd896f627f692896e68c914 Author: H.J. Lu Date: Tue Aug 27 13:11:39 2024 -0700 ipa: Don't disable function parameter analysis for fat LTO Update analyze_parms not to disable function parameter analysis for -ffat-lto-objects. Tested on x86-64, there are no differences in zstd with "-O2 -flto=auto" -g "vs -O2 -flto=auto -g -ffat-lto-objects". PR ipa/116410 * ipa-modref.cc (analyze_parms): Always analyze function parameter for LTO. Signed-off-by: H.J. Lu (cherry picked from commit 2f1689ea8e631ebb4ff3720d56ef0362f5898ff6) diff --git a/gcc/ipa-modref.cc b/gcc/ipa-modref.cc index 556816ab429..d41de9c59c1 100644 --- a/gcc/ipa-modref.cc +++ b/gcc/ipa-modref.cc @@ -2964,7 +2964,7 @@ analyze_parms (modref_summary *summary, modref_summary_lto *summary_lto, summary->arg_flags.safe_grow_cleared (count, true); summary->arg_flags[parm_index] = EAF_UNUSED; } - else if (summary_lto) + if (summary_lto) { if (parm_index >= summary_lto->arg_flags.length ()) summary_lto->arg_flags.safe_grow_cleared (count, true); @@ -3020,7 +3020,7 @@ analyze_parms (modref_summary *summary, modref_summary_lto *summary_lto, summary->arg_flags.safe_grow_cleared (count, true); summary->arg_flags[parm_index] = flags; } - else if (summary_lto) + if (summary_lto) { if (parm_index >= summary_lto->arg_flags.length ()) summary_lto->arg_flags.safe_grow_cleared (count, true); commit 71f9ca6c6933557104772d069db4f6ae253b02e5 Author: GCC Administrator Date: Fri Sep 6 00:19:57 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 37ed9375ce5..c7ac9ee1dae 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2024-09-05 H.J. Lu + + Backported from master: + 2024-09-03 H.J. Lu + + PR ipa/116410 + * ipa-modref.cc (analyze_parms): Always analyze function parameter + for LTO. + 2024-09-03 Haochen Jiang * config/i386/avx512fp16intrin.h diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c8bedcac256..a2aca132dd9 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240905 +20240906 commit fc14ff0c9e52c150a84c71796808569281a37140 Author: GCC Administrator Date: Sat Sep 7 00:18:42 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index a2aca132dd9..346d3ce6907 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240906 +20240907 commit 0f053a851954dea3be138e70b426d92bdbe551f0 Author: GCC Administrator Date: Sun Sep 8 00:20:33 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 346d3ce6907..e8f2bd3ba1b 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240907 +20240908 commit 0dba9570a403534bc56331b8f63a45e3a01639e0 Author: GCC Administrator Date: Mon Sep 9 00:18:31 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e8f2bd3ba1b..94bd4791818 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240908 +20240909 commit b48e7c28b62688537680dc177c1bc28d66157611 Author: GCC Administrator Date: Tue Sep 10 00:25:59 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 94bd4791818..83b31cda118 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240909 +20240910 commit b64a99840ee8ba5311095ce68f7b89ccd320d597 Author: GCC Administrator Date: Wed Sep 11 00:19:52 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 83b31cda118..2abf40291e6 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240910 +20240911 commit 682cc3f90d0ba42123ef6bb838f28039524e3ea8 Author: GCC Administrator Date: Thu Sep 12 00:18:11 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 2abf40291e6..8dcf10768ee 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240911 +20240912 commit 0344276a0015415e2076c79e2f56c980efae004f Author: GCC Administrator Date: Fri Sep 13 00:19:01 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 8dcf10768ee..bc240099074 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240912 +20240913 commit 6aceb85821ad96b13b7d9dd5c51091f42bb869d7 Author: GCC Administrator Date: Sat Sep 14 00:18:08 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index bc240099074..84c2c89a522 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240913 +20240914 commit 5c8f84c5ddcfc855dbbf4a6a30b216a6e8b1c0b6 Author: GCC Administrator Date: Sun Sep 15 00:18:06 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 84c2c89a522..5cbbe5defc0 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240914 +20240915 commit ebdc85b6ce4f2067431331a1a37440e952efd49b Author: H.J. Lu Date: Fri Sep 6 05:24:07 2024 -0700 x86-64: Don't use temp for argument in a TImode register Don't use temp for a PARALLEL BLKmode argument of an EXPR_LIST expression in a TImode register. Otherwise, the TImode variable will be put in the GPR save area which guarantees only 8-byte alignment. gcc/ PR target/116621 * config/i386/i386.cc (ix86_gimplify_va_arg): Don't use temp for a PARALLEL BLKmode container of an EXPR_LIST expression in a TImode register. gcc/testsuite/ PR target/116621 * gcc.target/i386/pr116621.c: New test. Signed-off-by: H.J. Lu (cherry picked from commit fa7bbb065c63aa802e0bbb04d605407dad58cf94) diff --git a/gcc/config/i386/i386.cc b/gcc/config/i386/i386.cc index 2d272bdaf1a..b52eb0d5f7b 100644 --- a/gcc/config/i386/i386.cc +++ b/gcc/config/i386/i386.cc @@ -4780,13 +4780,31 @@ ix86_gimplify_va_arg (tree valist, tree type, gimple_seq *pre_p, examine_argument (nat_mode, type, 0, &needed_intregs, &needed_sseregs); - need_temp = (!REG_P (container) + bool container_in_reg = false; + if (REG_P (container)) + container_in_reg = true; + else if (GET_CODE (container) == PARALLEL + && GET_MODE (container) == BLKmode + && XVECLEN (container, 0) == 1) + { + /* Check if it is a PARALLEL BLKmode container of an EXPR_LIST + expression in a TImode register. In this case, temp isn't + needed. Otherwise, the TImode variable will be put in the + GPR save area which guarantees only 8-byte alignment. */ + rtx x = XVECEXP (container, 0, 0); + if (GET_CODE (x) == EXPR_LIST + && REG_P (XEXP (x, 0)) + && XEXP (x, 1) == const0_rtx) + container_in_reg = true; + } + + need_temp = (!container_in_reg && ((needed_intregs && TYPE_ALIGN (type) > 64) || TYPE_ALIGN (type) > 128)); /* In case we are passing structure, verify that it is consecutive block on the register save area. If not we need to do moves. */ - if (!need_temp && !REG_P (container)) + if (!need_temp && !container_in_reg) { /* Verify that all registers are strictly consecutive */ if (SSE_REGNO_P (REGNO (XEXP (XVECEXP (container, 0, 0), 0)))) diff --git a/gcc/testsuite/gcc.target/i386/pr116621.c b/gcc/testsuite/gcc.target/i386/pr116621.c new file mode 100644 index 00000000000..704266458a8 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr116621.c @@ -0,0 +1,43 @@ +/* { dg-do run } */ +/* { dg-options "-O2" } */ + +#include +#include + +union S8302 +{ + union + { + double b; + int c; + } a; + long double d; + unsigned short int f[5]; +}; + +union S8302 s8302; +extern void check8302va (int i, ...); + +int +main (void) +{ + memset (&s8302, '\0', sizeof (s8302)); + s8302.a.b = -221438.250000; + check8302va (1, s8302); + return 0; +} + +__attribute__((noinline, noclone)) +void +check8302va (int z, ...) +{ + union S8302 arg, *p; + va_list ap; + + __builtin_va_start (ap, z); + p = &s8302; + arg = __builtin_va_arg (ap, union S8302); + if (p->a.b != arg.a.b) + __builtin_abort (); + __builtin_va_end (ap); +} commit 772393c20b3c128ba5791460b0b5fa8f2fa3312d Author: GCC Administrator Date: Mon Sep 16 00:18:36 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index c7ac9ee1dae..c50ed2b2451 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,13 @@ +2024-09-15 H.J. Lu + + Backported from master: + 2024-09-08 H.J. Lu + + PR target/116621 + * config/i386/i386.cc (ix86_gimplify_va_arg): Don't use temp for + a PARALLEL BLKmode container of an EXPR_LIST expression in a + TImode register. + 2024-09-05 H.J. Lu Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 5cbbe5defc0..657376e5779 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240915 +20240916 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 0fb37aaf192..ef4fe51a5af 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-09-15 H.J. Lu + + Backported from master: + 2024-09-08 H.J. Lu + + PR target/116621 + * gcc.target/i386/pr116621.c: New test. + 2024-09-03 Haochen Jiang * gcc.target/i386/avx512fp16-vfpclassph-1c.c: New test. commit 46bf97c534ca7e202f741d9f6ebe72f73f076302 Author: GCC Administrator Date: Tue Sep 17 00:18:27 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 657376e5779..9fc61c92b8c 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240916 +20240917 commit 9046f9aeae0f926e7365d39809a80855e7dc184a Author: Marek Polacek Date: Mon Sep 16 16:42:38 2024 -0400 c++: crash with anon VAR_DECL [PR116676] r12-3495 added maybe_warn_about_constant_value which will crash if it gets a nameless VAR_DECL, which is what happens in this PR. We created this VAR_DECL in cp_parser_decomposition_declaration. PR c++/116676 gcc/cp/ChangeLog: * constexpr.cc (maybe_warn_about_constant_value): Check DECL_NAME. gcc/testsuite/ChangeLog: * g++.dg/cpp1z/constexpr-116676.C: New test. Reviewed-by: Jason Merrill (cherry picked from commit dfe0d4389a3ce43179563a63046ad3e74d615a08) diff --git a/gcc/cp/constexpr.cc b/gcc/cp/constexpr.cc index 41f862e7056..20abbee3600 100644 --- a/gcc/cp/constexpr.cc +++ b/gcc/cp/constexpr.cc @@ -6434,6 +6434,7 @@ maybe_warn_about_constant_value (location_t loc, tree decl) && warn_interference_size && !OPTION_SET_P (param_destruct_interfere_size) && DECL_CONTEXT (decl) == std_node + && DECL_NAME (decl) && id_equal (DECL_NAME (decl), "hardware_destructive_interference_size") && (LOCATION_FILE (input_location) != main_input_filename || module_exporting_p ()) diff --git a/gcc/testsuite/g++.dg/cpp1z/constexpr-116676.C b/gcc/testsuite/g++.dg/cpp1z/constexpr-116676.C new file mode 100644 index 00000000000..1cb65f10a1d --- /dev/null +++ b/gcc/testsuite/g++.dg/cpp1z/constexpr-116676.C @@ -0,0 +1,57 @@ +// PR c++/116676 +// { dg-do compile { target c++17 } } + +namespace std { +typedef __SIZE_TYPE__ size_t; + + template + struct remove_reference + { typedef _Tp type; }; + + template + struct remove_reference<_Tp&> + { typedef _Tp type; }; + + template + struct remove_reference<_Tp&&> + { typedef _Tp type; }; + +template +constexpr typename std::remove_reference<_Tp>::type && +move(_Tp &&__t) noexcept { + return static_cast::type &&>(__t); +} +template struct tuple_size; +template struct tuple_element; +template class __pair_base {}; +template +struct pair { + _T1 first; + _T2 second; + template + explicit constexpr pair(const _T1 &__a, const _T2 &__b) + : first(__a), second(__b) {} +}; +template +struct tuple_size> +{ +static constexpr size_t value = 2; +}; +template +struct tuple_element<0, pair<_Tp1, _Tp2>> { + typedef _Tp1 type; +}; +template +struct tuple_element<1, pair<_Tp1, _Tp2>> { + typedef _Tp2 type; +}; + +template +constexpr typename tuple_element<_Int, pair<_Tp1, _Tp2>>::type & +get(pair<_Tp1, _Tp2> &&__in) noexcept { + return (std::move(__in).first); +} +int t; +auto [a, b] = std::pair{t, 1}; +} + commit 0ab2379e3acdf7efeec8112818fd844ec191663a Author: GCC Administrator Date: Wed Sep 18 00:19:22 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 9fc61c92b8c..6d5458fa160 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240917 +20240918 diff --git a/gcc/cp/ChangeLog b/gcc/cp/ChangeLog index 0f551f9a95a..b2f55e13d7d 100644 --- a/gcc/cp/ChangeLog +++ b/gcc/cp/ChangeLog @@ -1,3 +1,11 @@ +2024-09-17 Marek Polacek + + Backported from master: + 2024-09-17 Marek Polacek + + PR c++/116676 + * constexpr.cc (maybe_warn_about_constant_value): Check DECL_NAME. + 2024-06-20 Release Manager * GCC 12.4.0 released. diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index ef4fe51a5af..2f2efe212dd 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-09-17 Marek Polacek + + Backported from master: + 2024-09-17 Marek Polacek + + PR c++/116676 + * g++.dg/cpp1z/constexpr-116676.C: New test. + 2024-09-15 H.J. Lu Backported from master: commit f467bbb06d6498f54fa535b095dbfd3f872dd5fd Author: GCC Administrator Date: Thu Sep 19 00:20:51 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 6d5458fa160..f2530dded09 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240918 +20240919 commit 8483527158024d200b3a9e4edecbe188fa22fdaa Author: Haochen Jiang Date: Wed Sep 18 11:20:15 2024 +0800 doc: Add more alias option and reorder Intel CPU -march documentation This patch is backported from GCC15 with some tweaks. Since r15-3539, there are requests coming in to add other alias option documentation. This patch will add all of them, including corei7, corei7-avx, core-avx-i, core-avx2, atom and slm. Also in the patch, I reordered that part of documentation, currently all the CPUs/products are just all over the place. I regrouped them by date-to-now products (since the very first CPU to latest Panther Lake), P-core (since the clients become hybrid cores, starting from Sapphire Rapids) and E-core (since Bonnell). In GCC14 and eariler GCC, Xeon Phi CPUs are still there, I put them after E-core CPUs. And in the patch, I refined the product names in documentation. gcc/ChangeLog: * doc/invoke.texi: Add corei7, corei7-avx, core-avx-i, core-avx2, atom, and slm. Reorder the -march documentation by splitting them into date-to-now products, P-core, E-core and Xeon Phi. Refine the product names in documentation. diff --git a/gcc/doc/invoke.texi b/gcc/doc/invoke.texi index fbfa3241e7f..5db66718d10 100644 --- a/gcc/doc/invoke.texi +++ b/gcc/doc/invoke.texi @@ -31449,6 +31449,7 @@ Intel Core 2 CPU with 64-bit extensions, MMX, SSE, SSE2, SSE3, SSSE3, CX16, SAHF and FXSR instruction set support. @item nehalem +@itemx corei7 Intel Nehalem CPU with 64-bit extensions, MMX, SSE, SSE2, SSE3, SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF and FXSR instruction set support. @@ -31457,17 +31458,20 @@ Intel Westmere CPU with 64-bit extensions, MMX, SSE, SSE2, SSE3, SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR and PCLMUL instruction set support. @item sandybridge +@itemx corei7-avx Intel Sandy Bridge CPU with 64-bit extensions, MMX, SSE, SSE2, SSE3, SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE and PCLMUL instruction set support. @item ivybridge +@itemx core-avx-i Intel Ivy Bridge CPU with 64-bit extensions, MMX, SSE, SSE2, SSE3, SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, RDRND and F16C instruction set support. @item haswell -Intel Haswell CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, SSSE3, +@itemx core-avx2 +Intel Haswell CPU with 64-bit extensions, MMX, SSE, SSE2, SSE3, SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, RDRND, F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE and HLE instruction set support. @@ -31483,47 +31487,6 @@ SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, RDRND, F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, AES, CLFLUSHOPT, XSAVEC, XSAVES and SGX instruction set support. -@item bonnell -Intel Bonnell CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3 and SSSE3 -instruction set support. - -@item silvermont -Intel Silvermont CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, SSSE3, -SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, PCLMUL, PREFETCHW and RDRND -instruction set support. - -@item goldmont -Intel Goldmont CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, SSSE3, -SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, PCLMUL, PREFETCHW, RDRND, AES, SHA, -RDSEED, XSAVE, XSAVEC, XSAVES, XSAVEOPT, CLFLUSHOPT and FSGSBASE instruction -set support. - -@item goldmont-plus -Intel Goldmont Plus CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, -SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, PCLMUL, PREFETCHW, RDRND, AES, -SHA, RDSEED, XSAVE, XSAVEC, XSAVES, XSAVEOPT, CLFLUSHOPT, FSGSBASE, PTWRITE, -RDPID and SGX instruction set support. - -@item tremont -Intel Tremont CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, SSSE3, -SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, PCLMUL, PREFETCHW, RDRND, AES, SHA, -RDSEED, XSAVE, XSAVEC, XSAVES, XSAVEOPT, CLFLUSHOPT, FSGSBASE, PTWRITE, RDPID, -SGX, CLWB, GFNI-SSE, MOVDIRI, MOVDIR64B, CLDEMOTE and WAITPKG instruction set -support. - -@item knl -Intel Knight's Landing CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, -SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, -RDRND, F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, -AVX512PF, AVX512ER, AVX512F, AVX512CD and PREFETCHWT1 instruction set support. - -@item knm -Intel Knights Mill CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, -SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, -RDRND, F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, -AVX512PF, AVX512ER, AVX512F, AVX512CD and PREFETCHWT1, AVX5124VNNIW, -AVX5124FMAPS and AVX512VPOPCNTDQ instruction set support. - @item skylake-avx512 Intel Skylake Server CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, @@ -31531,16 +31494,30 @@ RDRND, F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, AES, CLFLUSHOPT, XSAVEC, XSAVES, SGX, AVX512F, CLWB, AVX512VL, AVX512BW, AVX512DQ and AVX512CD instruction set support. +@item cascadelake +Intel Cascade Lake CPU with 64-bit extensions, MMX, SSE, SSE2, SSE3, SSSE3, +SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, RDRND, +F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, AES, +CLFLUSHOPT, XSAVEC, XSAVES, SGX, AVX512F, CLWB, AVX512VL, AVX512BW, AVX512DQ, +AVX512CD and AVX512VNNI instruction set support. + @item cannonlake -Intel Cannonlake Server CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, +Intel Cannon Lake Server CPU with 64-bit extensions, MMX, SSE, SSE2, SSE3, SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, RDRND, F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, AES, CLFLUSHOPT, XSAVEC, XSAVES, SGX, AVX512F, AVX512VL, AVX512BW, AVX512DQ, AVX512CD, PKU, AVX512VBMI, AVX512IFMA and SHA instruction set support. +@item cooperlake +Intel Cooper Lake CPU with 64-bit extensions, MMX, SSE, SSE2, SSE3, SSSE3, +SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, RDRND, +F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, AES, +CLFLUSHOPT, XSAVEC, XSAVES, SGX, AVX512F, CLWB, AVX512VL, AVX512BW, AVX512DQ, +AVX512CD, AVX512VNNI and AVX512BF16 instruction set support. + @item icelake-client -Intel Icelake Client CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, +Intel Ice Lake Client CPU with 64-bit extensions, MMX, SSE, SSE2, SSE3, SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, RDRND, F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, AES, CLFLUSHOPT, XSAVEC, XSAVES, SGX, AVX512F, AVX512VL, AVX512BW, AVX512DQ, @@ -31548,7 +31525,7 @@ AVX512CD, PKU, AVX512VBMI, AVX512IFMA, SHA, AVX512VNNI, GFNI, VAES, AVX512VBMI2 , VPCLMULQDQ, AVX512BITALG, RDPID and AVX512VPOPCNTDQ instruction set support. @item icelake-server -Intel Icelake Server CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, +Intel Ice Lake Server CPU with 64-bit extensions, MMX, SSE, SSE2, SSE3, SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, RDRND, F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, AES, CLFLUSHOPT, XSAVEC, XSAVES, SGX, AVX512F, AVX512VL, AVX512BW, AVX512DQ, @@ -31556,55 +31533,84 @@ AVX512CD, PKU, AVX512VBMI, AVX512IFMA, SHA, AVX512VNNI, GFNI, VAES, AVX512VBMI2 , VPCLMULQDQ, AVX512BITALG, RDPID, AVX512VPOPCNTDQ, PCONFIG, WBNOINVD and CLWB instruction set support. -@item cascadelake -Intel Cascadelake CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, SSSE3, -SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, RDRND, -F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, AES, -CLFLUSHOPT, XSAVEC, XSAVES, SGX, AVX512F, CLWB, AVX512VL, AVX512BW, AVX512DQ, -AVX512CD and AVX512VNNI instruction set support. - -@item cooperlake -Intel cooperlake CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, SSSE3, +@item tigerlake +Intel Tiger Lake CPU with 64-bit extensions, MMX, SSE, SSE2, SSE3, SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, RDRND, F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, AES, -CLFLUSHOPT, XSAVEC, XSAVES, SGX, AVX512F, CLWB, AVX512VL, AVX512BW, AVX512DQ, -AVX512CD, AVX512VNNI and AVX512BF16 instruction set support. +CLFLUSHOPT, XSAVEC, XSAVES, SGX, AVX512F, AVX512VL, AVX512BW, AVX512DQ, +AVX512CD, PKU, AVX512VBMI, AVX512IFMA, SHA, AVX512VNNI, GFNI, VAES, AVX512VBMI2, +VPCLMULQDQ, AVX512BITALG, RDPID, AVX512VPOPCNTDQ, MOVDIRI, MOVDIR64B, CLWB, +AVX512VP2INTERSECT and KEYLOCKER instruction set support. -@item tigerlake -Intel Tigerlake CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, SSSE3, +@item rocketlake +Intel Rocket Lake CPU with 64-bit extensions, MMX, SSE, SSE2, SSE3, SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, RDRND, F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, AES, -CLFLUSHOPT, XSAVEC, XSAVES, SGX, AVX512F, AVX512VL, AVX512BW, AVX512DQ, AVX512CD +CLFLUSHOPT, XSAVEC, XSAVES, AVX512F, AVX512VL, AVX512BW, AVX512DQ, AVX512CD, PKU, AVX512VBMI, AVX512IFMA, SHA, AVX512VNNI, GFNI, VAES, AVX512VBMI2, -VPCLMULQDQ, AVX512BITALG, RDPID, AVX512VPOPCNTDQ, MOVDIRI, MOVDIR64B, CLWB, -AVX512VP2INTERSECT and KEYLOCKER instruction set support. +VPCLMULQDQ, AVX512BITALG, RDPID and AVX512VPOPCNTDQ instruction set support. + +@item alderlake +Intel Alder Lake CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, SSSE3, +SSE4.1, SSE4.2, POPCNT, AES, PREFETCHW, PCLMUL, RDRND, XSAVE, XSAVEC, XSAVES, +XSAVEOPT, FSGSBASE, PTWRITE, RDPID, SGX, GFNI-SSE, CLWB, MOVDIRI, MOVDIR64B, +CLDEMOTE, WAITPKG, ADCX, AVX, AVX2, BMI, BMI2, F16C, FMA, LZCNT, PCONFIG, PKU, +VAES, VPCLMULQDQ, SERIALIZE, HRESET, KL, WIDEKL and AVX-VNNI instruction set +support. @item sapphirerapids -Intel sapphirerapids CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, -SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, -RDRND, F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, -AES, CLFLUSHOPT, XSAVEC, XSAVES, SGX, AVX512F, AVX512VL, AVX512BW, AVX512DQ, +Intel Sapphire Rapids CPU with 64-bit extensions, MMX, SSE, SSE2, SSE3, SSSE3, +SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, RDRND, +F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, AES, +CLFLUSHOPT, XSAVEC, XSAVES, SGX, AVX512F, AVX512VL, AVX512BW, AVX512DQ, AVX512CD, PKU, AVX512VBMI, AVX512IFMA, SHA, AVX512VNNI, GFNI, VAES, AVX512VBMI2, VPCLMULQDQ, AVX512BITALG, RDPID, AVX512VPOPCNTDQ, PCONFIG, WBNOINVD, CLWB, MOVDIRI, MOVDIR64B, ENQCMD, CLDEMOTE, PTWRITE, WAITPKG, SERIALIZE, TSXLDTRK, UINTR, AMX-BF16, AMX-TILE, AMX-INT8, AVX-VNNI, AVX512-FP16 and AVX512BF16 instruction set support. -@item alderlake -Intel Alderlake CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, SSSE3, -SSE4.1, SSE4.2, POPCNT, AES, PREFETCHW, PCLMUL, RDRND, XSAVE, XSAVEC, XSAVES, -XSAVEOPT, FSGSBASE, PTWRITE, RDPID, SGX, GFNI-SSE, CLWB, MOVDIRI, MOVDIR64B, -CLDEMOTE, WAITPKG, ADCX, AVX, AVX2, BMI, BMI2, F16C, FMA, LZCNT, PCONFIG, PKU, -VAES, VPCLMULQDQ, SERIALIZE, HRESET, KL, WIDEKL and AVX-VNNI instruction set +@item bonnell +@itemx atom +Intel Bonnell CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3 and SSSE3 +instruction set support. + +@item silvermont +@itemx slm +Intel Silvermont CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, SSSE3, +SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, PCLMUL, PREFETCHW and RDRND +instruction set support. + +@item goldmont +Intel Goldmont CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, SSSE3, +SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, PCLMUL, PREFETCHW, RDRND, AES, SHA, +RDSEED, XSAVE, XSAVEC, XSAVES, XSAVEOPT, CLFLUSHOPT and FSGSBASE instruction +set support. + +@item goldmont-plus +Intel Goldmont Plus CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, +SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, PCLMUL, PREFETCHW, RDRND, AES, +SHA, RDSEED, XSAVE, XSAVEC, XSAVES, XSAVEOPT, CLFLUSHOPT, FSGSBASE, PTWRITE, +RDPID and SGX instruction set support. + +@item tremont +Intel Tremont CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, SSSE3, +SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, PCLMUL, PREFETCHW, RDRND, AES, SHA, +RDSEED, XSAVE, XSAVEC, XSAVES, XSAVEOPT, CLFLUSHOPT, FSGSBASE, PTWRITE, RDPID, +SGX, CLWB, GFNI-SSE, MOVDIRI, MOVDIR64B, CLDEMOTE and WAITPKG instruction set support. -@item rocketlake -Intel Rocketlake CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, SSSE3 -, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, RDRND, -F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, AES, -CLFLUSHOPT, XSAVEC, XSAVES, AVX512F, AVX512VL, AVX512BW, AVX512DQ, AVX512CD -PKU, AVX512VBMI, AVX512IFMA, SHA, AVX512VNNI, GFNI, VAES, AVX512VBMI2, -VPCLMULQDQ, AVX512BITALG, RDPID and AVX512VPOPCNTDQ instruction set support. +@item knl +Intel Knights Landing CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, +SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, +RDRND, F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, +AVX512PF, AVX512ER, AVX512F, AVX512CD and PREFETCHWT1 instruction set support. + +@item knm +Intel Knights Mill CPU with 64-bit extensions, MOVBE, MMX, SSE, SSE2, SSE3, +SSSE3, SSE4.1, SSE4.2, POPCNT, CX16, SAHF, FXSR, AVX, XSAVE, PCLMUL, FSGSBASE, +RDRND, F16C, AVX2, BMI, BMI2, LZCNT, FMA, MOVBE, HLE, RDSEED, ADCX, PREFETCHW, +AVX512PF, AVX512ER, AVX512F, AVX512CD and PREFETCHWT1, AVX5124VNNIW, +AVX5124FMAPS and AVX512VPOPCNTDQ instruction set support. @item k6 AMD K6 CPU with MMX instruction set support. commit 4fe0b88159a9548f90db4e846f2033d51e1506c7 Author: Stefan Schulze Frielinghaus Date: Fri Sep 20 14:08:32 2024 +0200 s390: Fix strict_low_part generation In s390_expand_insv(), if generating code for ICM et al. src is a MEM and gen_lowpart might force src into a register such that we end up with patterns which do not match anymore. Use adjust_address() instead in order to preserve a MEM. Furthermore, it is not straight forward to enforce a subreg. For example, in case of a paradoxical subreg, gen_lowpart() may return a register. In order to compensate this, s390_gen_lowpart_subreg() emits a reference to a pseudo which does not coincide with its definition which is wrong. Additionally, if dest is a paradoxical subreg, then do not try to emit a strict_low_part since it could mean that dest was not initialized even though this might be fixed up later by init-regs. Splitter for insn *get_tp_64, *zero_extendhisi2_31, *zero_extendqisi2_31, *zero_extendqihi2_31 are applied after reload. Thus, operands[0] is a hard register and gen_lowpart (m, operands[0]) just returns the hard register for mode m which is fine to use as an argument for strict_low_part, i.e., we do not need to enforce subregs here since after reload subregs are supposed to be eliminated anyway. This fixes gcc.dg/torture/pr111821.c. gcc/ChangeLog: * config/s390/s390-protos.h (s390_gen_lowpart_subreg): Remove. * config/s390/s390.cc (s390_gen_lowpart_subreg): Remove. (s390_expand_insv): Use adjust_address() and emit a strict_low_part only in case of a natural subreg. * config/s390/s390.md: Use gen_lowpart() instead of s390_gen_lowpart_subreg(). (cherry picked from commit 9ebc9fbdddfe1ec85355b068354315a4da8e1ca0) diff --git a/gcc/config/s390/s390-protos.h b/gcc/config/s390/s390-protos.h index 78117c36e03..7ce4c52abe9 100644 --- a/gcc/config/s390/s390-protos.h +++ b/gcc/config/s390/s390-protos.h @@ -50,7 +50,6 @@ extern void s390_set_has_landing_pad_p (bool); extern bool s390_hard_regno_rename_ok (unsigned int, unsigned int); extern int s390_class_max_nregs (enum reg_class, machine_mode); extern bool s390_return_addr_from_memory(void); -extern rtx s390_gen_lowpart_subreg (machine_mode, rtx); extern bool s390_fma_allowed_p (machine_mode); #if S390_USE_TARGET_ATTRIBUTE extern tree s390_valid_target_attribute_tree (tree args, diff --git a/gcc/config/s390/s390.cc b/gcc/config/s390/s390.cc index ae0cf9ef5b9..b66fc5be871 100644 --- a/gcc/config/s390/s390.cc +++ b/gcc/config/s390/s390.cc @@ -457,31 +457,6 @@ s390_return_addr_from_memory () return cfun_gpr_save_slot(RETURN_REGNUM) == SAVE_SLOT_STACK; } -/* Generate a SUBREG for the MODE lowpart of EXPR. - - In contrast to gen_lowpart it will always return a SUBREG - expression. This is useful to generate STRICT_LOW_PART - expressions. */ -rtx -s390_gen_lowpart_subreg (machine_mode mode, rtx expr) -{ - rtx lowpart = gen_lowpart (mode, expr); - - /* There might be no SUBREG in case it could be applied to the hard - REG rtx or it could be folded with a paradoxical subreg. Bring - it back. */ - if (!SUBREG_P (lowpart)) - { - machine_mode reg_mode = TARGET_ZARCH ? DImode : SImode; - gcc_assert (REG_P (lowpart)); - lowpart = gen_lowpart_SUBREG (mode, - gen_rtx_REG (reg_mode, - REGNO (lowpart))); - } - - return lowpart; -} - /* Return nonzero if it's OK to use fused multiply-add for MODE. */ bool s390_fma_allowed_p (machine_mode mode) @@ -6544,15 +6519,21 @@ s390_expand_insv (rtx dest, rtx op1, rtx op2, rtx src) /* Emit a strict_low_part pattern if possible. */ if (smode_bsize == bitsize && bitpos == mode_bsize - smode_bsize) { - rtx low_dest = s390_gen_lowpart_subreg (smode, dest); - rtx low_src = gen_lowpart (smode, src); - - switch (smode) + rtx low_dest = gen_lowpart (smode, dest); + if (SUBREG_P (low_dest) && !paradoxical_subreg_p (low_dest)) { - case E_QImode: emit_insn (gen_movstrictqi (low_dest, low_src)); return true; - case E_HImode: emit_insn (gen_movstricthi (low_dest, low_src)); return true; - case E_SImode: emit_insn (gen_movstrictsi (low_dest, low_src)); return true; - default: break; + poly_int64 offset = GET_MODE_SIZE (mode) - GET_MODE_SIZE (smode); + rtx low_src = adjust_address (src, smode, offset); + switch (smode) + { + case E_QImode: emit_insn (gen_movstrictqi (low_dest, low_src)); + return true; + case E_HImode: emit_insn (gen_movstricthi (low_dest, low_src)); + return true; + case E_SImode: emit_insn (gen_movstrictsi (low_dest, low_src)); + return true; + default: break; + } } } diff --git a/gcc/config/s390/s390.md b/gcc/config/s390/s390.md index b8dcf2bb58a..2fc79969537 100644 --- a/gcc/config/s390/s390.md +++ b/gcc/config/s390/s390.md @@ -1971,12 +1971,11 @@ "TARGET_ZARCH" "#" "&& reload_completed" - [(set (match_dup 2) (match_dup 4)) + [(set (match_dup 2) (match_dup 3)) (set (match_dup 0) (ashift:DI (match_dup 0) (const_int 32))) - (set (strict_low_part (match_dup 3)) (match_dup 5))] + (set (strict_low_part (match_dup 2)) (match_dup 4))] "operands[2] = gen_lowpart (SImode, operands[0]); - operands[3] = s390_gen_lowpart_subreg (SImode, operands[0]); - s390_split_access_reg (operands[1], &operands[5], &operands[4]);") + s390_split_access_reg (operands[1], &operands[4], &operands[3]);") ; Splitters for storing TLS pointer to %a0:DI. @@ -5013,7 +5012,7 @@ (parallel [(set (strict_low_part (match_dup 2)) (match_dup 1)) (clobber (reg:CC CC_REGNUM))])] - "operands[2] = s390_gen_lowpart_subreg (HImode, operands[0]);") + "operands[2] = gen_lowpart (HImode, operands[0]);") (define_insn_and_split "*zero_extendqisi2_31" [(set (match_operand:SI 0 "register_operand" "=&d") @@ -5023,7 +5022,7 @@ "&& reload_completed" [(set (match_dup 0) (const_int 0)) (set (strict_low_part (match_dup 2)) (match_dup 1))] - "operands[2] = s390_gen_lowpart_subreg (QImode, operands[0]);") + "operands[2] = gen_lowpart (QImode, operands[0]);") ; ; zero_extendqihi2 instruction pattern(s). @@ -5055,7 +5054,7 @@ "&& reload_completed" [(set (match_dup 0) (const_int 0)) (set (strict_low_part (match_dup 2)) (match_dup 1))] - "operands[2] = s390_gen_lowpart_subreg (QImode, operands[0]);") + "operands[2] = gen_lowpart (QImode, operands[0]);") ; ; fixuns_trunc(dd|td|sf|df|tf)(si|di)2 expander commit 0f32c312508f11d2c6e9e363fa5b88d69eb9252c Author: Eric Botcazou Date: Fri Sep 20 12:32:13 2024 +0200 Fix small thinko in IPA mod/ref pass When a memory copy operation is analyzed by analyze_ssa_name, if both the load and store are made through the same SSA name, the store is overlooked. gcc/ * ipa-modref.cc (modref_eaf_analysis::analyze_ssa_name): Always process both the load and the store of a memory copy operation. gcc/testsuite/ * gcc.dg/ipa/modref-4.c: New test. diff --git a/gcc/ipa-modref.cc b/gcc/ipa-modref.cc index d41de9c59c1..c3e3fd50d17 100644 --- a/gcc/ipa-modref.cc +++ b/gcc/ipa-modref.cc @@ -2599,8 +2599,9 @@ modref_eaf_analysis::analyze_ssa_name (tree name, bool deferred) is used arbitrarily. */ if (memory_access_to (gimple_assign_rhs1 (assign), name)) m_lattice[index].merge (deref_flags (0, false)); + /* Handle *name = *exp. */ - else if (memory_access_to (gimple_assign_lhs (assign), name)) + if (memory_access_to (gimple_assign_lhs (assign), name)) m_lattice[index].merge_direct_store (); } /* Handle lhs = *name. */ diff --git a/gcc/testsuite/gcc.dg/ipa/modref-4.c b/gcc/testsuite/gcc.dg/ipa/modref-4.c new file mode 100644 index 00000000000..71ed1ca5f5b --- /dev/null +++ b/gcc/testsuite/gcc.dg/ipa/modref-4.c @@ -0,0 +1,34 @@ +/* { dg-options "-O" } */ +/* { dg-do run } */ + +static __attribute__((noipa)) int foo (void) +{ + return 1; +} + +int main (void) +{ + struct S { int a; int b; }; + struct T { struct S s; }; + + struct T t = { { 0, 0 } }; + struct T u; + + __attribute__((noinline)) void bar (void) + { + if (foo ()) + { + u = t; + /* OK with u.s.a = 0; */ + } + } + + u.s.a = 1; + + bar (); + + if (u.s.a != 0) + __builtin_abort (); + + return 0; +} commit 645a11f70ee63232f8a3f54cf4415074454cc6c9 Author: GCC Administrator Date: Fri Sep 20 17:37:53 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index c50ed2b2451..6db0a3c2152 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,27 @@ +2024-09-20 Eric Botcazou + + * ipa-modref.cc (modref_eaf_analysis::analyze_ssa_name): Always + process both the load and the store of a memory copy operation. + +2024-09-20 Stefan Schulze Frielinghaus + + Backported from master: + 2024-09-12 Stefan Schulze Frielinghaus + + * config/s390/s390-protos.h (s390_gen_lowpart_subreg): Remove. + * config/s390/s390.cc (s390_gen_lowpart_subreg): Remove. + (s390_expand_insv): Use adjust_address() and emit a + strict_low_part only in case of a natural subreg. + * config/s390/s390.md: Use gen_lowpart() instead of + s390_gen_lowpart_subreg(). + +2024-09-19 Haochen Jiang + + * doc/invoke.texi: Add corei7, corei7-avx, core-avx-i, + core-avx2, atom, and slm. Reorder the -march documentation by + splitting them into date-to-now products, P-core, E-core and + Xeon Phi. Refine the product names in documentation. + 2024-09-15 H.J. Lu Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index f2530dded09..09a8fff25e6 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240919 +20240920 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 2f2efe212dd..c15477dd05b 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,7 @@ +2024-09-20 Eric Botcazou + + * gcc.dg/ipa/modref-4.c: New test. + 2024-09-17 Marek Polacek Backported from master: commit cb25c5dd6b315dc216c7a5640dc89c5d74ffea34 Author: Harald Anlauf Date: Thu Sep 5 21:30:25 2024 +0200 Fortran: fix ICE in gfc_create_module_variable [PR100273] gcc/fortran/ChangeLog: PR fortran/100273 * trans-decl.cc (gfc_create_module_variable): Handle module variable also when it is needed for the result specification of a contained function. gcc/testsuite/ChangeLog: PR fortran/100273 * gfortran.dg/pr100273.f90: New test. (cherry picked from commit 1f462b5072a5e82c35921f7e3bdf3959c4a49dc9) diff --git a/gcc/fortran/trans-decl.cc b/gcc/fortran/trans-decl.cc index 0e91553108f..884978ad981 100644 --- a/gcc/fortran/trans-decl.cc +++ b/gcc/fortran/trans-decl.cc @@ -5251,7 +5251,8 @@ gfc_create_module_variable (gfc_symbol * sym) /* Create the variable. */ pushdecl (decl); gcc_assert (sym->ns->proc_name->attr.flavor == FL_MODULE - || (sym->ns->parent->proc_name->attr.flavor == FL_MODULE + || ((sym->ns->parent->proc_name->attr.flavor == FL_MODULE + || sym->ns->parent->proc_name->attr.flavor == FL_PROCEDURE) && sym->fn_result_spec)); DECL_CONTEXT (decl) = sym->ns->proc_name->backend_decl; rest_of_decl_compilation (decl, 1, 0); diff --git a/gcc/testsuite/gfortran.dg/pr100273.f90 b/gcc/testsuite/gfortran.dg/pr100273.f90 new file mode 100644 index 00000000000..f71947ad802 --- /dev/null +++ b/gcc/testsuite/gfortran.dg/pr100273.f90 @@ -0,0 +1,26 @@ +! { dg-do compile } +! PR fortran/100273 - ICE in gfc_create_module_variable +! +! Contributed by G.Steinmetz + +module m + implicit none +contains + character(4) function g(k) + integer :: k + g = f(k) + contains + function f(n) + character(3), parameter :: a(2) = ['1 ', '123'] + integer :: n + character(len_trim(a(n))) :: f + f = 'abc' + end + end +end +program p + use m + implicit none + print *, '>>' // g(1) // '<<' + print *, '>>' // g(2) // '<<' +end commit a761f1007f066aec55b26dcd600f5d28185ec2ac Author: GCC Administrator Date: Sat Sep 21 00:19:46 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 09a8fff25e6..6df304d1af8 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240920 +20240921 diff --git a/gcc/fortran/ChangeLog b/gcc/fortran/ChangeLog index 1f360364d33..cf1b77de38b 100644 --- a/gcc/fortran/ChangeLog +++ b/gcc/fortran/ChangeLog @@ -1,3 +1,13 @@ +2024-09-20 Harald Anlauf + + Backported from master: + 2024-09-05 Harald Anlauf + + PR fortran/100273 + * trans-decl.cc (gfc_create_module_variable): Handle module + variable also when it is needed for the result specification + of a contained function. + 2024-08-05 Paul Thomas Backported from master: diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index c15477dd05b..5fa325e4c6c 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-09-20 Harald Anlauf + + Backported from master: + 2024-09-05 Harald Anlauf + + PR fortran/100273 + * gfortran.dg/pr100273.f90: New test. + 2024-09-20 Eric Botcazou * gcc.dg/ipa/modref-4.c: New test. commit 2a6e9bfcd3904d1ddc9e359d366da1dbbd49b971 Author: GCC Administrator Date: Sun Sep 22 00:20:46 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 6df304d1af8..c43582073d8 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240921 +20240922 commit 917b6c6a897da7741c0e404c48ce358852749cc0 Author: GCC Administrator Date: Mon Sep 23 00:19:53 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c43582073d8..37e5425b64e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240922 +20240923 commit 52bb3a257d774b6d55b3524e3f2897e1db9b6264 Author: GCC Administrator Date: Tue Sep 24 00:20:06 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 37e5425b64e..f4255371a62 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240923 +20240924 commit 50c8048de95e180f632385348b09eda12893a22c Author: GCC Administrator Date: Wed Sep 25 00:20:31 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index f4255371a62..e6fc5ab9d83 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240924 +20240925 commit 596d857e68d727703458639af05a6c4b3ea1ddb1 Author: GCC Administrator Date: Thu Sep 26 00:21:00 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e6fc5ab9d83..6444ce3348b 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240925 +20240926 commit 63a5a1fbb7b5431ea3aa228490abdeeef4625183 Author: GCC Administrator Date: Fri Sep 27 00:20:45 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 6444ce3348b..b84faf4bb87 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240926 +20240927 commit 8d29e1c4ceaea4d3ceec6b51de5b7c31a6bc5f85 Author: Stefan Schulze Frielinghaus Date: Fri Sep 27 12:45:42 2024 +0200 s390: Fix AQ and AR constraints Ensure for AQ and AR constraints that the resulting displacement after adding any positive offset less than the size of the object being referenced is still valid. gcc/ChangeLog: * config/s390/s390.cc (s390_mem_constraint): Check displacement for AQ and AR constraints. (cherry picked from commit 1a71ff3b89aadc7fa0af0bca269d74bb23c1a957) diff --git a/gcc/config/s390/s390.cc b/gcc/config/s390/s390.cc index b66fc5be871..2be3a873b89 100644 --- a/gcc/config/s390/s390.cc +++ b/gcc/config/s390/s390.cc @@ -3405,6 +3405,18 @@ s390_mem_constraint (const char *str, rtx op) if ((reload_completed || reload_in_progress) ? !offsettable_memref_p (op) : !offsettable_nonstrict_memref_p (op)) return 0; + /* offsettable_memref_p ensures only that any positive offset added to + the address forms a valid general address. For AQ and AR constraints + we also have to verify that the resulting displacement after adding + any positive offset less than the size of the object being referenced + is still valid. */ + if (str[1] == 'Q' || str[1] == 'R') + { + int o = GET_MODE_SIZE (GET_MODE (op)) - 1; + rtx tmp = adjust_address (op, QImode, o); + if (!s390_check_qrst_address (str[1], XEXP (tmp, 0), true)) + return 0; + } return s390_check_qrst_address (str[1], XEXP (op, 0), true); case 'B': /* Check for non-literal-pool variants of memory constraints. */ commit 7051fa5fa4eaa24785a64072490c1e0c65039915 Author: Stefan Schulze Frielinghaus Date: Fri Sep 27 12:45:42 2024 +0200 s390: Fix TF to FPRX2 conversion [PR115860] Currently subregs originating from *tf_to_fprx2_0 and *tf_to_fprx2_1 survive register allocation. This in turn leads to wrong register renaming. Keeping the current approach would mean we need two insns for *tf_to_fprx2_0 and *tf_to_fprx2_1, respectively. Something along the lines (define_insn "*tf_to_fprx2_0" [(set (subreg:DF (match_operand:FPRX2 0 "nonimmediate_operand" "=f") 0) (unspec:DF [(match_operand:TF 1 "general_operand" "v")] UNSPEC_TF_TO_FPRX2_0))] "TARGET_VXE" "#") (define_insn "*tf_to_fprx2_0" [(set (match_operand:DF 0 "nonimmediate_operand" "=f") (unspec:DF [(match_operand:TF 1 "general_operand" "v")] UNSPEC_TF_TO_FPRX2_0))] "TARGET_VXE" "vpdi\t%v0,%v1,%v0,1 [(set_attr "op_type" "VRR")]) and similar for *tf_to_fprx2_1. Note, pre register allocation operand 0 has mode FPRX2 and afterwards DF once subregs have been eliminated. Since we always copy a whole vector register into a floating-point register pair, another way to fix this is to merge *tf_to_fprx2_0 and *tf_to_fprx2_1 into a single insn which means we don't have to use subregs at all. The downside of this is that the assembler template contains two instructions, now. The upside is that we don't have to come up with some artificial insn before RA which might be more readable/maintainable. That is implemented by this patch. In commit r11-4872-ge627cda5686592, the output operand specifier %V was introduced which is used in tf_to_fprx2 only, now. Instead of coming up with its counterpart %F for floating-point registers, which would also only be used in tf_to_fprx2, I print the operands directly. This renders %V unused which is why it is removed by this patch. gcc/ChangeLog: PR target/115860 * config/s390/s390.cc (print_operand): Remove operand specifier %V. * config/s390/s390.md (UNSPEC_TF_TO_FPRX2): New. * config/s390/vector.md (*tf_to_fprx2_0): Remove. (*tf_to_fprx2_1): Remove. (tf_to_fprx2): New. gcc/testsuite/ChangeLog: * gcc.target/s390/vector/long-double-asm-abi.c: Adapt scan-assembler directive. * gcc.target/s390/vector/long-double-to-i64.c: Adapt scan-assembler directive. * gcc.target/s390/pr115860-1.c: New test. (cherry picked from commit 46c2538435dfc50dd5c67c4e03ce387d1f6ebe9b) diff --git a/gcc/config/s390/s390.cc b/gcc/config/s390/s390.cc index 2be3a873b89..a8f804ffe4f 100644 --- a/gcc/config/s390/s390.cc +++ b/gcc/config/s390/s390.cc @@ -8028,7 +8028,6 @@ print_operand_address (FILE *file, rtx addr) CONST_VECTOR: Generate a bitmask for vgbm instruction. 'x': print integer X as if it's an unsigned halfword. 'v': print register number as vector register (v1 instead of f1). - 'V': print the second word of a TFmode operand as vector register. */ void @@ -8221,13 +8220,13 @@ print_operand (FILE *file, rtx x, int code) case REG: /* Print FP regs as fx instead of vx when they are accessed through non-vector mode. */ - if ((code == 'v' || code == 'V') + if (code == 'v' || VECTOR_NOFP_REG_P (x) || (FP_REG_P (x) && VECTOR_MODE_P (GET_MODE (x))) || (VECTOR_REG_P (x) && (GET_MODE_SIZE (GET_MODE (x)) / s390_class_max_nregs (FP_REGS, GET_MODE (x))) > 8)) - fprintf (file, "%%v%s", reg_names[REGNO (x) + (code == 'V')] + 2); + fprintf (file, "%%v%s", reg_names[REGNO (x)] + 2); else fprintf (file, "%s", reg_names[REGNO (x)]); break; diff --git a/gcc/config/s390/s390.md b/gcc/config/s390/s390.md index 2fc79969537..335aff9884e 100644 --- a/gcc/config/s390/s390.md +++ b/gcc/config/s390/s390.md @@ -244,6 +244,8 @@ UNSPEC_VEC_ELTSWAP + UNSPEC_TF_TO_FPRX2 + UNSPEC_NNPA_VCLFNHS_V8HI UNSPEC_NNPA_VCLFNLS_V8HI UNSPEC_NNPA_VCRNFS_V8HI diff --git a/gcc/config/s390/vector.md b/gcc/config/s390/vector.md index 75912280c23..ac3816a6f5c 100644 --- a/gcc/config/s390/vector.md +++ b/gcc/config/s390/vector.md @@ -900,36 +900,45 @@ "vmrlg\t%0,%1,%2"; [(set_attr "op_type" "VRR")]) - -(define_insn "*tf_to_fprx2_0" - [(set (subreg:DF (match_operand:FPRX2 0 "nonimmediate_operand" "+f") 0) - (subreg:DF (match_operand:TF 1 "general_operand" "v") 0))] - "TARGET_VXE" - ; M4 == 1 corresponds to %v0[0] = %v1[0]; %v0[1] = %v0[1]; - "vpdi\t%v0,%v1,%v0,1" - [(set_attr "op_type" "VRR")]) - -(define_insn "*tf_to_fprx2_1" - [(set (subreg:DF (match_operand:FPRX2 0 "nonimmediate_operand" "+f") 8) - (subreg:DF (match_operand:TF 1 "general_operand" "v") 8))] +(define_insn "tf_to_fprx2" + [(set (match_operand:FPRX2 0 "register_operand" "=f,f ,f") + (unspec:FPRX2 [(match_operand:TF 1 "general_operand" "v,AR,AT")] + UNSPEC_TF_TO_FPRX2))] "TARGET_VXE" - ; M4 == 5 corresponds to %V0[0] = %v1[1]; %V0[1] = %V0[1]; - "vpdi\t%V0,%v1,%V0,5" - [(set_attr "op_type" "VRR")]) - -(define_insn_and_split "tf_to_fprx2" - [(set (match_operand:FPRX2 0 "nonimmediate_operand" "=f,f") - (subreg:FPRX2 (match_operand:TF 1 "general_operand" "v,AR") 0))] - "TARGET_VXE" - "#" - "!(MEM_P (operands[1]) && MEM_VOLATILE_P (operands[1]))" - [(set (match_dup 2) (match_dup 3)) - (set (match_dup 4) (match_dup 5))] { - operands[2] = simplify_gen_subreg (DFmode, operands[0], FPRX2mode, 0); - operands[3] = simplify_gen_subreg (DFmode, operands[1], TFmode, 0); - operands[4] = simplify_gen_subreg (DFmode, operands[0], FPRX2mode, 8); - operands[5] = simplify_gen_subreg (DFmode, operands[1], TFmode, 8); + char buf[64]; + const char *reg_pair = reg_names[REGNO (operands[0]) + 1]; + switch (which_alternative) + { + case 0: + if (REGNO (operands[0]) == REGNO (operands[1])) + { + reg_pair += 2; // get rid of prefix %f + snprintf (buf, sizeof (buf), "vpdi\t%%%%v%s,%%v1,%%%%v%s,5", reg_pair, reg_pair); + output_asm_insn (buf, operands); + return ""; + } + else + { + reg_pair += 2; // get rid of prefix %f + snprintf (buf, sizeof (buf), "ldr\t%%f0,%%f1;vpdi\t%%%%v%s,%%v1,%%%%v%s,5", reg_pair, reg_pair); + output_asm_insn (buf, operands); + return ""; + } + case 1: + { + snprintf (buf, sizeof (buf), "ld\t%%f0,%%1;ld\t%%%s,8+%%1", reg_pair); + output_asm_insn (buf, operands); + return ""; + } + case 2: + { + snprintf (buf, sizeof (buf), "ldy\t%%f0,%%1;ldy\t%%%s,8+%%1", reg_pair); + output_asm_insn (buf, operands); + return ""; + } + default: gcc_unreachable (); + } }) @@ -2548,9 +2557,8 @@ ; There is no instruction for rounding an extended BFP operand in a VR into ; a signed integer, therefore copy it into a FPR pair first. (define_expand "fix_trunctf2_vr" - [(set (subreg:DF (match_dup 2) 0) - (subreg:DF (match_operand:TF 1 "register_operand" "") 0)) - (set (subreg:DF (match_dup 2) 8) (subreg:DF (match_dup 1) 8)) + [(set (match_dup 2) + (unspec:FPRX2 [(match_operand:TF 1 "register_operand")] UNSPEC_TF_TO_FPRX2)) (parallel [(set (match_operand:GPR 0 "register_operand" "") (fix:GPR (match_dup 2))) (unspec:GPR [(const_int BFP_RND_TOWARD_0)] UNSPEC_ROUND) @@ -2582,9 +2590,8 @@ ; There is no instruction for rounding an extended BFP operand in a VR into ; an unsigned integer, therefore copy it into a FPR pair first. (define_expand "fixuns_trunctf2_vr" - [(set (subreg:DF (match_dup 2) 0) - (subreg:DF (match_operand:TF 1 "register_operand" "") 0)) - (set (subreg:DF (match_dup 2) 8) (subreg:DF (match_dup 1) 8)) + [(set (match_dup 2) + (unspec:FPRX2 [(match_operand:TF 1 "register_operand")] UNSPEC_TF_TO_FPRX2)) (parallel [(set (match_operand:GPR 0 "register_operand" "") (unsigned_fix:GPR (match_dup 2))) (unspec:GPR [(const_int BFP_RND_TOWARD_0)] UNSPEC_ROUND) diff --git a/gcc/testsuite/gcc.target/s390/pr115860-1.c b/gcc/testsuite/gcc.target/s390/pr115860-1.c new file mode 100644 index 00000000000..abcddeaed95 --- /dev/null +++ b/gcc/testsuite/gcc.target/s390/pr115860-1.c @@ -0,0 +1,26 @@ +/* { dg-do run } */ +/* { dg-require-effective-target s390_vxe } */ +/* { dg-options "-O2 -march=z14 -mzarch" } */ + +__attribute__ ((noipa)) +long long trunctf (long double x) +{ + /* Ensure via ++x that x is in a register. */ + ++x; + return x; +} + +__attribute__ ((noipa)) +long long trunctf_from_mem (long double x) +{ + return x; +} + +int main (void) +{ + if (trunctf (0x7ffffffffffffffeLL) != 0x7fffffffffffffffLL) + __builtin_abort (); + if (trunctf_from_mem (0x7fffffffffffffffLL) != 0x7fffffffffffffffLL) + __builtin_abort (); + return 0; +} diff --git a/gcc/testsuite/gcc.target/s390/vector/long-double-asm-abi.c b/gcc/testsuite/gcc.target/s390/vector/long-double-asm-abi.c index f9f2d1286e2..ac3576801f0 100644 --- a/gcc/testsuite/gcc.target/s390/vector/long-double-asm-abi.c +++ b/gcc/testsuite/gcc.target/s390/vector/long-double-asm-abi.c @@ -14,7 +14,7 @@ xsqrt (long double x) /* Check that the generated code is very small and straightforward. In particular, there must be no unnecessary copying and no stack frame. */ -/* { dg-final { scan-assembler {\n\tld\t[^\n]*\n\tld\t[^\n]*\n(#[^\n]*\n)*\tsqxbr\t.*\n(#[^\n]*\n)*\tstd\t[^\n]*\n\tstd\t[^\n]*\n\tbr\t%r14\n} } } */ +/* { dg-final { scan-assembler {\n\tld\t[^\n]*;ld\t[^\n]*\n(#[^\n]*\n)*\tsqxbr\t.*\n(#[^\n]*\n)*\tstd\t[^\n]*\n\tstd\t[^\n]*\n\tbr\t%r14\n} } } */ int main (void) diff --git a/gcc/testsuite/gcc.target/s390/vector/long-double-to-i64.c b/gcc/testsuite/gcc.target/s390/vector/long-double-to-i64.c index 2dbbb5d1c03..ed47fd9b858 100644 --- a/gcc/testsuite/gcc.target/s390/vector/long-double-to-i64.c +++ b/gcc/testsuite/gcc.target/s390/vector/long-double-to-i64.c @@ -10,8 +10,6 @@ long_double_to_i64 (long double x) return x; } -/* { dg-final { scan-assembler-times {\n\tvpdi\t%v\d+,%v\d+,%v\d+,1\n} 1 } } */ -/* { dg-final { scan-assembler-times {\n\tvpdi\t%v\d+,%v\d+,%v\d+,5\n} 1 } } */ /* { dg-final { scan-assembler-times {\n\tcgxbr\t} 1 } } */ int commit e282606b6cfa0981ddf234984450007f815bd860 Author: GCC Administrator Date: Sat Sep 28 00:20:43 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 6db0a3c2152..e7795f0dd6d 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,24 @@ +2024-09-27 Stefan Schulze Frielinghaus + + Backported from master: + 2024-09-13 Stefan Schulze Frielinghaus + + PR target/115860 + * config/s390/s390.cc (print_operand): Remove operand specifier + %V. + * config/s390/s390.md (UNSPEC_TF_TO_FPRX2): New. + * config/s390/vector.md (*tf_to_fprx2_0): Remove. + (*tf_to_fprx2_1): Remove. + (tf_to_fprx2): New. + +2024-09-27 Stefan Schulze Frielinghaus + + Backported from master: + 2024-09-13 Stefan Schulze Frielinghaus + + * config/s390/s390.cc (s390_mem_constraint): Check displacement + for AQ and AR constraints. + 2024-09-20 Eric Botcazou * ipa-modref.cc (modref_eaf_analysis::analyze_ssa_name): Always diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index b84faf4bb87..d1c17555461 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240927 +20240928 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 5fa325e4c6c..77308773b19 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,14 @@ +2024-09-27 Stefan Schulze Frielinghaus + + Backported from master: + 2024-09-13 Stefan Schulze Frielinghaus + + * gcc.target/s390/vector/long-double-asm-abi.c: Adapt + scan-assembler directive. + * gcc.target/s390/vector/long-double-to-i64.c: Adapt + scan-assembler directive. + * gcc.target/s390/pr115860-1.c: New test. + 2024-09-20 Harald Anlauf Backported from master: commit 2e66eb7e7eae82bcd6675e79eabbdd6decfa9fe5 Author: H.J. Lu Date: Wed Sep 25 16:39:04 2024 +0800 x86: Don't use address override with segment regsiter Address override only applies to the (reg32) part in the thread address fs:(reg32). Don't rewrite thread address like (set (reg:CCZ 17 flags) (compare:CCZ (reg:SI 98 [ __gmpfr_emax.0_1 ]) (mem/c:SI (plus:SI (plus:SI (unspec:SI [ (const_int 0 [0]) ] UNSPEC_TP) (reg:SI 107)) (const:SI (unspec:SI [ (symbol_ref:SI ("previous_emax") [flags 0x1a] ) ] UNSPEC_DTPOFF))) [1 previous_emax+0 S4 A32]))) if address override is used to avoid the invalid memory operand like cmpl %fs:previous_emax@dtpoff(%eax), %r12d gcc/ PR target/116839 * config/i386/i386.cc (ix86_rewrite_tls_address_1): Make it static. Return if TLS address is thread register plus an integer register. gcc/testsuite/ PR target/116839 * gcc.target/i386/pr116839.c: New file. Signed-off-by: H.J. Lu (cherry picked from commit c79cc30862d7255ca15884aa956d1ccfa279d86a) diff --git a/gcc/config/i386/i386.cc b/gcc/config/i386/i386.cc index b52eb0d5f7b..bf8553e3dd0 100644 --- a/gcc/config/i386/i386.cc +++ b/gcc/config/i386/i386.cc @@ -11787,7 +11787,7 @@ ix86_tls_address_pattern_p (rtx op) } /* Rewrite *LOC so that it refers to a default TLS address space. */ -void +static void ix86_rewrite_tls_address_1 (rtx *loc) { subrtx_ptr_iterator::array_type array; @@ -11809,6 +11809,13 @@ ix86_rewrite_tls_address_1 (rtx *loc) if (GET_CODE (u) == UNSPEC && XINT (u, 1) == UNSPEC_TP) { + /* NB: Since address override only applies to the + (reg32) part in fs:(reg32), return if address + override is used. */ + if (Pmode != word_mode + && REG_P (XEXP (*x, 1 - i))) + return; + addr_space_t as = DEFAULT_TLS_SEG_REG; *x = XEXP (*x, 1 - i); diff --git a/gcc/testsuite/gcc.target/i386/pr116839.c b/gcc/testsuite/gcc.target/i386/pr116839.c new file mode 100644 index 00000000000..e5df8256251 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr116839.c @@ -0,0 +1,48 @@ +/* { dg-do compile { target { ! ia32 } } } */ +/* { dg-require-effective-target maybe_x32 } */ +/* { dg-options "-mx32 -O2 -fPIC -mtls-dialect=gnu2" } */ +/* { dg-final { scan-assembler-not "cmpl\[ \t\]+%fs:previous_emax@dtpoff\\(%eax\\)" } } */ + +typedef long mpfr_prec_t; +typedef long mpfr_exp_t; +typedef struct { + mpfr_prec_t _mpfr_prec; +} __mpfr_struct; +typedef __mpfr_struct mpfr_t[1]; +extern _Thread_local mpfr_exp_t __gmpfr_emax; +static _Thread_local mpfr_exp_t previous_emax; +static _Thread_local mpfr_t bound_emax; +extern const mpfr_t __gmpfr_const_log2_RNDD; +extern const mpfr_t __gmpfr_const_log2_RNDU; + +typedef enum { + MPFR_RNDN=0, + MPFR_RNDZ, + MPFR_RNDU, + MPFR_RNDD, + MPFR_RNDA, + MPFR_RNDF, + MPFR_RNDNA=-1 +} mpfr_rnd_t; +typedef __mpfr_struct *mpfr_ptr; +typedef const __mpfr_struct *mpfr_srcptr; +void mpfr_mul (mpfr_ptr, mpfr_srcptr, mpfr_rnd_t); + +void +foo (void) +{ + mpfr_exp_t saved_emax; + + if (__gmpfr_emax != previous_emax) + { + saved_emax = __gmpfr_emax; + + bound_emax->_mpfr_prec = 32; + + mpfr_mul (bound_emax, saved_emax < 0 ? + __gmpfr_const_log2_RNDD : __gmpfr_const_log2_RNDU, + MPFR_RNDU); + previous_emax = saved_emax; + __gmpfr_emax = saved_emax; + } +} commit 54806268b47775449c7e237f8f03e922d6da26f6 Author: Jan Hubicka Date: Mon Mar 18 10:22:44 2024 +0100 Add AMD znver5 processor enablement with scheduler model 2024-02-14 Jan Hubicka Karthiban Anbazhagan gcc/ChangeLog: * common/config/i386/cpuinfo.h (get_amd_cpu): Recognize znver5. * common/config/i386/i386-common.cc (processor_names): Add znver5. (processor_alias_table): Likewise. * common/config/i386/i386-cpuinfo.h (processor_types): Add new zen family. (processor_subtypes): Add znver5. * config.gcc (x86_64-*-* |...): Likewise. * config/i386/driver-i386.cc (host_detect_local_cpu): Let march=native detect znver5 cpu's. * config/i386/i386-c.cc (ix86_target_macros_internal): Add znver5. * config/i386/i386-options.cc (m_ZNVER5): New definition (processor_cost_table): Add znver5. * config/i386/i386.cc (ix86_reassociation_width): Likewise. * config/i386/i386.h (processor_type): Add PROCESSOR_ZNVER5 (PTA_ZNVER5): New definition. * config/i386/i386.md (define_attr "cpu"): Add znver5. (Scheduling descriptions) Add znver5.md. * config/i386/x86-tune-costs.h (znver5_cost): New definition. * config/i386/x86-tune-sched.cc (ix86_issue_rate): Add znver5. (ix86_adjust_cost): Likewise. * config/i386/x86-tune.def (avx512_move_by_pieces): Add m_ZNVER5. (avx512_store_by_pieces): Add m_ZNVER5. * doc/extend.texi: Add znver5. * doc/invoke.texi: Likewise. * config/i386/znver4.md: Rename to zn4zn5.md; combine znver4 and znver5 Scheduler. gcc/testsuite/ChangeLog: * g++.target/i386/mv29.C: Handle znver5 arch. * gcc.target/i386/funcspec-56.inc:Likewise. (cherry picked from commit d0aa0af9a9b7dd709a8c7ff6604ed6b7da0fc23a) diff --git a/gcc/common/config/i386/cpuinfo.h b/gcc/common/config/i386/cpuinfo.h index 316ad3cb3e9..d79534331f7 100644 --- a/gcc/common/config/i386/cpuinfo.h +++ b/gcc/common/config/i386/cpuinfo.h @@ -282,6 +282,22 @@ get_amd_cpu (struct __processor_model *cpu_model, cpu_model->__cpu_subtype = AMDFAM19H_ZNVER3; } break; + case 0x1a: + cpu_model->__cpu_type = AMDFAM1AH; + if (model <= 0x77) + { + cpu = "znver5"; + CHECK___builtin_cpu_is ("znver5"); + cpu_model->__cpu_subtype = AMDFAM1AH_ZNVER5; + } + else if (has_cpu_feature (cpu_model, cpu_features2, + FEATURE_AVX512VP2INTERSECT)) + { + cpu = "znver5"; + CHECK___builtin_cpu_is ("znver5"); + cpu_model->__cpu_subtype = AMDFAM1AH_ZNVER5; + } + break; default: break; } diff --git a/gcc/common/config/i386/i386-common.cc b/gcc/common/config/i386/i386-common.cc index e2594cae4cc..a01172cab2f 100644 --- a/gcc/common/config/i386/i386-common.cc +++ b/gcc/common/config/i386/i386-common.cc @@ -1831,7 +1831,8 @@ const char *const processor_names[] = "znver1", "znver2", "znver3", - "znver4" + "znver4", + "znver5" }; /* Guarantee that the array is aligned with enum processor_type. */ @@ -2067,6 +2068,9 @@ const pta processor_alias_table[] = {"znver4", PROCESSOR_ZNVER4, CPU_ZNVER4, PTA_ZNVER4, M_CPU_SUBTYPE (AMDFAM19H_ZNVER4), P_PROC_AVX512F}, + {"znver5", PROCESSOR_ZNVER5, CPU_ZNVER5, + PTA_ZNVER5, + M_CPU_SUBTYPE (AMDFAM1AH_ZNVER5), P_PROC_AVX512F}, {"btver1", PROCESSOR_BTVER1, CPU_GENERIC, PTA_64BIT | PTA_MMX | PTA_SSE | PTA_SSE2 | PTA_SSE3 | PTA_SSSE3 | PTA_SSE4A | PTA_ABM | PTA_CX16 | PTA_PRFCHW diff --git a/gcc/common/config/i386/i386-cpuinfo.h b/gcc/common/config/i386/i386-cpuinfo.h index 82996ebb32d..18a6b4bd0bc 100644 --- a/gcc/common/config/i386/i386-cpuinfo.h +++ b/gcc/common/config/i386/i386-cpuinfo.h @@ -58,6 +58,7 @@ enum processor_types INTEL_GOLDMONT_PLUS, INTEL_TREMONT, AMDFAM19H, + AMDFAM1AH, CPU_TYPE_MAX, BUILTIN_CPU_TYPE_MAX = CPU_TYPE_MAX }; @@ -92,6 +93,7 @@ enum processor_subtypes AMDFAM19H_ZNVER3, INTEL_COREI7_ROCKETLAKE, AMDFAM19H_ZNVER4, + AMDFAM1AH_ZNVER5, CPU_SUBTYPE_MAX }; diff --git a/gcc/config.gcc b/gcc/config.gcc index 5c378c698ff..b0eef5572c9 100644 --- a/gcc/config.gcc +++ b/gcc/config.gcc @@ -663,9 +663,9 @@ c7 esther" # 64-bit x86 processors supported by --with-arch=. Each processor # MUST be separated by exactly one space. x86_64_archs="amdfam10 athlon64 athlon64-sse3 barcelona bdver1 bdver2 \ -bdver3 bdver4 znver1 znver2 znver3 znver4 btver1 btver2 k8 k8-sse3 opteron \ -opteron-sse3 nocona core2 corei7 corei7-avx core-avx-i core-avx2 atom \ -slm nehalem westmere sandybridge ivybridge haswell broadwell bonnell \ +bdver3 bdver4 znver1 znver2 znver3 znver4 znver5 btver1 btver2 k8 k8-sse3 \ +opteron opteron-sse3 nocona core2 corei7 corei7-avx core-avx-i core-avx2 \ +atom slm nehalem westmere sandybridge ivybridge haswell broadwell bonnell \ silvermont knl knm skylake-avx512 cannonlake icelake-client icelake-server \ skylake goldmont goldmont-plus tremont cascadelake tigerlake cooperlake \ sapphirerapids alderlake rocketlake eden-x2 nano nano-1000 nano-2000 nano-3000 \ @@ -3744,6 +3744,10 @@ case ${target} in arch=znver4 cpu=znver4 ;; + znver5-*) + arch=znver5 + cpu=znver5 + ;; bdver4-*) arch=bdver4 cpu=bdver4 @@ -3873,6 +3877,10 @@ case ${target} in arch=znver4 cpu=znver4 ;; + znver5-*) + arch=znver5 + cpu=znver5 + ;; bdver4-*) arch=bdver4 cpu=bdver4 diff --git a/gcc/config/i386/driver-i386.cc b/gcc/config/i386/driver-i386.cc index 3b5161aeddc..8d19838b10e 100644 --- a/gcc/config/i386/driver-i386.cc +++ b/gcc/config/i386/driver-i386.cc @@ -464,6 +464,8 @@ const char *host_detect_local_cpu (int argc, const char **argv) processor = PROCESSOR_GEODE; else if (has_feature (FEATURE_MOVBE) && family == 22) processor = PROCESSOR_BTVER2; + else if (has_feature (FEATURE_AVX512VP2INTERSECT)) + processor = PROCESSOR_ZNVER5; else if (has_feature (FEATURE_AVX512F)) processor = PROCESSOR_ZNVER4; else if (has_feature (FEATURE_VAES)) @@ -769,6 +771,9 @@ const char *host_detect_local_cpu (int argc, const char **argv) case PROCESSOR_ZNVER4: cpu = "znver4"; break; + case PROCESSOR_ZNVER5: + cpu = "znver5"; + break; case PROCESSOR_BTVER1: cpu = "btver1"; break; diff --git a/gcc/config/i386/i386-c.cc b/gcc/config/i386/i386-c.cc index 3fec4c7e245..989e23e1b9f 100644 --- a/gcc/config/i386/i386-c.cc +++ b/gcc/config/i386/i386-c.cc @@ -136,6 +136,10 @@ ix86_target_macros_internal (HOST_WIDE_INT isa_flag, def_or_undef (parse_in, "__znver4"); def_or_undef (parse_in, "__znver4__"); break; + case PROCESSOR_ZNVER5: + def_or_undef (parse_in, "__znver5"); + def_or_undef (parse_in, "__znver5__"); + break; case PROCESSOR_BTVER1: def_or_undef (parse_in, "__btver1"); def_or_undef (parse_in, "__btver1__"); @@ -333,6 +337,9 @@ ix86_target_macros_internal (HOST_WIDE_INT isa_flag, case PROCESSOR_ZNVER4: def_or_undef (parse_in, "__tune_znver4__"); break; + case PROCESSOR_ZNVER5: + def_or_undef (parse_in, "__tune_znver5__"); + break; case PROCESSOR_BTVER1: def_or_undef (parse_in, "__tune_btver1__"); break; diff --git a/gcc/config/i386/i386-options.cc b/gcc/config/i386/i386-options.cc index ad496ea5a8e..2daa722b2f3 100644 --- a/gcc/config/i386/i386-options.cc +++ b/gcc/config/i386/i386-options.cc @@ -158,11 +158,12 @@ along with GCC; see the file COPYING3. If not see #define m_ZNVER2 (HOST_WIDE_INT_1U<integer move cost is 2. */ + + /* reg-reg moves are done by renaming and thus they are even cheaper than + 1 cycle. Because reg-reg move cost is 2 and following tables correspond + to doubles of latencies, we do not model this correctly. It does not + seem to make practical difference to bump prices up even more. */ + 6, /* cost for loading QImode using + movzbl. */ + {6, 6, 6}, /* cost of loading integer registers + in QImode, HImode and SImode. + Relative to reg-reg move (2). */ + {8, 8, 8}, /* cost of storing integer + registers. */ + 2, /* cost of reg,reg fld/fst. */ + {14, 14, 17}, /* cost of loading fp registers + in SFmode, DFmode and XFmode. */ + {12, 12, 16}, /* cost of storing fp registers + in SFmode, DFmode and XFmode. */ + 2, /* cost of moving MMX register. */ + {6, 6}, /* cost of loading MMX registers + in SImode and DImode. */ + {8, 8}, /* cost of storing MMX registers + in SImode and DImode. */ + 2, 2, 3, /* cost of moving XMM,YMM,ZMM + register. */ + {6, 6, 10, 10, 12}, /* cost of loading SSE registers + in 32,64,128,256 and 512-bit. */ + {8, 8, 8, 12, 12}, /* cost of storing SSE registers + in 32,64,128,256 and 512-bit. */ + 6, 8, /* SSE->integer and integer->SSE + moves. */ + 8, 8, /* mask->integer and integer->mask moves */ + {6, 6, 6}, /* cost of loading mask register + in QImode, HImode, SImode. */ + {8, 8, 8}, /* cost if storing mask register + in QImode, HImode, SImode. */ + 2, /* cost of moving mask register. */ + /* End of register allocator costs. */ + }, + + COSTS_N_INSNS (1), /* cost of an add instruction. */ + /* TODO: Lea with 3 components has cost 2. */ + COSTS_N_INSNS (1), /* cost of a lea instruction. */ + COSTS_N_INSNS (1), /* variable shift costs. */ + COSTS_N_INSNS (1), /* constant shift costs. */ + {COSTS_N_INSNS (3), /* cost of starting multiply for QI. */ + COSTS_N_INSNS (3), /* HI. */ + COSTS_N_INSNS (3), /* SI. */ + COSTS_N_INSNS (3), /* DI. */ + COSTS_N_INSNS (3)}, /* other. */ + 0, /* cost of multiply per each bit + set. */ + {COSTS_N_INSNS (10), /* cost of a divide/mod for QI. */ + COSTS_N_INSNS (11), /* HI. */ + COSTS_N_INSNS (13), /* SI. */ + COSTS_N_INSNS (16), /* DI. */ + COSTS_N_INSNS (16)}, /* other. */ + COSTS_N_INSNS (1), /* cost of movsx. */ + COSTS_N_INSNS (1), /* cost of movzx. */ + 8, /* "large" insn. */ + 9, /* MOVE_RATIO. */ + 6, /* CLEAR_RATIO */ + {6, 6, 6}, /* cost of loading integer registers + in QImode, HImode and SImode. + Relative to reg-reg move (2). */ + {8, 8, 8}, /* cost of storing integer + registers. */ + {6, 6, 10, 10, 12}, /* cost of loading SSE registers + in 32bit, 64bit, 128bit, 256bit and 512bit */ + {8, 8, 8, 12, 12}, /* cost of storing SSE register + in 32bit, 64bit, 128bit, 256bit and 512bit */ + {6, 6, 6, 6, 6}, /* cost of unaligned loads. */ + {8, 8, 8, 8, 8}, /* cost of unaligned stores. */ + 2, 2, 2, /* cost of moving XMM,YMM,ZMM + register. */ + 6, /* cost of moving SSE register to integer. */ + /* VGATHERDPD is 17 uops and throughput is 4, VGATHERDPS is 24 uops, + throughput 5. Approx 7 uops do not depend on vector size and every load + is 5 uops. */ + 14, 10, /* Gather load static, per_elt. */ + 14, 20, /* Gather store static, per_elt. */ + 32, /* size of l1 cache. */ + 1024, /* size of l2 cache. */ + 64, /* size of prefetch block. */ + /* New AMD processors never drop prefetches; if they cannot be performed + immediately, they are queued. We set number of simultaneous prefetches + to a large constant to reflect this (it probably is not a good idea not + to limit number of prefetches at all, as their execution also takes some + time). */ + 100, /* number of parallel prefetches. */ + 3, /* Branch cost. */ + COSTS_N_INSNS (7), /* cost of FADD and FSUB insns. */ + COSTS_N_INSNS (7), /* cost of FMUL instruction. */ + /* Latency of fdiv is 8-15. */ + COSTS_N_INSNS (15), /* cost of FDIV instruction. */ + COSTS_N_INSNS (1), /* cost of FABS instruction. */ + COSTS_N_INSNS (1), /* cost of FCHS instruction. */ + /* Latency of fsqrt is 4-10. */ + COSTS_N_INSNS (25), /* cost of FSQRT instruction. */ + + COSTS_N_INSNS (1), /* cost of cheap SSE instruction. */ + COSTS_N_INSNS (3), /* cost of ADDSS/SD SUBSS/SD insns. */ + COSTS_N_INSNS (3), /* cost of MULSS instruction. */ + COSTS_N_INSNS (3), /* cost of MULSD instruction. */ + COSTS_N_INSNS (4), /* cost of FMA SS instruction. */ + COSTS_N_INSNS (4), /* cost of FMA SD instruction. */ + COSTS_N_INSNS (10), /* cost of DIVSS instruction. */ + /* 9-13. */ + COSTS_N_INSNS (13), /* cost of DIVSD instruction. */ + COSTS_N_INSNS (14), /* cost of SQRTSS instruction. */ + COSTS_N_INSNS (20), /* cost of SQRTSD instruction. */ + /* Zen can execute 4 integer operations per cycle. FP operations + take 3 cycles and it can execute 2 integer additions and 2 + multiplications thus reassociation may make sense up to with of 6. + SPEC2k6 bencharks suggests + that 4 works better than 6 probably due to register pressure. + + Integer vector operations are taken by FP unit and execute 3 vector + plus/minus operations per cycle but only one multiply. This is adjusted + in ix86_reassociation_width. */ + 4, 4, 3, 6, /* reassoc int, fp, vec_int, vec_fp. */ + znver2_memcpy, + znver2_memset, + COSTS_N_INSNS (4), /* cond_taken_branch_cost. */ + COSTS_N_INSNS (2), /* cond_not_taken_branch_cost. */ + "16", /* Loop alignment. */ + "16", /* Jump alignment. */ + "0:0:8", /* Label alignment. */ + "16", /* Func alignment. */ +}; + /* skylake_cost should produce code tuned for Skylake familly of CPUs. */ static stringop_algs skylake_memcpy[2] = { {libcall, diff --git a/gcc/config/i386/x86-tune-sched.cc b/gcc/config/i386/x86-tune-sched.cc index 2ead7ac557b..ebfde596249 100644 --- a/gcc/config/i386/x86-tune-sched.cc +++ b/gcc/config/i386/x86-tune-sched.cc @@ -68,6 +68,7 @@ ix86_issue_rate (void) case PROCESSOR_ZNVER2: case PROCESSOR_ZNVER3: case PROCESSOR_ZNVER4: + case PROCESSOR_ZNVER5: case PROCESSOR_CORE2: case PROCESSOR_NEHALEM: case PROCESSOR_SANDYBRIDGE: @@ -401,6 +402,7 @@ ix86_adjust_cost (rtx_insn *insn, int dep_type, rtx_insn *dep_insn, int cost, case PROCESSOR_ZNVER2: case PROCESSOR_ZNVER3: case PROCESSOR_ZNVER4: + case PROCESSOR_ZNVER5: /* Stack engine allows to execute push&pop instructions in parall. */ if ((insn_type == TYPE_PUSH || insn_type == TYPE_POP) && (dep_insn_type == TYPE_PUSH || dep_insn_type == TYPE_POP)) diff --git a/gcc/config/i386/x86-tune.def b/gcc/config/i386/x86-tune.def index cc6c69e7d4b..f5bf331242a 100644 --- a/gcc/config/i386/x86-tune.def +++ b/gcc/config/i386/x86-tune.def @@ -558,12 +558,12 @@ DEF_TUNE (X86_TUNE_AVX256_STORE_BY_PIECES, "avx256_store_by_pieces", /* X86_TUNE_AVX512_MOVE_BY_PIECES: Optimize move_by_pieces with 512-bit AVX instructions. */ DEF_TUNE (X86_TUNE_AVX512_MOVE_BY_PIECES, "avx512_move_by_pieces", - m_SAPPHIRERAPIDS | m_ZNVER4) + m_SAPPHIRERAPIDS | m_ZNVER4 | m_ZNVER5) /* X86_TUNE_AVX512_STORE_BY_PIECES: Optimize store_by_pieces with 512-bit AVX instructions. */ DEF_TUNE (X86_TUNE_AVX512_STORE_BY_PIECES, "avx512_store_by_pieces", - m_SAPPHIRERAPIDS | m_ZNVER4) + m_SAPPHIRERAPIDS | m_ZNVER4 | m_ZNVER5) /*****************************************************************************/ /*****************************************************************************/ diff --git a/gcc/config/i386/znver4.md b/gcc/config/i386/zn4zn5.md similarity index 55% rename from gcc/config/i386/znver4.md rename to gcc/config/i386/zn4zn5.md index d0b239822a8..ba9cfbb5dfc 100644 --- a/gcc/config/i386/znver4.md +++ b/gcc/config/i386/zn4zn5.md @@ -1,4 +1,4 @@ -;; Copyright (C) 2012-2022 Free Software Foundation, Inc. +;; Copyright (C) 2012-2024 Free Software Foundation, Inc. ;; ;; This file is part of GCC. ;; @@ -21,7 +21,7 @@ (define_attr "znver4_decode" "direct,vector,double" (const_string "direct")) -;; AMD znver4 Scheduling +;; AMD znver4 and znver5 Scheduling ;; Modeling automatons for zen decoders, integer execution pipes, ;; AGU pipes, branch, floating point execution and fp store units. (define_automaton "znver4, znver4_ieu, znver4_idiv, znver4_fdiv, znver4_agu, znver4_fpu, znver4_fp_store") @@ -44,32 +44,44 @@ (define_reservation "znver4-double" "znver4-direct") -;; Integer unit 4 ALU pipes. +;; Integer unit 4 ALU pipes in znver4 6 ALU pipes in znver5. (define_cpu_unit "znver4-ieu0" "znver4_ieu") (define_cpu_unit "znver4-ieu1" "znver4_ieu") (define_cpu_unit "znver4-ieu2" "znver4_ieu") (define_cpu_unit "znver4-ieu3" "znver4_ieu") +(define_cpu_unit "znver5-ieu4" "znver4_ieu") +(define_cpu_unit "znver5-ieu5" "znver4_ieu") + ;; Znver4 has an additional branch unit. (define_cpu_unit "znver4-bru0" "znver4_ieu") + (define_reservation "znver4-ieu" "znver4-ieu0|znver4-ieu1|znver4-ieu2|znver4-ieu3") +(define_reservation "znver5-ieu" "znver4-ieu0|znver4-ieu1|znver4-ieu2|znver4-ieu3|znver5-ieu4|znver5-ieu5") -;; 3 AGU pipes in znver4 +;; 3 AGU pipes in znver4 and 4 AGU pipes in znver5 (define_cpu_unit "znver4-agu0" "znver4_agu") (define_cpu_unit "znver4-agu1" "znver4_agu") (define_cpu_unit "znver4-agu2" "znver4_agu") +(define_cpu_unit "znver5-agu3" "znver4_agu") + (define_reservation "znver4-agu-reserve" "znver4-agu0|znver4-agu1|znver4-agu2") +(define_reservation "znver5-agu-reserve" "znver4-agu0|znver4-agu1|znver4-agu2|znver5-agu3") ;; Load is 4 cycles. We do not model reservation of load unit. (define_reservation "znver4-load" "znver4-agu-reserve") (define_reservation "znver4-store" "znver4-agu-reserve") +(define_reservation "znver5-load" "znver5-agu-reserve") +(define_reservation "znver5-store" "znver5-agu-reserve") ;; vectorpath (microcoded) instructions are single issue instructions. ;; So, they occupy all the integer units. +;; This is used for both Znver4 and Znver5, since reserving extra units not used otherwise +;; is harmless. (define_reservation "znver4-ivector" "znver4-ieu0+znver4-ieu1 - +znver4-ieu2+znver4-ieu3+znver4-bru0 - +znver4-agu0+znver4-agu1+znver4-agu2") + +znver4-ieu2+znver4-ieu3+znver5-ieu4+znver5-ieu5+znver4-bru0 + +znver4-agu0+znver4-agu1+znver4-agu2+znver5-agu3") -;; Floating point unit 4 FP pipes. +;; Floating point unit 4 FP pipes in znver4 and znver5. (define_cpu_unit "znver4-fpu0" "znver4_fpu") (define_cpu_unit "znver4-fpu1" "znver4_fpu") (define_cpu_unit "znver4-fpu2" "znver4_fpu") @@ -77,10 +89,6 @@ (define_reservation "znver4-fpu" "znver4-fpu0|znver4-fpu1|znver4-fpu2|znver4-fpu3") -(define_reservation "znver4-fvector" "znver4-fpu0+znver4-fpu1 - +znver4-fpu2+znver4-fpu3 - +znver4-agu0+znver4-agu1+znver4-agu2") - ;; DIV units (define_cpu_unit "znver4-idiv" "znver4_idiv") (define_cpu_unit "znver4-fdiv" "znver4_fdiv") @@ -89,6 +97,19 @@ ;; throughput is limited to only one per cycle. (define_cpu_unit "znver4-fp-store" "znver4_fp_store") +;; Floating point store unit 2 FP pipes in znver5. +(define_cpu_unit "znver5-fp-store0" "znver4_fp_store") +(define_cpu_unit "znver5-fp-store1" "znver4_fp_store") + +;; This is used for both Znver4 and Znver5, since reserving extra units not used otherwise +;; is harmless. +(define_reservation "znver4-fvector" "znver4-fpu0+znver4-fpu1 + +znver4-fpu2+znver4-fpu3+znver5-fp-store0+znver5-fp-store1 + +znver4-agu0+znver4-agu1+znver4-agu2+znver5-agu3") + +(define_reservation "znver5-fp-store256" "znver5-fp-store0|znver5-fp-store1") +(define_reservation "znver5-fp-store-512" "znver5-fp-store0+znver5-fp-store1") + ;; Integer Instructions ;; Move instructions @@ -100,6 +121,13 @@ (eq_attr "memory" "none")))) "znver4-double,znver4-ieu") +(define_insn_reservation "znver5_imov_double" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "znver1_decode" "double") + (and (eq_attr "type" "imov") + (eq_attr "memory" "none")))) + "znver4-double,znver5-ieu") + (define_insn_reservation "znver4_imov_double_load" 5 (and (eq_attr "cpu" "znver4") (and (eq_attr "znver1_decode" "double") @@ -107,6 +135,13 @@ (eq_attr "memory" "load")))) "znver4-double,znver4-load,znver4-ieu") +(define_insn_reservation "znver5_imov_double_load" 5 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "znver1_decode" "double") + (and (eq_attr "type" "imov") + (eq_attr "memory" "load")))) + "znver4-double,znver5-load,znver5-ieu") + ;; imov, imovx (define_insn_reservation "znver4_imov" 1 (and (eq_attr "cpu" "znver4") @@ -114,12 +149,24 @@ (eq_attr "memory" "none"))) "znver4-direct,znver4-ieu") +(define_insn_reservation "znver5_imov" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "imov,imovx") + (eq_attr "memory" "none"))) + "znver4-direct,znver5-ieu") + (define_insn_reservation "znver4_imov_load" 5 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "imov,imovx") (eq_attr "memory" "load"))) "znver4-direct,znver4-load,znver4-ieu") +(define_insn_reservation "znver5_imov_load" 5 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "imov,imovx") + (eq_attr "memory" "load"))) + "znver4-direct,znver5-load,znver5-ieu") + ;; Push Instruction (define_insn_reservation "znver4_push" 1 (and (eq_attr "cpu" "znver4") @@ -127,12 +174,24 @@ (eq_attr "memory" "store"))) "znver4-direct,znver4-store") +(define_insn_reservation "znver5_push" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "push") + (eq_attr "memory" "store"))) + "znver4-direct,znver5-store") + (define_insn_reservation "znver4_push_mem" 5 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "push") (eq_attr "memory" "both"))) "znver4-direct,znver4-load,znver4-store") +(define_insn_reservation "znver5_push_mem" 5 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "push") + (eq_attr "memory" "both"))) + "znver4-direct,znver5-load,znver5-store") + ;; Pop instruction (define_insn_reservation "znver4_pop" 4 (and (eq_attr "cpu" "znver4") @@ -140,16 +199,28 @@ (eq_attr "memory" "load"))) "znver4-direct,znver4-load") +(define_insn_reservation "znver5_pop" 4 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "pop") + (eq_attr "memory" "load"))) + "znver4-direct,znver5-load") + (define_insn_reservation "znver4_pop_mem" 5 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "pop") (eq_attr "memory" "both"))) "znver4-direct,znver4-load,znver4-store") +(define_insn_reservation "znver5_pop_mem" 5 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "pop") + (eq_attr "memory" "both"))) + "znver4-direct,znver5-load,znver5-store") + ;; Integer Instructions or General instructions ;; Multiplications (define_insn_reservation "znver4_imul" 3 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "imul") (eq_attr "memory" "none"))) "znver4-direct,znver4-ieu1") @@ -160,30 +231,36 @@ (eq_attr "memory" "load"))) "znver4-direct,znver4-load,znver4-ieu1") +(define_insn_reservation "znver5_imul_load" 7 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "imul") + (eq_attr "memory" "load"))) + "znver4-direct,znver5-load,znver4-ieu1") + ;; Divisions (define_insn_reservation "znver4_idiv_DI" 18 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "idiv") (and (eq_attr "mode" "DI") (eq_attr "memory" "none")))) "znver4-double,znver4-idiv*10") (define_insn_reservation "znver4_idiv_SI" 12 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "idiv") (and (eq_attr "mode" "SI") (eq_attr "memory" "none")))) "znver4-double,znver4-idiv*6") (define_insn_reservation "znver4_idiv_HI" 10 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "idiv") (and (eq_attr "mode" "HI") (eq_attr "memory" "none")))) "znver4-double,znver4-idiv*4") (define_insn_reservation "znver4_idiv_QI" 9 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "idiv") (and (eq_attr "mode" "QI") (eq_attr "memory" "none")))) @@ -196,6 +273,13 @@ (eq_attr "memory" "load")))) "znver4-double,znver4-load,znver4-idiv*10") +(define_insn_reservation "znver5_idiv_DI_load" 22 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "idiv") + (and (eq_attr "mode" "DI") + (eq_attr "memory" "load")))) + "znver4-double,znver5-load,znver4-idiv*10") + (define_insn_reservation "znver4_idiv_SI_load" 16 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "idiv") @@ -203,6 +287,13 @@ (eq_attr "memory" "load")))) "znver4-double,znver4-load,znver4-idiv*6") +(define_insn_reservation "znver5_idiv_SI_load" 16 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "idiv") + (and (eq_attr "mode" "SI") + (eq_attr "memory" "load")))) + "znver4-double,znver5-load,znver4-idiv*6") + (define_insn_reservation "znver4_idiv_HI_load" 14 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "idiv") @@ -210,6 +301,13 @@ (eq_attr "memory" "load")))) "znver4-double,znver4-load,znver4-idiv*4") +(define_insn_reservation "znver5_idiv_HI_load" 14 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "idiv") + (and (eq_attr "mode" "HI") + (eq_attr "memory" "load")))) + "znver4-double,znver5-load,znver4-idiv*4") + (define_insn_reservation "znver4_idiv_QI_load" 13 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "idiv") @@ -217,6 +315,13 @@ (eq_attr "memory" "load")))) "znver4-double,znver4-load,znver4-idiv*4") +(define_insn_reservation "znver5_idiv_QI_load" 13 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "idiv") + (and (eq_attr "mode" "QI") + (eq_attr "memory" "load")))) + "znver4-double,znver5-load,znver4-idiv*4") + ;; INTEGER/GENERAL Instructions (define_insn_reservation "znver4_insn" 1 (and (eq_attr "cpu" "znver4") @@ -224,14 +329,26 @@ (eq_attr "memory" "none,unknown"))) "znver4-direct,znver4-ieu") +(define_insn_reservation "znver5_insn" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "alu,alu1,negnot,rotate1,ishift1,test,incdec,icmp") + (eq_attr "memory" "none,unknown"))) + "znver4-direct,znver5-ieu") + (define_insn_reservation "znver4_insn_load" 5 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "alu,alu1,negnot,rotate1,ishift1,test,incdec,icmp") (eq_attr "memory" "load"))) "znver4-direct,znver4-load,znver4-ieu") +(define_insn_reservation "znver5_insn_load" 5 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "alu,alu1,negnot,rotate1,ishift1,test,incdec,icmp") + (eq_attr "memory" "load"))) + "znver4-direct,znver5-load,znver5-ieu") + (define_insn_reservation "znver4_insn2" 1 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "icmov,setcc") (eq_attr "memory" "none,unknown"))) "znver4-direct,znver4-ieu0|znver4-ieu3") @@ -242,8 +359,14 @@ (eq_attr "memory" "load"))) "znver4-direct,znver4-load,znver4-ieu0|znver4-ieu3") +(define_insn_reservation "znver5_insn2_load" 5 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "icmov,setcc") + (eq_attr "memory" "load"))) + "znver4-direct,znver5-load,znver4-ieu0|znver4-ieu3") + (define_insn_reservation "znver4_rotate" 1 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "rotate") (eq_attr "memory" "none,unknown"))) "znver4-direct,znver4-ieu1|znver4-ieu2") @@ -254,27 +377,51 @@ (eq_attr "memory" "load"))) "znver4-direct,znver4-load,znver4-ieu1|znver4-ieu2") +(define_insn_reservation "znver5_rotate_load" 5 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "rotate") + (eq_attr "memory" "load"))) + "znver4-direct,znver5-load,znver4-ieu1|znver4-ieu2") + (define_insn_reservation "znver4_insn_store" 1 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "alu,alu1,negnot,rotate1,ishift1,test,incdec,icmp") (eq_attr "memory" "store"))) "znver4-direct,znver4-ieu,znver4-store") +(define_insn_reservation "znver5_insn_store" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "alu,alu1,negnot,rotate1,ishift1,test,incdec,icmp") + (eq_attr "memory" "store"))) + "znver4-direct,znver4-ieu,znver5-store") + (define_insn_reservation "znver4_insn2_store" 1 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "icmov,setcc") (eq_attr "memory" "store"))) "znver4-direct,znver4-ieu0|znver4-ieu3,znver4-store") +(define_insn_reservation "znver5_insn2_store" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "icmov,setcc") + (eq_attr "memory" "store"))) + "znver4-direct,znver4-ieu0|znver4-ieu3,znver5-store") + (define_insn_reservation "znver4_rotate_store" 1 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "rotate") (eq_attr "memory" "store"))) "znver4-direct,znver4-ieu1|znver4-ieu2,znver4-store") +(define_insn_reservation "znver5_rotate_store" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "rotate") + (eq_attr "memory" "store"))) + "znver4-direct,znver4-ieu1|znver4-ieu2,znver5-store") + ;; alu1 instructions (define_insn_reservation "znver4_alu1_vector" 3 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "znver1_decode" "vector") (and (eq_attr "type" "alu1") (eq_attr "memory" "none,unknown")))) @@ -287,15 +434,27 @@ (eq_attr "memory" "load")))) "znver4-vector,znver4-load,znver4-ivector*3") +(define_insn_reservation "znver5_alu1_vector_load" 7 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "znver1_decode" "vector") + (and (eq_attr "type" "alu1") + (eq_attr "memory" "load")))) + "znver4-vector,znver5-load,znver4-ivector*3") + ;; Call Instruction (define_insn_reservation "znver4_call" 1 (and (eq_attr "cpu" "znver4") (eq_attr "type" "call,callv")) "znver4-double,znver4-ieu0|znver4-bru0,znver4-store") +(define_insn_reservation "znver5_call" 1 + (and (eq_attr "cpu" "znver5") + (eq_attr "type" "call,callv")) + "znver4-double,znver4-ieu0|znver4-bru0,znver5-store") + ;; Branches (define_insn_reservation "znver4_branch" 1 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "ibr") (eq_attr "memory" "none"))) "znver4-direct,znver4-ieu0|znver4-bru0") @@ -306,8 +465,14 @@ (eq_attr "memory" "load"))) "znver4-direct,znver4-load,znver4-ieu0|znver4-bru0") +(define_insn_reservation "znver5_branch_load" 5 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ibr") + (eq_attr "memory" "load"))) + "znver4-direct,znver5-load,znver4-ieu0|znver4-bru0") + (define_insn_reservation "znver4_branch_vector" 2 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "ibr") (eq_attr "memory" "none,unknown"))) "znver4-vector,znver4-ivector*2") @@ -318,21 +483,36 @@ (eq_attr "memory" "load"))) "znver4-vector,znver4-load,znver4-ivector*2") +(define_insn_reservation "znver5_branch_vector_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ibr") + (eq_attr "memory" "load"))) + "znver4-vector,znver5-load,znver4-ivector*2") + ;; LEA instruction with simple addressing (define_insn_reservation "znver4_lea" 1 (and (eq_attr "cpu" "znver4") (eq_attr "type" "lea")) "znver4-direct,znver4-ieu") +(define_insn_reservation "znver5_lea" 1 + (and (eq_attr "cpu" "znver5") + (eq_attr "type" "lea")) + "znver4-direct,znver5-ieu") ;; Leave (define_insn_reservation "znver4_leave" 1 (and (eq_attr "cpu" "znver4") (eq_attr "type" "leave")) "znver4-double,znver4-ieu,znver4-store") +(define_insn_reservation "znver5_leave" 1 + (and (eq_attr "cpu" "znver5") + (eq_attr "type" "leave")) + "znver4-double,znver5-ieu,znver5-store") + ;; STR and ISHIFT are microcoded. (define_insn_reservation "znver4_str" 3 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "str") (eq_attr "memory" "none"))) "znver4-vector,znver4-ivector*3") @@ -343,8 +523,14 @@ (eq_attr "memory" "load"))) "znver4-vector,znver4-load,znver4-ivector*3") +(define_insn_reservation "znver5_str_load" 7 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "str") + (eq_attr "memory" "load"))) + "znver4-vector,znver5-load,znver4-ivector*3") + (define_insn_reservation "znver4_ishift" 2 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "ishift") (eq_attr "memory" "none"))) "znver4-vector,znver4-ivector*2") @@ -355,9 +541,15 @@ (eq_attr "memory" "load"))) "znver4-vector,znver4-load,znver4-ivector*2") +(define_insn_reservation "znver5_ishift_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ishift") + (eq_attr "memory" "load"))) + "znver4-vector,znver5-load,znver4-ivector*2") + ;; Other vector type (define_insn_reservation "znver4_ieu_vector" 5 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "other,multi") (eq_attr "memory" "none,unknown"))) "znver4-vector,znver4-ivector*5") @@ -368,15 +560,21 @@ (eq_attr "memory" "load"))) "znver4-vector,znver4-load,znver4-ivector*5") +(define_insn_reservation "znver5_ieu_vector_load" 9 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "other,multi") + (eq_attr "memory" "load"))) + "znver4-vector,znver5-load,znver4-ivector*5") + ;; Floating Point ;; FP movs (define_insn_reservation "znver4_fp_cmov" 4 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (eq_attr "type" "fcmov")) "znver4-vector,znver4-fvector*3") (define_insn_reservation "znver4_fp_mov_direct" 1 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (eq_attr "type" "fmov")) "znver4-direct,znver4-fpu0|znver4-fpu1") @@ -388,6 +586,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu0|znver4-fpu1") +(define_insn_reservation "znver5_fp_mov_direct_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "znver1_decode" "direct") + (and (eq_attr "type" "fmov") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu0|znver4-fpu1") + ;;FST (define_insn_reservation "znver4_fp_mov_direct_store" 6 (and (eq_attr "cpu" "znver4") @@ -396,6 +601,13 @@ (eq_attr "memory" "store")))) "znver4-direct,znver4-fpu0|znver4-fpu1,znver4-fp-store") +(define_insn_reservation "znver5_fp_mov_direct_store" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "znver1_decode" "direct") + (and (eq_attr "type" "fmov") + (eq_attr "memory" "store")))) + "znver4-direct,znver4-fpu0|znver4-fpu1,znver5-fp-store256") + ;;FILD (define_insn_reservation "znver4_fp_mov_double_load" 13 (and (eq_attr "cpu" "znver4") @@ -404,6 +616,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu1") +(define_insn_reservation "znver5_fp_mov_double_load" 13 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "znver1_decode" "double") + (and (eq_attr "type" "fmov") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu1") + ;;FIST (define_insn_reservation "znver4_fp_mov_double_store" 7 (and (eq_attr "cpu" "znver4") @@ -412,9 +631,16 @@ (eq_attr "memory" "store")))) "znver4-double,znver4-fpu1,znver4-fp-store") +(define_insn_reservation "znver5_fp_mov_double_store" 7 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "znver1_decode" "double") + (and (eq_attr "type" "fmov") + (eq_attr "memory" "store")))) + "znver4-double,znver4-fpu1,znver5-fp-store256") + ;; FSQRT (define_insn_reservation "znver4_fsqrt" 22 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "fpspc") (and (eq_attr "mode" "XF") (eq_attr "memory" "none")))) @@ -422,20 +648,20 @@ ;; FPSPC instructions (define_insn_reservation "znver4_fp_spc" 6 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "fpspc") (eq_attr "memory" "none"))) "znver4-vector,znver4-fvector*6") (define_insn_reservation "znver4_fp_insn_vector" 6 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "znver1_decode" "vector") (eq_attr "type" "mmxcvt,sselog1,ssemov"))) "znver4-vector,znver4-fvector*6") ;; FADD, FSUB, FMUL (define_insn_reservation "znver4_fp_op_mul" 7 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "fop,fmul") (eq_attr "memory" "none"))) "znver4-direct,znver4-fpu0") @@ -446,9 +672,14 @@ (eq_attr "memory" "load"))) "znver4-direct,znver4-load,znver4-fpu0") +(define_insn_reservation "znver5_fp_op_mul_load" 12 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "fop,fmul") + (eq_attr "memory" "load"))) + "znver4-direct,znver5-load,znver4-fpu0") ;; FDIV (define_insn_reservation "znver4_fp_div" 15 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "fdiv") (eq_attr "memory" "none"))) "znver4-direct,znver4-fdiv*6") @@ -459,6 +690,12 @@ (eq_attr "memory" "load"))) "znver4-direct,znver4-load,znver4-fdiv*6") +(define_insn_reservation "znver5_fp_div_load" 20 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "fdiv") + (eq_attr "memory" "load"))) + "znver4-direct,znver5-load,znver4-fdiv*6") + (define_insn_reservation "znver4_fp_idiv_load" 24 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "fdiv") @@ -466,15 +703,27 @@ (eq_attr "memory" "load")))) "znver4-double,znver4-load,znver4-fdiv*6") +(define_insn_reservation "znver5_fp_idiv_load" 24 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "fdiv") + (and (eq_attr "fp_int_src" "true") + (eq_attr "memory" "load")))) + "znver4-double,znver5-load,znver4-fdiv*6") + ;; FABS, FCHS (define_insn_reservation "znver4_fp_fsgn" 1 (and (eq_attr "cpu" "znver4") (eq_attr "type" "fsgn")) "znver4-direct,znver4-fpu0|znver4-fpu1") +(define_insn_reservation "znver5_fp_fsgn" 1 + (and (eq_attr "cpu" "znver5") + (eq_attr "type" "fsgn")) + "znver4-direct,znver4-fpu1|znver4-fpu2") + ;; FCMP (define_insn_reservation "znver4_fp_fcmp" 3 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "fcmp") (eq_attr "memory" "none"))) "znver4-direct,znver4-fpu1") @@ -486,14 +735,21 @@ (eq_attr "memory" "none")))) "znver4-double,znver4-fpu1,znver4-fpu2") +(define_insn_reservation "znver5_fp_fcmp_double" 4 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "fcmp") + (and (eq_attr "znver1_decode" "double") + (eq_attr "memory" "none")))) + "znver4-double,znver4-fpu1,znver5-fp-store256") + ;; MMX, SSE, SSEn.n instructions (define_insn_reservation "znver4_fp_mmx " 1 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (eq_attr "type" "mmx")) "znver4-direct,znver4-fpu1|znver4-fpu2") (define_insn_reservation "znver4_mmx_add_cmp" 1 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "mmxadd,mmxcmp") (eq_attr "memory" "none"))) "znver4-direct,znver4-fpu") @@ -504,32 +760,62 @@ (eq_attr "memory" "load"))) "znver4-direct,znver4-load,znver4-fpu") +(define_insn_reservation "znver5_mmx_add_cmp_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "mmxadd,mmxcmp") + (eq_attr "memory" "load"))) + "znver4-direct,znver5-load,znver4-fpu") + (define_insn_reservation "znver4_mmx_insn" 1 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "mmxcvt,sseshuf,sseshuf1,mmxshft") (eq_attr "memory" "none"))) "znver4-direct,znver4-fpu1|znver4-fpu2") +(define_insn_reservation "znver5_mmx_insn" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "mmxcvt,sseshuf,sseshuf1,mmxshft") + (eq_attr "memory" "none"))) + "znver4-direct,znver4-fpu0|znver4-fpu1|znver4-fpu2|znver4-fpu3") + (define_insn_reservation "znver4_mmx_insn_load" 6 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "mmxcvt,sseshuf,sseshuf1,mmxshft") (eq_attr "memory" "load"))) "znver4-direct,znver4-load,znver4-fpu1|znver4-fpu2") +(define_insn_reservation "znver5_mmx_insn_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "mmxcvt,sseshuf,sseshuf1,mmxshft") + (eq_attr "memory" "load"))) + "znver4-direct,znver5-load,znver4-fpu0|znver4-fpu1|znver4-fpu2|znver4-fpu3") + (define_insn_reservation "znver4_mmx_mov" 1 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "mmxmov") (eq_attr "memory" "store"))) "znver4-direct,znver4-fp-store") +(define_insn_reservation "znver5_mmx_mov" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "mmxmov") + (eq_attr "memory" "store"))) + "znver4-direct,znver5-fp-store256") + (define_insn_reservation "znver4_mmx_mov_load" 6 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "mmxmov") (eq_attr "memory" "both"))) "znver4-direct,znver4-load,znver4-fp-store") +(define_insn_reservation "znver5_mmx_mov_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "mmxmov") + (eq_attr "memory" "both"))) + "znver4-direct,znver5-load,znver5-fp-store256") + (define_insn_reservation "znver4_mmx_mul" 3 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "mmxmul") (eq_attr "memory" "none"))) "znver4-direct,znver4-fpu0|znver4-fpu3") @@ -540,9 +826,15 @@ (eq_attr "memory" "load"))) "znver4-direct,znver4-load,znver4-fpu0|znver4-fpu3") +(define_insn_reservation "znver5_mmx_mul_load" 8 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "mmxmul") + (eq_attr "memory" "load"))) + "znver4-direct,znver5-load,znver4-fpu0|znver4-fpu3") + ;; AVX instructions (define_insn_reservation "znver4_sse_log" 1 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "sselog") (and (eq_attr "mode" "V4SF,V8SF,V2DF,V4DF,QI,HI,SI,DI,TI,OI") (eq_attr "memory" "none")))) @@ -555,6 +847,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu") +(define_insn_reservation "znver5_sse_log_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sselog") + (and (eq_attr "mode" "V4SF,V8SF,V2DF,V4DF,QI,HI,SI,DI,TI,OI") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu") + (define_insn_reservation "znver4_sse_log1" 1 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sselog1") @@ -562,6 +861,13 @@ (eq_attr "memory" "store")))) "znver4-direct,znver4-fpu1|znver4-fpu2,znver4-fp-store") +(define_insn_reservation "znver5_sse_log1" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sselog1") + (and (eq_attr "mode" "V4SF,V8SF,V2DF,V4DF,QI,HI,SI,DI,TI,OI") + (eq_attr "memory" "store")))) + "znver4-direct,znver4-fpu1|znver4-fpu2,znver5-fp-store256") + (define_insn_reservation "znver4_sse_log1_load" 6 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sselog1") @@ -569,20 +875,39 @@ (eq_attr "memory" "both")))) "znver4-direct,znver4-load,znver4-fpu1|znver4-fpu2,znver4-fp-store") +(define_insn_reservation "znver5_sse_log1_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sselog1") + (and (eq_attr "mode" "V4SF,V8SF,V2DF,V4DF,QI,HI,SI,DI,TI,OI") + (eq_attr "memory" "both")))) + "znver4-direct,znver5-load,znver4-fpu1|znver4-fpu2,znver5-fp-store256") + (define_insn_reservation "znver4_sse_comi" 1 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssecomi") (eq_attr "memory" "store"))) "znver4-double,znver4-fpu2|znver4-fpu3,znver4-fp-store") +(define_insn_reservation "znver5_sse_comi" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssecomi") + (eq_attr "memory" "store"))) + "znver4-double,znver4-fpu2|znver4-fpu3,znver5-fp-store256") + (define_insn_reservation "znver4_sse_comi_load" 6 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssecomi") (eq_attr "memory" "both"))) "znver4-double,znver4-load,znver4-fpu2|znver4-fpu3,znver4-fp-store") +(define_insn_reservation "znver5_sse_comi_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssecomi") + (eq_attr "memory" "both"))) + "znver4-double,znver5-load,znver4-fpu2|znver4-fpu3,znver5-fp-store256") + (define_insn_reservation "znver4_sse_test" 1 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "prefix_extra" "1") (and (eq_attr "type" "ssecomi") (eq_attr "memory" "none")))) @@ -595,8 +920,15 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu1|znver4-fpu2") +(define_insn_reservation "znver5_sse_test_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "prefix_extra" "1") + (and (eq_attr "type" "ssecomi") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu1|znver4-fpu2") + (define_insn_reservation "znver4_sse_imul" 3 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "sseimul") (and (eq_attr "mode" "QI,HI,SI,DI,TI,OI") (eq_attr "memory" "none")))) @@ -609,8 +941,15 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu0|znver4-fpu1") +(define_insn_reservation "znver5_sse_imul_load" 8 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseimul") + (and (eq_attr "mode" "QI,HI,SI,DI,TI,OI") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu0|znver4-fpu1") + (define_insn_reservation "znver4_sse_mov" 1 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "ssemov") (and (eq_attr "mode" "QI,HI,SI,DI,TI,OI") (eq_attr "memory" "none")))) @@ -623,6 +962,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu1|znver4-fpu2") +(define_insn_reservation "znver5_sse_mov_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssemov") + (and (eq_attr "mode" "QI,HI,SI,DI,TI,OI") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu1|znver4-fpu2") + (define_insn_reservation "znver4_sse_mov_store" 1 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssemov") @@ -630,8 +976,15 @@ (eq_attr "memory" "store")))) "znver4-direct,znver4-fpu1|znver4-fpu2,znver4-fp-store") +(define_insn_reservation "znver5_sse_mov_store" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssemov") + (and (eq_attr "mode" "QI,HI,SI,DI,TI,OI") + (eq_attr "memory" "store")))) + "znver4-direct,znver4-fpu1|znver4-fpu2,znver5-fp-store256") + (define_insn_reservation "znver4_sse_mov_fp" 1 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "ssemov") (and (eq_attr "mode" "V16SF,V8DF,V8SF,V4DF,V4SF,V2DF,V2SF,V1DF,SF") (eq_attr "memory" "none")))) @@ -644,6 +997,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu") +(define_insn_reservation "znver5_sse_mov_fp_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssemov") + (and (eq_attr "mode" "V16SF,V8DF,V8SF,V4DF,V4SF,V2DF,V2SF,V1DF,SF") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu") + (define_insn_reservation "znver4_sse_mov_fp_store" 1 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssemov") @@ -651,8 +1011,22 @@ (eq_attr "memory" "store")))) "znver4-direct,znver4-fp-store") +(define_insn_reservation "znver5_sse_mov_fp_store" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssemov") + (and (eq_attr "mode" "V8SF,V4DF,V4SF,V2DF,V2SF,V1DF,SF") + (eq_attr "memory" "store")))) + "znver4-direct,znver5-fp-store256") + +(define_insn_reservation "znver5_sse_mov_fp_store_512" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssemov") + (and (eq_attr "mode" "V16SF,V8DF") + (eq_attr "memory" "store")))) + "znver4-direct,znver5-fp-store-512") + (define_insn_reservation "znver4_sse_add" 3 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "sseadd") (and (eq_attr "mode" "V8SF,V4DF,V4SF,V2DF,V2SF,V1DF,SF") (eq_attr "memory" "none")))) @@ -665,8 +1039,15 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu2|znver4-fpu3") +(define_insn_reservation "znver5_sse_add_load" 8 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseadd") + (and (eq_attr "mode" "V8SF,V4DF,V4SF,V2DF,V2SF,V1DF,SF") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu2|znver4-fpu3") + (define_insn_reservation "znver4_sse_add1" 4 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "sseadd1") (and (eq_attr "mode" "V8SF,V4DF,V4SF,V2DF,V2SF,V1DF,SF") (eq_attr "memory" "none")))) @@ -679,8 +1060,15 @@ (eq_attr "memory" "load")))) "znver4-vector,znver4-load,znver4-fvector*2") +(define_insn_reservation "znver5_sse_add1_load" 9 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseadd1") + (and (eq_attr "mode" "V8SF,V4DF,V4SF,V2DF,V2SF,V1DF,SF") + (eq_attr "memory" "load")))) + "znver4-vector,znver5-load,znver4-fvector*2") + (define_insn_reservation "znver4_sse_iadd" 1 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "sseiadd") (and (eq_attr "mode" "QI,HI,SI,DI,TI,OI") (eq_attr "memory" "none")))) @@ -693,8 +1081,15 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu") +(define_insn_reservation "znver5_sse_iadd_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseiadd") + (and (eq_attr "mode" "QI,HI,SI,DI,TI,OI") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu") + (define_insn_reservation "znver4_sse_mul" 3 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "ssemul") (and (eq_attr "mode" "V8SF,V4DF,V4SF,V2DF,V2SF,V1DF,SF") (eq_attr "memory" "none")))) @@ -707,15 +1102,22 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu0|znver4-fpu1") +(define_insn_reservation "znver5_sse_mul_load" 8 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssemul") + (and (eq_attr "mode" "V8SF,V4DF,V4SF,V2DF,V2SF,V1DF,SF") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu0|znver4-fpu1") + (define_insn_reservation "znver4_sse_div_pd" 13 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "ssediv") (and (eq_attr "mode" "V4DF,V2DF,V1DF") (eq_attr "memory" "none")))) "znver4-direct,znver4-fdiv*5") (define_insn_reservation "znver4_sse_div_ps" 10 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "ssediv") (and (eq_attr "mode" "V8SF,V4SF,V2SF,SF") (eq_attr "memory" "none")))) @@ -728,6 +1130,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fdiv*5") +(define_insn_reservation "znver5_sse_div_pd_load" 18 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssediv") + (and (eq_attr "mode" "V4DF,V2DF,V1DF") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fdiv*5") + (define_insn_reservation "znver4_sse_div_ps_load" 15 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssediv") @@ -735,8 +1144,15 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fdiv*3") +(define_insn_reservation "znver5_sse_div_ps_load" 15 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssediv") + (and (eq_attr "mode" "V8SF,V4SF,V2SF,SF") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fdiv*3") + (define_insn_reservation "znver4_sse_cmp_avx" 1 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "ssecmp") (and (eq_attr "prefix" "vex") (eq_attr "memory" "none")))) @@ -749,20 +1165,39 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu0|znver4-fpu1") +(define_insn_reservation "znver5_sse_cmp_avx_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssecmp") + (and (eq_attr "prefix" "vex") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu0|znver4-fpu1") + (define_insn_reservation "znver4_sse_comi_avx" 1 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssecomi") (eq_attr "memory" "store"))) "znver4-direct,znver4-fpu2+znver4-fpu3,znver4-fp-store") +(define_insn_reservation "znver5_sse_comi_avx" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssecomi") + (eq_attr "memory" "store"))) + "znver4-direct,znver4-fpu2+znver4-fpu3,znver5-fp-store256") + (define_insn_reservation "znver4_sse_comi_avx_load" 6 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssecomi") (eq_attr "memory" "both"))) "znver4-direct,znver4-load,znver4-fpu2+znver4-fpu3,znver4-fp-store") +(define_insn_reservation "znver5_sse_comi_avx_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssecomi") + (eq_attr "memory" "both"))) + "znver4-direct,znver5-load,znver4-fpu2+znver4-fpu3,znver5-fp-store256") + (define_insn_reservation "znver4_sse_cvt" 3 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "ssecvt") (and (eq_attr "mode" "V8SF,V4DF,V4SF,V2DF,V2SF,V1DF,SF") (eq_attr "memory" "none")))) @@ -775,8 +1210,15 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu2|znver4-fpu3") +(define_insn_reservation "znver5_sse_cvt_load" 8 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssecvt") + (and (eq_attr "mode" "V8SF,V4DF,V4SF,V2DF,V2SF,V1DF,SF") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu2|znver4-fpu3") + (define_insn_reservation "znver4_sse_icvt" 3 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "ssecvt") (and (eq_attr "mode" "SI") (eq_attr "memory" "none")))) @@ -789,6 +1231,13 @@ (eq_attr "memory" "store")))) "znver4-double,znver4-fpu2|znver4-fpu3,znver4-fp-store") +(define_insn_reservation "znver5_sse_icvt_store" 4 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssecvt") + (and (eq_attr "mode" "SI") + (eq_attr "memory" "store")))) + "znver4-double,znver4-fpu2|znver4-fpu3,znver5-fp-store256") + (define_insn_reservation "znver4_sse_shuf" 1 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sseshuf") @@ -796,6 +1245,13 @@ (eq_attr "memory" "none")))) "znver4-direct,znver4-fpu1|znver4-fpu2") +(define_insn_reservation "znver5_sse_shuf" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseshuf") + (and (eq_attr "mode" "V8SF,V4DF,V4SF,V2DF,V2SF,V1DF,SF") + (eq_attr "memory" "none")))) + "znver4-direct,znver4-fpu1|znver4-fpu2|znver4-fpu3") + (define_insn_reservation "znver4_sse_shuf_load" 6 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sseshuf") @@ -803,8 +1259,15 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu") +(define_insn_reservation "znver5_sse_shuf_load" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseshuf") + (and (eq_attr "mode" "V8SF,V4DF,V4SF,V2DF,V2SF,V1DF,SF") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu") + (define_insn_reservation "znver4_sse_ishuf" 3 - (and (eq_attr "cpu" "znver4") + (and (eq_attr "cpu" "znver4,znver5") (and (eq_attr "type" "sseshuf") (and (eq_attr "mode" "OI") (eq_attr "memory" "none")))) @@ -817,6 +1280,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu1|znver4-fpu2") +(define_insn_reservation "znver5_sse_ishuf_load" 8 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseshuf") + (and (eq_attr "mode" "OI") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu1|znver4-fpu2") + ;; AVX512 instructions (define_insn_reservation "znver4_sse_log_evex" 1 (and (eq_attr "cpu" "znver4") @@ -825,6 +1295,13 @@ (eq_attr "memory" "none")))) "znver4-direct,znver4-fpu0*2|znver4-fpu1*2|znver4-fpu2*2|znver4-fpu3*2") +(define_insn_reservation "znver5_sse_log_evex" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sselog") + (and (eq_attr "mode" "V16SF,V8DF,XI") + (eq_attr "memory" "none")))) + "znver4-direct,znver4-fpu0|znver4-fpu1|znver4-fpu2|znver4-fpu3") + (define_insn_reservation "znver4_sse_log_evex_load" 7 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sselog") @@ -832,6 +1309,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu0*2|znver4-fpu1*2|znver4-fpu2*2|znver4-fpu3*2") +(define_insn_reservation "znver5_sse_log_evex_load" 7 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sselog") + (and (eq_attr "mode" "V16SF,V8DF,XI") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu0|znver4-fpu1|znver4-fpu2|znver4-fpu3") + (define_insn_reservation "znver4_sse_log1_evex" 1 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sselog1") @@ -839,6 +1323,13 @@ (eq_attr "memory" "none")))) "znver4-direct,znver4-fpu1*2|znver4-fpu2*2,znver4-fp-store") +(define_insn_reservation "znver5_sse_log1_evex" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sselog1") + (and (eq_attr "mode" "V16SF,V8DF,XI") + (eq_attr "memory" "none")))) + "znver4-direct,znver4-fpu1|znver4-fpu2,znver5-fp-store-512") + (define_insn_reservation "znver4_sse_log1_evex_load" 7 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sselog1") @@ -846,6 +1337,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu1*2|znver4-fpu2*2,znver4-fp-store") +(define_insn_reservation "znver5_sse_log1_evex_load" 7 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sselog1") + (and (eq_attr "mode" "V16SF,V8DF,XI") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu1|znver4-fpu2,znver5-fp-store-512") + (define_insn_reservation "znver4_sse_mul_evex" 3 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssemul") @@ -853,6 +1351,13 @@ (eq_attr "memory" "none")))) "znver4-direct,znver4-fpu0*2|znver4-fpu1*2") +(define_insn_reservation "znver5_sse_mul_evex" 3 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssemul") + (and (eq_attr "mode" "V16SF,V8DF") + (eq_attr "memory" "none")))) + "znver4-direct,znver4-fpu0|znver4-fpu1") + (define_insn_reservation "znver4_sse_mul_evex_load" 9 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssemul") @@ -860,6 +1365,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu0*2|znver4-fpu1*2") +(define_insn_reservation "znver5_sse_mul_evex_load" 9 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssemul") + (and (eq_attr "mode" "V16SF,V8DF") + (eq_attr "memory" "load")))) + "znver4-direct,znver4-load,znver4-fpu0|znver4-fpu1") + (define_insn_reservation "znver4_sse_imul_evex" 3 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sseimul") @@ -867,6 +1379,13 @@ (eq_attr "memory" "none")))) "znver4-direct,znver4-fpu0*2|znver4-fpu3*2") +(define_insn_reservation "znver5_sse_imul_evex" 3 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseimul") + (and (eq_attr "mode" "XI") + (eq_attr "memory" "none")))) + "znver4-direct,znver4-fpu0|znver4-fpu3") + (define_insn_reservation "znver4_sse_imul_evex_load" 9 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sseimul") @@ -874,6 +1393,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu0*2|znver4-fpu1*2") +(define_insn_reservation "znver5_sse_imul_evex_load" 9 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseimul") + (and (eq_attr "mode" "XI") + (eq_attr "memory" "load")))) + "znver4-direct,znver4-load,znver4-fpu0|znver4-fpu1") + (define_insn_reservation "znver4_sse_mov_evex" 4 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssemov") @@ -881,6 +1407,13 @@ (eq_attr "memory" "none")))) "znver4-direct,znver4-fpu1*2|znver4-fpu2*2") +(define_insn_reservation "znver5_sse_mov_evex" 2 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssemov") + (and (eq_attr "mode" "XI") + (eq_attr "memory" "none")))) + "znver4-direct,znver4-fpu1|znver4-fpu2") + (define_insn_reservation "znver4_sse_mov_evex_load" 10 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssemov") @@ -888,6 +1421,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu1*2|znver4-fpu2*2") +(define_insn_reservation "znver5_sse_mov_evex_load" 8 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssemov") + (and (eq_attr "mode" "XI") + (eq_attr "memory" "load")))) + "znver4-direct,znver4-load,znver4-fpu1|znver4-fpu2") + (define_insn_reservation "znver4_sse_mov_evex_store" 5 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssemov") @@ -895,6 +1435,13 @@ (eq_attr "memory" "store")))) "znver4-direct,znver4-fpu1*2|znver4-fpu2*2,znver4-fp-store") +(define_insn_reservation "znver5_sse_mov_evex_store" 3 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssemov") + (and (eq_attr "mode" "XI") + (eq_attr "memory" "store")))) + "znver4-direct,znver4-fpu1|znver4-fpu2,znver5-fp-store-512") + (define_insn_reservation "znver4_sse_add_evex" 3 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sseadd") @@ -902,6 +1449,13 @@ (eq_attr "memory" "none")))) "znver4-direct,znver4-fpu2*2|znver4-fpu3*2") +(define_insn_reservation "znver5_sse_add_evex" 2 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseadd") + (and (eq_attr "mode" "V16SF,V8DF") + (eq_attr "memory" "none")))) + "znver4-direct,znver4-fpu2|znver4-fpu3") + (define_insn_reservation "znver4_sse_add_evex_load" 9 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sseadd") @@ -909,6 +1463,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu2*2|znver4-fpu3*2") +(define_insn_reservation "znver5_sse_add_evex_load" 8 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseadd") + (and (eq_attr "mode" "V16SF,V8DF") + (eq_attr "memory" "load")))) + "znver4-direct,znver4-load,znver4-fpu2|znver4-fpu3") + (define_insn_reservation "znver4_sse_iadd_evex" 1 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sseiadd") @@ -916,6 +1477,13 @@ (eq_attr "memory" "none")))) "znver4-direct,znver4-fpu0*2|znver4-fpu1*2|znver4-fpu2*2|znver4-fpu3*2") +(define_insn_reservation "znver5_sse_iadd_evex" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseiadd") + (and (eq_attr "mode" "XI") + (eq_attr "memory" "none")))) + "znver4-direct,znver4-fpu0|znver4-fpu1|znver4-fpu2|znver4-fpu3") + (define_insn_reservation "znver4_sse_iadd_evex_load" 7 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sseiadd") @@ -923,6 +1491,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu0*2|znver4-fpu1*2|znver4-fpu2*2|znver4-fpu3*2") +(define_insn_reservation "znver5_sse_iadd_evex_load" 7 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseiadd") + (and (eq_attr "mode" "XI") + (eq_attr "memory" "load")))) + "znver4-direct,znver4-load,znver4-fpu0|znver4-fpu1|znver4-fpu2|znver4-fpu3") + (define_insn_reservation "znver4_sse_div_pd_evex" 13 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssediv") @@ -930,6 +1505,13 @@ (eq_attr "memory" "none")))) "znver4-direct,znver4-fdiv*9") +(define_insn_reservation "znver5_sse_div_pd_evex" 13 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssediv") + (and (eq_attr "mode" "V8DF") + (eq_attr "memory" "none")))) + "znver4-direct,znver4-fdiv*9") + (define_insn_reservation "znver4_sse_div_ps_evex" 10 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssediv") @@ -937,6 +1519,13 @@ (eq_attr "memory" "none")))) "znver4-direct,znver4-fdiv*6") +(define_insn_reservation "znver5_sse_div_ps_evex" 10 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssediv") + (and (eq_attr "mode" "V16SF") + (eq_attr "memory" "none")))) + "znver4-direct,znver4-fdiv*6") + (define_insn_reservation "znver4_sse_div_pd_evex_load" 19 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssediv") @@ -944,6 +1533,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fdiv*9") +(define_insn_reservation "znver5_sse_div_pd_evex_load" 19 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssediv") + (and (eq_attr "mode" "V8DF") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fdiv*9") + (define_insn_reservation "znver4_sse_div_ps_evex_load" 16 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssediv") @@ -951,6 +1547,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fdiv*6") +(define_insn_reservation "znver5_sse_div_ps_evex_load" 16 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssediv") + (and (eq_attr "mode" "V16SF") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fdiv*6") + (define_insn_reservation "znver4_sse_cmp_avx128" 3 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssecmp") @@ -959,6 +1562,14 @@ (eq_attr "memory" "none"))))) "znver4-direct,znver4-fpu0*2|znver4-fpu1*2") +(define_insn_reservation "znver5_sse_cmp_avx128" 3 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssecmp") + (and (eq_attr "mode" "V4SF,V2DF,V2SF,V1DF,SF") + (and (eq_attr "prefix" "evex") + (eq_attr "memory" "none"))))) + "znver4-direct,znver4-fpu1|znver4-fpu2") + (define_insn_reservation "znver4_sse_cmp_avx128_load" 9 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssecmp") @@ -967,6 +1578,14 @@ (eq_attr "memory" "load"))))) "znver4-direct,znver4-load,znver4-fpu0*2|znver4-fpu1*2") +(define_insn_reservation "znver5_sse_cmp_avx128_load" 9 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssecmp") + (and (eq_attr "mode" "V4SF,V2DF,V2SF,V1DF,SF") + (and (eq_attr "prefix" "evex") + (eq_attr "memory" "load"))))) + "znver4-direct,znver5-load,znver4-fpu1|znver4-fpu2") + (define_insn_reservation "znver4_sse_cmp_avx256" 4 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssecmp") @@ -975,6 +1594,14 @@ (eq_attr "memory" "none"))))) "znver4-direct,znver4-fpu0*2|znver4-fpu1*2") +(define_insn_reservation "znver5_sse_cmp_avx256" 4 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssecmp") + (and (eq_attr "mode" "V8SF,V4DF") + (and (eq_attr "prefix" "evex") + (eq_attr "memory" "none"))))) + "znver4-direct,znver4-fpu1|znver4-fpu2") + (define_insn_reservation "znver4_sse_cmp_avx256_load" 10 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssecmp") @@ -983,6 +1610,14 @@ (eq_attr "memory" "load"))))) "znver4-direct,znver4-load,znver4-fpu0*2|znver4-fpu1*2") +(define_insn_reservation "znver5_sse_cmp_avx256_load" 10 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssecmp") + (and (eq_attr "mode" "V8SF,V4DF") + (and (eq_attr "prefix" "evex") + (eq_attr "memory" "load"))))) + "znver4-direct,znver5-load,znver4-fpu1|znver4-fpu2") + (define_insn_reservation "znver4_sse_cmp_avx512" 5 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssecmp") @@ -991,6 +1626,14 @@ (eq_attr "memory" "none"))))) "znver4-direct,znver4-fpu0*2|znver4-fpu1*2") +(define_insn_reservation "znver5_sse_cmp_avx512" 5 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssecmp") + (and (eq_attr "mode" "V16SF,V8DF") + (and (eq_attr "prefix" "evex") + (eq_attr "memory" "none"))))) + "znver4-direct,znver4-fpu1|znver4-fpu2") + (define_insn_reservation "znver4_sse_cmp_avx512_load" 11 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssecmp") @@ -999,6 +1642,14 @@ (eq_attr "memory" "load"))))) "znver4-direct,znver4-load,znver4-fpu0*2|znver4-fpu1*2") +(define_insn_reservation "znver5_sse_cmp_avx512_load" 11 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssecmp") + (and (eq_attr "mode" "V16SF,V8DF") + (and (eq_attr "prefix" "evex") + (eq_attr "memory" "load"))))) + "znver4-direct,znver5-load,znver4-fpu1|znver4-fpu2") + (define_insn_reservation "znver4_sse_cvt_evex" 6 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssecvt") @@ -1006,6 +1657,13 @@ (eq_attr "memory" "none")))) "znver4-direct,znver4-fpu1*2|znver4-fpu2*2,znver4-fpu2*2|znver4-fpu3*2") +(define_insn_reservation "znver5_sse_cvt_evex" 6 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssecvt") + (and (eq_attr "mode" "V16SF,V8DF") + (eq_attr "memory" "none")))) + "znver4-direct,znver4-fpu1|znver4-fpu2,znver4-fpu2|znver4-fpu3") + (define_insn_reservation "znver4_sse_cvt_evex_load" 12 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssecvt") @@ -1013,6 +1671,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu1*2|znver4-fpu2*2,znver4-fpu2*2|znver4-fpu3*2") +(define_insn_reservation "znver5_sse_cvt_evex_load" 12 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssecvt") + (and (eq_attr "mode" "V16SF,V8DF") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu1|znver4-fpu2,znver4-fpu2|znver4-fpu3") + (define_insn_reservation "znver4_sse_shuf_evex" 1 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sseshuf") @@ -1020,6 +1685,13 @@ (eq_attr "memory" "none")))) "znver4-direct,znver4-fpu0*2|znver4-fpu1*2|znver4-fpu2*2|znver4-fpu3*2") +(define_insn_reservation "znver5_sse_shuf_evex" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseshuf") + (and (eq_attr "mode" "V16SF,V8DF") + (eq_attr "memory" "none")))) + "znver4-direct,znver4-fpu0|znver4-fpu1|znver4-fpu2|znver4-fpu3") + (define_insn_reservation "znver4_sse_shuf_evex_load" 7 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sseshuf") @@ -1027,6 +1699,13 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu0*2|znver4-fpu1*2|znver4-fpu2*2|znver4-fpu3*2") +(define_insn_reservation "znver5_sse_shuf_evex_load" 7 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseshuf") + (and (eq_attr "mode" "V16SF,V8DF") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu0|znver4-fpu1|znver4-fpu2|znver4-fpu3") + (define_insn_reservation "znver4_sse_ishuf_evex" 4 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sseshuf") @@ -1034,6 +1713,13 @@ (eq_attr "memory" "none")))) "znver4-direct,znver4-fpu1*2|znver4-fpu2*2") +(define_insn_reservation "znver5_sse_ishuf_evex" 5 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseshuf") + (and (eq_attr "mode" "XI") + (eq_attr "memory" "none")))) + "znver4-direct,znver4-fpu1|znver4-fpu2") + (define_insn_reservation "znver4_sse_ishuf_evex_load" 10 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sseshuf") @@ -1041,18 +1727,37 @@ (eq_attr "memory" "load")))) "znver4-direct,znver4-load,znver4-fpu1*2|znver4-fpu2*2") +(define_insn_reservation "znver5_sse_ishuf_evex_load" 10 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseshuf") + (and (eq_attr "mode" "XI") + (eq_attr "memory" "load")))) + "znver4-direct,znver5-load,znver4-fpu1|znver4-fpu2") + (define_insn_reservation "znver4_sse_muladd" 4 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "ssemuladd") (eq_attr "memory" "none"))) "znver4-direct,znver4-fpu0*2|znver4-fpu1*2") +(define_insn_reservation "znver5_sse_muladd" 4 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "ssemuladd") + (eq_attr "memory" "none"))) + "znver4-direct,znver4-fpu0|znver4-fpu1") + (define_insn_reservation "znver4_sse_muladd_load" 10 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "sseshuf") (eq_attr "memory" "load"))) "znver4-direct,znver4-load,znver4-fpu0*2|znver4-fpu1*2") +(define_insn_reservation "znver5_sse_muladd_load" 10 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "sseshuf") + (eq_attr "memory" "load"))) + "znver4-direct,znver5-load,znver4-fpu1|znver4-fpu2") + ;; AVX512 mask instructions (define_insn_reservation "znver4_sse_mskmov" 2 @@ -1061,8 +1766,20 @@ (eq_attr "memory" "none"))) "znver4-direct,znver4-fpu0*2|znver4-fpu1*2") +(define_insn_reservation "znver5_sse_mskmov" 2 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "mskmov") + (eq_attr "memory" "none"))) + "znver4-direct,znver4-fpu0|znver4-fpu1") + (define_insn_reservation "znver4_sse_msklog" 1 (and (eq_attr "cpu" "znver4") (and (eq_attr "type" "msklog") (eq_attr "memory" "none"))) "znver4-direct,znver4-fpu2*2|znver4-fpu3*2") + +(define_insn_reservation "znver5_sse_msklog" 1 + (and (eq_attr "cpu" "znver5") + (and (eq_attr "type" "msklog") + (eq_attr "memory" "none"))) + "znver4-direct,znver4-fpu0|znver4-fpu3") diff --git a/gcc/doc/extend.texi b/gcc/doc/extend.texi index 0eb9bdcfac5..39fae6c8cd4 100644 --- a/gcc/doc/extend.texi +++ b/gcc/doc/extend.texi @@ -22024,6 +22024,9 @@ AMD Family 19h Zen version 3. @item znver4 AMD Family 19h Zen version 4. + +@item znver5 +AMD Family 1ah Zen version 5. @end table Here is an example: diff --git a/gcc/doc/invoke.texi b/gcc/doc/invoke.texi index 5db66718d10..926b72982e2 100644 --- a/gcc/doc/invoke.texi +++ b/gcc/doc/invoke.texi @@ -31703,6 +31703,16 @@ WBNOINVD, PKU, VPCLMULQDQ, VAES, AVX512F, AVX512DQ, AVX512IFMA, AVX512CD, AVX512BW, AVX512VL, AVX512BF16, AVX512VBMI, AVX512VBMI2, AVX512VNNI, AVX512BITALG, AVX512VPOPCNTDQ, GFNI and 64-bit instruction set extensions.) +@item znver5 +AMD Family 1ah core based CPUs with x86-64 instruction set support. (This +supersets BMI, BMI2, CLWB, F16C, FMA, FSGSBASE, AVX, AVX2, ADCX, RDSEED, +MWAITX, SHA, CLZERO, AES, PCLMUL, CX16, MOVBE, MMX, SSE, SSE2, SSE3, SSE4A, +SSSE3, SSE4.1, SSE4.2, ABM, XSAVEC, XSAVES, CLFLUSHOPT, POPCNT, RDPID, +WBNOINVD, PKU, VPCLMULQDQ, VAES, AVX512F, AVX512DQ, AVX512IFMA, AVX512CD, +AVX512BW, AVX512VL, AVX512BF16, AVX512VBMI, AVX512VBMI2, AVX512VNNI, +AVX512BITALG, AVX512VPOPCNTDQ, GFNI, AVXVNNI, MOVDIRI, MOVDIR64B, +AVX512VP2INTERSECT, PREFETCHI and 64-bit instruction set extensions.) + @item btver1 CPUs based on AMD Family 14h cores with x86-64 instruction set support. (This supersets MMX, SSE, SSE2, SSE3, SSSE3, SSE4A, CX16, ABM and 64-bit diff --git a/gcc/testsuite/g++.target/i386/mv29.C b/gcc/testsuite/g++.target/i386/mv29.C index a8dd8ac4803..ab229534edd 100644 --- a/gcc/testsuite/g++.target/i386/mv29.C +++ b/gcc/testsuite/g++.target/i386/mv29.C @@ -53,6 +53,10 @@ int __attribute__ ((target("arch=znver4"))) foo () { return 10; } +int __attribute__ ((target("arch=znver5"))) foo () { + return 11; +} + int main () { int val = foo (); @@ -77,6 +81,8 @@ int main () assert (val == 9); else if (__builtin_cpu_is ("znver4")) assert (val == 10); + else if (__builtin_cpu_is ("znver5")) + assert (val == 11); else assert (val == 0); diff --git a/gcc/testsuite/gcc.target/i386/funcspec-56.inc b/gcc/testsuite/gcc.target/i386/funcspec-56.inc index f34e7a977d2..c96c5cf72a9 100644 --- a/gcc/testsuite/gcc.target/i386/funcspec-56.inc +++ b/gcc/testsuite/gcc.target/i386/funcspec-56.inc @@ -200,6 +200,7 @@ extern void test_arch_znver1 (void) __attribute__((__target__("arch= extern void test_arch_znver2 (void) __attribute__((__target__("arch=znver2"))); extern void test_arch_znver3 (void) __attribute__((__target__("arch=znver3"))); extern void test_arch_znver4 (void) __attribute__((__target__("arch=znver4"))); +extern void test_arch_znver5 (void) __attribute__((__target__("arch=znver5"))); extern void test_tune_nocona (void) __attribute__((__target__("tune=nocona"))); extern void test_tune_core2 (void) __attribute__((__target__("tune=core2"))); @@ -223,6 +224,7 @@ extern void test_tune_znver1 (void) __attribute__((__target__("tune= extern void test_tune_znver2 (void) __attribute__((__target__("tune=znver2"))); extern void test_tune_znver3 (void) __attribute__((__target__("tune=znver3"))); extern void test_tune_znver4 (void) __attribute__((__target__("tune=znver4"))); +extern void test_tune_znver5 (void) __attribute__((__target__("tune=znver5"))); extern void test_fpmath_sse (void) __attribute__((__target__("sse2,fpmath=sse"))); extern void test_fpmath_387 (void) __attribute__((__target__("sse2,fpmath=387"))); commit c77b1c833e84b62928a729556c502e1311782b2d Author: Richard Biener Date: Tue Jul 16 10:45:27 2024 +0200 Fixup unaligned load/store cost for znver5 Currently unaligned YMM and ZMM load and store costs are cheaper than aligned which causes the vectorizer to purposely mis-align accesses by adding an alignment prologue. It looks like the unaligned costs were simply copied from the bogus znver4 costs. The following makes the unaligned costs equal to the aligned costs like in the fixed znver4 version. * config/i386/x86-tune-costs.h (znver5_cost): Update unaligned load and store cost from the aligned costs. (cherry picked from commit 896393791ee34ffc176c87d232dfee735db3aaab) diff --git a/gcc/config/i386/x86-tune-costs.h b/gcc/config/i386/x86-tune-costs.h index 11a9dd0ff9e..b8e7ab9372e 100644 --- a/gcc/config/i386/x86-tune-costs.h +++ b/gcc/config/i386/x86-tune-costs.h @@ -2028,8 +2028,8 @@ struct processor_costs znver5_cost = { in 32bit, 64bit, 128bit, 256bit and 512bit */ {8, 8, 8, 12, 12}, /* cost of storing SSE register in 32bit, 64bit, 128bit, 256bit and 512bit */ - {6, 6, 6, 6, 6}, /* cost of unaligned loads. */ - {8, 8, 8, 8, 8}, /* cost of unaligned stores. */ + {6, 6, 10, 10, 12}, /* cost of unaligned loads. */ + {8, 8, 8, 12, 12}, /* cost of unaligned stores. */ 2, 2, 2, /* cost of moving XMM,YMM,ZMM register. */ 6, /* cost of moving SSE register to integer. */ commit 3cc85e9261f539071c2afac2de4b0ba0f3ee7244 Author: GCC Administrator Date: Sun Sep 29 00:19:38 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index e7795f0dd6d..e22a687bd1c 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,55 @@ +2024-09-28 Richard Biener + + Backported from master: + 2024-07-16 Richard Biener + + * config/i386/x86-tune-costs.h (znver5_cost): Update unaligned + load and store cost from the aligned costs. + +2024-09-28 Jan Hubicka + + Backported from master: + 2024-03-18 Jan Hubicka + Karthiban Anbazhagan + + * common/config/i386/cpuinfo.h (get_amd_cpu): Recognize znver5. + * common/config/i386/i386-common.cc (processor_names): Add znver5. + (processor_alias_table): Likewise. + * common/config/i386/i386-cpuinfo.h (processor_types): Add new zen + family. + (processor_subtypes): Add znver5. + * config.gcc (x86_64-*-* |...): Likewise. + * config/i386/driver-i386.cc (host_detect_local_cpu): Let + march=native detect znver5 cpu's. + * config/i386/i386-c.cc (ix86_target_macros_internal): Add + znver5. + * config/i386/i386-options.cc (m_ZNVER5): New definition + (processor_cost_table): Add znver5. + * config/i386/i386.cc (ix86_reassociation_width): Likewise. + * config/i386/i386.h (processor_type): Add PROCESSOR_ZNVER5 + (PTA_ZNVER5): New definition. + * config/i386/i386.md (define_attr "cpu"): Add znver5. + (Scheduling descriptions) Add znver5.md. + * config/i386/x86-tune-costs.h (znver5_cost): New definition. + * config/i386/x86-tune-sched.cc (ix86_issue_rate): Add znver5. + (ix86_adjust_cost): Likewise. + * config/i386/x86-tune.def (avx512_move_by_pieces): Add m_ZNVER5. + (avx512_store_by_pieces): Add m_ZNVER5. + * doc/extend.texi: Add znver5. + * doc/invoke.texi: Likewise. + * config/i386/znver4.md: Rename to zn4zn5.md; combine znver4 and znver5 Scheduler. + * config/i386/zn4zn5.md: New file. + +2024-09-28 H.J. Lu + + Backported from master: + 2024-09-25 H.J. Lu + + PR target/116839 + * config/i386/i386.cc (ix86_rewrite_tls_address_1): Make it + static. Return if TLS address is thread register plus an integer + register. + 2024-09-27 Stefan Schulze Frielinghaus Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d1c17555461..3745c09494a 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240928 +20240929 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 77308773b19..4694af8c584 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,20 @@ +2024-09-28 Jan Hubicka + + Backported from master: + 2024-03-18 Jan Hubicka + Karthiban Anbazhagan + + * g++.target/i386/mv29.C: Handle znver5 arch. + * gcc.target/i386/funcspec-56.inc:Likewise. + +2024-09-28 H.J. Lu + + Backported from master: + 2024-09-25 H.J. Lu + + PR target/116839 + * gcc.target/i386/pr116839.c: New file. + 2024-09-27 Stefan Schulze Frielinghaus Backported from master: commit c44494ee5f13fb1a74c05a405b023c0e2664da57 Author: GCC Administrator Date: Mon Sep 30 00:19:50 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 3745c09494a..bd531cb5559 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240929 +20240930 commit be6334fffdf2a7df3b7f92ea933b804664dfc383 Author: Jan Hubicka Date: Tue Sep 3 13:38:33 2024 +0200 Zen5 tuning part 1: avoid FMA chains testing matrix multiplication benchmarks shows that FMA on a critical chain is a perofrmance loss over separate multiply and add. While the latency of 4 is lower than multiply + add (3+2) the problem is that all values needs to be ready before computation starts. While on znver4 AVX512 code fared well with FMA, it was because of the split registers. Znver5 benefits from avoding FMA on all widths. This may be different with the mobile version though. On naive matrix multiplication benchmark the difference is 8% with -O3 only since with -Ofast loop interchange solves the problem differently. It is 30% win, for example, on S323 from TSVC: real_t s323(struct args_t * func_args) { // recurrences // coupled recurrence initialise_arrays(__func__); gettimeofday(&func_args->t1, NULL); for (int nl = 0; nl < iterations/2; nl++) { for (int i = 1; i < LEN_1D; i++) { a[i] = b[i-1] + c[i] * d[i]; b[i] = a[i] + c[i] * e[i]; } dummy(a, b, c, d, e, aa, bb, cc, 0.); } gettimeofday(&func_args->t2, NULL); return calc_checksum(__func__); } gcc/ChangeLog: * config/i386/x86-tune.def (X86_TUNE_AVOID_128FMA_CHAINS): Enable for znver5. (X86_TUNE_AVOID_256FMA_CHAINS): Likewise. (X86_TUNE_AVOID_512FMA_CHAINS): Likewise. (cherry picked from commit d6360b4083695970789fd65b9c515c11a5ce25b4) diff --git a/gcc/config/i386/x86-tune.def b/gcc/config/i386/x86-tune.def index f5bf331242a..249a239de77 100644 --- a/gcc/config/i386/x86-tune.def +++ b/gcc/config/i386/x86-tune.def @@ -499,16 +499,16 @@ DEF_TUNE (X86_TUNE_USE_SCATTER_8PARTS, "use_scatter_8parts", /* X86_TUNE_AVOID_128FMA_CHAINS: Avoid creating loops with tight 128bit or smaller FMA chain. */ -DEF_TUNE (X86_TUNE_AVOID_128FMA_CHAINS, "avoid_fma_chains", m_ZNVER1 | m_ZNVER2 | m_ZNVER3) +DEF_TUNE (X86_TUNE_AVOID_128FMA_CHAINS, "avoid_fma_chains", m_ZNVER) /* X86_TUNE_AVOID_256FMA_CHAINS: Avoid creating loops with tight 256bit or smaller FMA chain. */ DEF_TUNE (X86_TUNE_AVOID_256FMA_CHAINS, "avoid_fma256_chains", m_ZNVER2 | m_ZNVER3 - | m_ALDERLAKE | m_SAPPHIRERAPIDS | m_GENERIC | m_ZNVER4) + | m_ALDERLAKE | m_SAPPHIRERAPIDS | m_GENERIC | m_ZNVER4 | m_ZNVER5) /* X86_TUNE_AVOID_512FMA_CHAINS: Avoid creating loops with tight 512bit or smaller FMA chain. */ -DEF_TUNE (X86_TUNE_AVOID_512FMA_CHAINS, "avoid_fma512_chains", m_NONE) +DEF_TUNE (X86_TUNE_AVOID_512FMA_CHAINS, "avoid_fma512_chains", m_ZNVER5) /* X86_TUNE_V2DF_REDUCTION_PREFER_PHADDPD: Prefer haddpd for v2df vector reduction. */ commit 3a5daf1ecc81a31accc0c77e6d40802c908a092c Author: GCC Administrator Date: Tue Oct 1 00:21:17 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index e22a687bd1c..07592aedc6c 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,13 @@ +2024-09-30 Jan Hubicka + + Backported from master: + 2024-09-03 Jan Hubicka + + * config/i386/x86-tune.def (X86_TUNE_AVOID_128FMA_CHAINS): Enable for + znver5. + (X86_TUNE_AVOID_256FMA_CHAINS): Likewise. + (X86_TUNE_AVOID_512FMA_CHAINS): Likewise. + 2024-09-28 Richard Biener Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index bd531cb5559..76549858b25 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20240930 +20241001 commit 4137d48918c26548e8834f5526a31a255da78032 Author: GCC Administrator Date: Wed Oct 2 00:18:26 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 76549858b25..f464e65124d 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241001 +20241002 commit 8e5bd9b4b38f5b4fbd2a95d8f61168d9eeea97d3 Author: Richard Biener Date: Wed Sep 18 09:52:55 2024 +0200 tree-optimization/116585 - SSA corruption with split_constant_offset split_constant_offset when looking through SSA defs can end up picking SSA leafs that are subject to abnormal coalescing. This can lead to downstream consumers to insert code based on the result (like from dataref analysis) in places that violate constraints for abnormal coalescing. It's best to not expand defs whose operands are subject to abnormal coalescing - and not either do something when a subexpression has operands like that already. PR tree-optimization/116585 * tree-data-ref.cc (split_constant_offset_1): When either operand is subject to abnormal coalescing do no further processing. * gcc.dg/torture/pr116585.c: New testcase. (cherry picked from commit 1d0cb3b5fca69b81e69cfdb4aea0eebc1ac04750) diff --git a/gcc/testsuite/gcc.dg/torture/pr116585.c b/gcc/testsuite/gcc.dg/torture/pr116585.c new file mode 100644 index 00000000000..108c481e104 --- /dev/null +++ b/gcc/testsuite/gcc.dg/torture/pr116585.c @@ -0,0 +1,32 @@ +/* { dg-do compile } */ + +char *s1, *s2; +extern int* my_alloc (int); +extern int _setjmp (); +extern void bar(); +void foo(int s1len, int s2len) +{ + int e; + e = _setjmp (); + { + int l, i; + int *md = my_alloc(((sizeof(int)) * (s1len + 1) * (s2len))); + s1len++; + for (; s1len; l) + for (; s2len; l) + for (; s1len; i) + { + int j = 1; + for (; j < s2len; j++) + { + int cost; + if (s1[1] == s2[1]) + cost = 0; + else + cost = 1; + md[j * s1len ] = ((cost)); + } + } + bar(); + } +} diff --git a/gcc/tree-data-ref.cc b/gcc/tree-data-ref.cc index 706a49f226e..b7bca6a9d06 100644 --- a/gcc/tree-data-ref.cc +++ b/gcc/tree-data-ref.cc @@ -761,6 +761,14 @@ split_constant_offset_1 (tree type, tree op0, enum tree_code code, tree op1, if (INTEGRAL_TYPE_P (type) && TYPE_OVERFLOW_TRAPS (type)) return false; + if (TREE_CODE (op0) == SSA_NAME + && SSA_NAME_OCCURS_IN_ABNORMAL_PHI (op0)) + return false; + if (op1 + && TREE_CODE (op1) == SSA_NAME + && SSA_NAME_OCCURS_IN_ABNORMAL_PHI (op1)) + return false; + switch (code) { case INTEGER_CST: @@ -853,9 +861,6 @@ split_constant_offset_1 (tree type, tree op0, enum tree_code code, tree op1, case SSA_NAME: { - if (SSA_NAME_OCCURS_IN_ABNORMAL_PHI (op0)) - return false; - gimple *def_stmt = SSA_NAME_DEF_STMT (op0); enum tree_code subcode; commit 95435a16dfd929e79a40e25b0ba7019769d68dbb Author: GCC Administrator Date: Thu Oct 3 00:19:24 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 07592aedc6c..d103a1505b5 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,13 @@ +2024-10-02 Richard Biener + + Backported from master: + 2024-09-18 Richard Biener + + PR tree-optimization/116585 + * tree-data-ref.cc (split_constant_offset_1): When either + operand is subject to abnormal coalescing do no further + processing. + 2024-09-30 Jan Hubicka Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index f464e65124d..aab92d459f7 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241002 +20241003 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 4694af8c584..cbf89a282d3 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-10-02 Richard Biener + + Backported from master: + 2024-09-18 Richard Biener + + PR tree-optimization/116585 + * gcc.dg/torture/pr116585.c: New testcase. + 2024-09-28 Jan Hubicka Backported from master: commit a5e109c31aa7ce72d1182f9fb3924fd986e6b80b Author: GCC Administrator Date: Fri Oct 4 00:18:37 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index aab92d459f7..ec7372fbdf8 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241003 +20241004 commit 7a51946e00fa2fb2ffacdd7c8d0a80e056c1f7ff Author: H.J. Lu Date: Fri Oct 4 16:21:15 2024 +0800 x86: Disable stack protector for naked functions Since naked functions should not enable stack protector, define TARGET_STACK_PROTECT_RUNTIME_ENABLED_P to disable stack protector for naked functions. gcc/ PR target/116962 * config/i386/i386.cc (ix86_stack_protect_runtime_enabled_p): New function. (TARGET_STACK_PROTECT_RUNTIME_ENABLED_P): New. gcc/testsuite/ PR target/116962 * gcc.target/i386/pr116962.c: New file. Signed-off-by: H.J. Lu (cherry picked from commit 7d2845da112214f064e7b24531cc67e256b5177e) diff --git a/gcc/config/i386/i386.cc b/gcc/config/i386/i386.cc index 882de00610f..2087f8633eb 100644 --- a/gcc/config/i386/i386.cc +++ b/gcc/config/i386/i386.cc @@ -22644,6 +22644,13 @@ ix86_stack_protect_guard (void) return default_stack_protect_guard (); } +static bool +ix86_stack_protect_runtime_enabled_p (void) +{ + /* Naked functions should not enable stack protector. */ + return !ix86_function_naked (current_function_decl); +} + /* For 32-bit code we can save PIC register setup by using __stack_chk_fail_local hidden function instead of calling __stack_chk_fail directly. 64-bit code doesn't need to setup any PIC @@ -24650,6 +24657,10 @@ ix86_libgcc_floating_mode_supported_p #undef TARGET_STACK_PROTECT_GUARD #define TARGET_STACK_PROTECT_GUARD ix86_stack_protect_guard +#undef TARGET_STACK_PROTECT_RUNTIME_ENABLED_P +#define TARGET_STACK_PROTECT_RUNTIME_ENABLED_P \ + ix86_stack_protect_runtime_enabled_p + #if !TARGET_MACHO #undef TARGET_STACK_PROTECT_FAIL #define TARGET_STACK_PROTECT_FAIL ix86_stack_protect_fail diff --git a/gcc/testsuite/gcc.target/i386/pr116962.c b/gcc/testsuite/gcc.target/i386/pr116962.c new file mode 100644 index 00000000000..ced16eee746 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr116962.c @@ -0,0 +1,10 @@ +/* { dg-do compile { target fstack_protector } } */ +/* { dg-options "-O2 -fstack-protector-all" } */ +/* { dg-final { scan-assembler-not "__stack_chk_fail" } } */ + +__attribute__ ((naked)) +void +foo (void) +{ + asm ("ret"); +} commit bcad4307707f8b451780031c7409cd66f32c5944 Author: GCC Administrator Date: Sat Oct 5 00:19:25 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index d103a1505b5..21af457f1bb 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,13 @@ +2024-10-04 H.J. Lu + + Backported from master: + 2024-10-04 H.J. Lu + + PR target/116962 + * config/i386/i386.cc (ix86_stack_protect_runtime_enabled_p): New + function. + (TARGET_STACK_PROTECT_RUNTIME_ENABLED_P): New. + 2024-10-02 Richard Biener Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index ec7372fbdf8..21c32c43dc9 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241004 +20241005 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index cbf89a282d3..2ecd420c938 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-10-04 H.J. Lu + + Backported from master: + 2024-10-04 H.J. Lu + + PR target/116962 + * gcc.target/i386/pr116962.c: New file. + 2024-10-02 Richard Biener Backported from master: commit 00100084dbabf8acfd8b12e5b5025bd08ee52f75 Author: John David Anglin Date: Sat Oct 5 18:18:31 2024 -0400 hppa: Fix indirect_goto constraint Noticed testing LRA. 2024-10-05 John David Anglin gcc/ChangeLog: * config/pa/pa.md: Fix indirect_got constraint. diff --git a/gcc/config/pa/pa.md b/gcc/config/pa/pa.md index 43241958722..1d97cb6f486 100644 --- a/gcc/config/pa/pa.md +++ b/gcc/config/pa/pa.md @@ -7316,7 +7316,7 @@ }) (define_insn "indirect_goto" - [(unspec [(match_operand 0 "register_operand" "=r")] UNSPEC_GOTO)] + [(unspec [(match_operand 0 "register_operand" "r")] UNSPEC_GOTO)] "GET_MODE (operands[0]) == word_mode" "bv%* %%r0(%0)" [(set_attr "type" "branch") commit d8ec3da7656c88c06fa48335b046c4202a768aa6 Author: GCC Administrator Date: Sun Oct 6 00:20:33 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 21af457f1bb..e66dcf4aa0d 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,7 @@ +2024-10-05 John David Anglin + + * config/pa/pa.md: Fix indirect_got constraint. + 2024-10-04 H.J. Lu Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 21c32c43dc9..c7126e3e1a9 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241005 +20241006 commit c43ec27305c6fa6f3230974c598ad4664d2d258f Author: GCC Administrator Date: Mon Oct 7 00:19:15 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c7126e3e1a9..09c64f0d788 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241006 +20241007 commit c4d2f51741bbb1771219fbeaaf812fa73c36fc0f Author: Jonathan Wakely Date: Fri Jun 28 15:14:15 2024 +0100 libstdc++: Define __glibcxx_assert_fail for non-verbose build [PR115585] When the library is configured with --disable-libstdcxx-verbose the assertions just abort instead of calling __glibcxx_assert_fail, and so I didn't export that function for the non-verbose build. However, that option is documented to not change the library ABI, so we still need to export the symbol from the library. It could be needed by programs compiled against the headers from a verbose build. The non-verbose definition can just call abort so that it doesn't pull in I/O symbols, which are unwanted in a non-verbose build. libstdc++-v3/ChangeLog: PR libstdc++/115585 * src/c++11/assert_fail.cc (__glibcxx_assert_fail): Add definition for non-verbose builds. (cherry picked from commit 52370c839edd04df86d3ff2b71fcdca0c7376a7f) diff --git a/libstdc++-v3/src/c++11/assert_fail.cc b/libstdc++-v3/src/c++11/assert_fail.cc index 540e953da2e..774ffa70118 100644 --- a/libstdc++-v3/src/c++11/assert_fail.cc +++ b/libstdc++-v3/src/c++11/assert_fail.cc @@ -22,10 +22,10 @@ // see the files COPYING3 and COPYING.RUNTIME respectively. If not, see // . -#include // for std::fprintf, stderr #include // for std::abort #ifdef _GLIBCXX_VERBOSE_ASSERT +#include // for std::fprintf, stderr namespace std { [[__noreturn__]] @@ -41,4 +41,12 @@ namespace std abort(); } } +#else +namespace std +{ + [[__noreturn__]] + void + __glibcxx_assert_fail(const char*, int, const char*, const char*) noexcept + { abort(); } +} #endif commit 2ab55da5eba0aa7a92e15d8100d51cc977f9aca4 Author: Jonathan Wakely Date: Tue Sep 10 14:25:41 2024 +0100 libstdc++: std::string move assignment should not use POCCA trait [PR116641] The changes to implement LWG 2579 (r10-327-gdb33efde17932f) made std::string::assign use the propagate_on_container_copy_assignment (POCCA) trait, for consistency with operator=(const basic_string&). However, this also unintentionally affected operator=(basic_string&&) which calls assign(str) to make a deep copy when performing a move is not possible. The fix is for the move assignment operator to call _M_assign(str) instead of assign(str), as this just does the deep copy and doesn't check the POCCA trait first. The bug only affects the unlikely/useless combination of POCCA==true and POCMA==false, but we should fix it for correctness anyway. it should also make move assignment slightly cheaper to compile and execute, because we skip the extra code in assign(const basic_string&). libstdc++-v3/ChangeLog: PR libstdc++/116641 * include/bits/basic_string.h (operator=(basic_string&&)): Call _M_assign instead of assign. * testsuite/21_strings/basic_string/allocator/116641.cc: New test. (cherry picked from commit c07cf418fdde0c192e370a8d76a991cc7215e9c4) diff --git a/libstdc++-v3/include/bits/basic_string.h b/libstdc++-v3/include/bits/basic_string.h index e02b1b97c5c..b6ad19f2ad3 100644 --- a/libstdc++-v3/include/bits/basic_string.h +++ b/libstdc++-v3/include/bits/basic_string.h @@ -911,7 +911,7 @@ _GLIBCXX_BEGIN_NAMESPACE_CXX11 __str._M_data(__str._M_local_buf); } else // Need to do a deep copy - assign(__str); + _M_assign(__str); __str.clear(); return *this; } diff --git a/libstdc++-v3/testsuite/21_strings/basic_string/allocator/116641.cc b/libstdc++-v3/testsuite/21_strings/basic_string/allocator/116641.cc new file mode 100644 index 00000000000..a1a411b87fa --- /dev/null +++ b/libstdc++-v3/testsuite/21_strings/basic_string/allocator/116641.cc @@ -0,0 +1,53 @@ +// { dg-do run { target c++11 } } +// { dg-require-effective-target cxx11_abi } + +// Bug 116641 - std::string move assignment incorrectly depends on POCCA + +#include +#include + +template +struct Alloc +{ + using value_type = T; + using propagate_on_container_swap = std::false_type; + using propagate_on_container_copy_assignment = std::true_type; + using propagate_on_container_move_assignment = std::false_type; + + Alloc(int id) : id(id) { } + + template + Alloc(const Alloc& a) : id(a.id) { } + + T* allocate(unsigned long n) + { return std::allocator().allocate(n); } + + void deallocate(T* p, unsigned long n) + { std::allocator().deallocate(p, n); } + + Alloc& operator=(const Alloc&) { throw; } + + bool operator==(const Alloc& a) const { return id == a.id; } + bool operator!=(const Alloc& a) const { return id != a.id; } + + int id; +}; + +void +test_pr116641() +{ + Alloc a1(1), a2(2); + std::basic_string, Alloc> s1(a1), s2(a2); + + s1 = "allocator should not propagate on move assignment"; + VERIFY( s1.get_allocator() == a1 ); + VERIFY( s2.get_allocator() == a2 ); + s2 = std::move(s1); + VERIFY( s1.get_allocator() == a1 ); + VERIFY( s2.get_allocator() == a2 ); +} + +int main() +{ + test_pr116641(); +} commit f5ffdcfe3b58771b4482d94264cf076df83a2cdc Author: Jonathan Wakely Date: Wed Aug 28 12:38:18 2024 +0100 libstdc++: Fix autoconf check for O_NONBLOCK in I misused the AC_CHECK_DECL macro, assuming that it behaved like AC_CHECK_DECLS and always defined a HAVE_xxx macro if the decl was found. Instead, the [action-if-found] shell commands are needed to defined HAVE_O_NONBLOCK explicitly. libstdc++-v3/ChangeLog: * configure.ac: Fix check for O_NONBLOCK. * config.h.in: Regenerate. * configure: Regenerate. (cherry picked from commit b68561dd7925dfee1836f75d3fa8d33fff5c2498) diff --git a/libstdc++-v3/config.h.in b/libstdc++-v3/config.h.in index 639000d797a..4bf9b428228 100644 --- a/libstdc++-v3/config.h.in +++ b/libstdc++-v3/config.h.in @@ -295,6 +295,9 @@ /* Define if openat is available in . */ #undef HAVE_OPENAT +/* Define if O_NONBLOCK is defined in */ +#undef HAVE_O_NONBLOCK + /* Define if poll is available in . */ #undef HAVE_POLL diff --git a/libstdc++-v3/configure b/libstdc++-v3/configure index ccc23f1b352..3ea275e9ab2 100755 --- a/libstdc++-v3/configure +++ b/libstdc++-v3/configure @@ -77901,6 +77901,8 @@ if test "$ac_cv_have_decl_F_GETFL$ac_cv_have_decl_F_SETFL" = yesyes ; then " if test "x$ac_cv_have_decl_O_NONBLOCK" = xyes; then : +$as_echo "#define HAVE_O_NONBLOCK 1" >>confdefs.h + fi fi diff --git a/libstdc++-v3/configure.ac b/libstdc++-v3/configure.ac index dc0c61973d1..dad313a2376 100644 --- a/libstdc++-v3/configure.ac +++ b/libstdc++-v3/configure.ac @@ -503,7 +503,10 @@ AC_CHECK_HEADERS([fcntl.h sys/ioctl.h sys/socket.h sys/uio.h poll.h netdb.h arpa AC_CHECK_DECL(F_GETFL,,,[#include ]) AC_CHECK_DECL(F_SETFL,,,[#include ]) if test "$ac_cv_have_decl_F_GETFL$ac_cv_have_decl_F_SETFL" = yesyes ; then - AC_CHECK_DECL(O_NONBLOCK,,,[#include ]) + AC_CHECK_DECL(O_NONBLOCK, + AC_DEFINE(HAVE_O_NONBLOCK,1,[Define if O_NONBLOCK is defined in ]), + [], + [#include ]) fi # For Transactional Memory TS commit f6b9603a23e829212bbca95f9f9d592bd8a318cb Author: Jonathan Wakely Date: Fri Jul 5 20:00:04 2024 +0100 libstdc++: Use reserved form of [[__likely__]] in We should not use [[unlikely]] before C++20, so use [[__unlikely__]] instead. libstdc++-v3/ChangeLog: * include/std/variant (_Variant_storage::_M_reset): Use __unlikely__ form of attribute instead of unlikely. (cherry picked from commit 9f1cd51766f251aafe0f1b898892f79855892729) diff --git a/libstdc++-v3/include/std/variant b/libstdc++-v3/include/std/variant index 9abe7b9ed6d..75214c687df 100644 --- a/libstdc++-v3/include/std/variant +++ b/libstdc++-v3/include/std/variant @@ -491,7 +491,7 @@ namespace __variant constexpr void _M_reset() { - if (!_M_valid()) [[unlikely]] + if (!_M_valid()) [[__unlikely__]] return; std::__do_visit([](auto&& __this_mem) mutable commit 99a3f6587cf2fa2d50890f1ee8b28c4b0a23c7bc Author: Kim Gräsman Date: Tue Aug 27 17:11:29 2024 +0100 libstdc++: Fix @file for target-specific opt_random.h A few of these files self-identified as ext/random.tcc, update to use the actual basename. libstdc++-v3/ChangeLog: * config/cpu/aarch64/opt/ext/opt_random.h: Improve doxygen file docs. * config/cpu/i486/opt/ext/opt_random.h: Likewise. (cherry picked from commit c2ad7b2d5247cf2ddee98d7f46274775a3fa1268) diff --git a/libstdc++-v3/config/cpu/aarch64/opt/ext/opt_random.h b/libstdc++-v3/config/cpu/aarch64/opt/ext/opt_random.h index 1d5cf6e6f34..58c0d0347fb 100644 --- a/libstdc++-v3/config/cpu/aarch64/opt/ext/opt_random.h +++ b/libstdc++-v3/config/cpu/aarch64/opt/ext/opt_random.h @@ -22,7 +22,7 @@ // see the files COPYING3 and COPYING.RUNTIME respectively. If not, see // . -/** @file ext/random.tcc +/** @file ext/opt_random.h * This is an internal header file, included by other library headers. * Do not attempt to use it directly. @headername{ext/random} */ diff --git a/libstdc++-v3/config/cpu/i486/opt/ext/opt_random.h b/libstdc++-v3/config/cpu/i486/opt/ext/opt_random.h index bf3bbcce6f7..36889427b0d 100644 --- a/libstdc++-v3/config/cpu/i486/opt/ext/opt_random.h +++ b/libstdc++-v3/config/cpu/i486/opt/ext/opt_random.h @@ -22,7 +22,7 @@ // see the files COPYING3 and COPYING.RUNTIME respectively. If not, see // . -/** @file ext/random.tcc +/** @file ext/opt_random.h * This is an internal header file, included by other library headers. * Do not attempt to use it directly. @headername{ext/random} */ commit 556051a7bf9373dc8a0f607b5d1ae177a2b5afad Author: Kim Gräsman Date: Tue Aug 27 17:08:47 2024 +0100 libstdc++: Fix @headername for bits/cpp_type_traits.h There is no file ext/type_traits, point it to ext/type_traits.h instead. libstdc++-v3/ChangeLog: * include/bits/cpp_type_traits.h: Improve doxygen file docs. (cherry picked from commit f6ed7a61a7c906f8fb7f8059132225c9bc41f3b2) diff --git a/libstdc++-v3/include/bits/cpp_type_traits.h b/libstdc++-v3/include/bits/cpp_type_traits.h index 8f91bbedbed..550820100cc 100644 --- a/libstdc++-v3/include/bits/cpp_type_traits.h +++ b/libstdc++-v3/include/bits/cpp_type_traits.h @@ -24,7 +24,7 @@ /** @file bits/cpp_type_traits.h * This is an internal header file, included by other library headers. - * Do not attempt to use it directly. @headername{ext/type_traits} + * Do not attempt to use it directly. @headername{ext/type_traits.h} */ // Written by Gabriel Dos Reis commit 1f655ef43621cc022745c3aa9c77e3725b9280cd Author: Jonathan Wakely Date: Mon Jun 10 14:08:16 2024 +0100 libstdc++: Fix std::tr2::dynamic_bitset shift operations [PR115399] The shift operations for dynamic_bitset fail to zero out words where the non-zero bits were shifted to a completely different word. For a right shift we don't need to sanitize the unused bits in the high word, because we know they were already clear and a right shift doesn't change that. libstdc++-v3/ChangeLog: PR libstdc++/115399 * include/tr2/dynamic_bitset (operator>>=): Remove redundant call to _M_do_sanitize. * include/tr2/dynamic_bitset.tcc (_M_do_left_shift): Zero out low bits in words that should no longer be populated. (_M_do_right_shift): Likewise for high bits. * testsuite/tr2/dynamic_bitset/pr115399.cc: New test. (cherry picked from commit bd3a312728fbf8c35a09239b9180269f938f872e) diff --git a/libstdc++-v3/include/tr2/dynamic_bitset b/libstdc++-v3/include/tr2/dynamic_bitset index 0d2160d611d..3bed740624b 100644 --- a/libstdc++-v3/include/tr2/dynamic_bitset +++ b/libstdc++-v3/include/tr2/dynamic_bitset @@ -815,10 +815,7 @@ namespace tr2 operator>>=(size_type __pos) { if (__builtin_expect(__pos < this->_M_Nb, 1)) - { - this->_M_do_right_shift(__pos); - this->_M_do_sanitize(); - } + this->_M_do_right_shift(__pos); else this->_M_do_reset(); return *this; diff --git a/libstdc++-v3/include/tr2/dynamic_bitset.tcc b/libstdc++-v3/include/tr2/dynamic_bitset.tcc index 8392ba6ffe6..41f4dc29120 100644 --- a/libstdc++-v3/include/tr2/dynamic_bitset.tcc +++ b/libstdc++-v3/include/tr2/dynamic_bitset.tcc @@ -60,8 +60,7 @@ namespace tr2 this->_M_w[__wshift] = this->_M_w[0] << __offset; } - //// std::fill(this->_M_w.begin(), this->_M_w.begin() + __wshift, - //// static_cast<_WordT>(0)); + std::fill_n(this->_M_w.begin(), __wshift, _WordT(0)); } } @@ -88,8 +87,7 @@ namespace tr2 this->_M_w[__limit] = this->_M_w[_M_w.size()-1] >> __offset; } - ////std::fill(this->_M_w.begin() + __limit + 1, this->_M_w.end(), - //// static_cast<_WordT>(0)); + std::fill_n(this->_M_w.end() - __wshift, __wshift, _WordT(0)); } } diff --git a/libstdc++-v3/testsuite/tr2/dynamic_bitset/pr115399.cc b/libstdc++-v3/testsuite/tr2/dynamic_bitset/pr115399.cc new file mode 100644 index 00000000000..e626e4a5d15 --- /dev/null +++ b/libstdc++-v3/testsuite/tr2/dynamic_bitset/pr115399.cc @@ -0,0 +1,37 @@ +// { dg-do run { target c++11 } } + +// PR libstdc++/115399 +// std::tr2::dynamic_bitset shift behaves differently from std::bitset + +#include +#include + +void +test_left_shift() +{ + std::tr2::dynamic_bitset<> b(65); + b[0] = 1; + auto b2 = b << 64; + VERIFY(b2[64] == 1); + VERIFY(b2[0] == 0); + b <<= 64; + VERIFY( b2 == b ); +} + +void +test_right_shift() +{ + std::tr2::dynamic_bitset<> b(65); + b[64] = 1; + auto b2 = b >> 64; + VERIFY(b2[64] == 0); + VERIFY(b2[0] == 1); + b >>= 64; + VERIFY( b2 == b ); +} + +int main() +{ + test_left_shift(); + test_right_shift(); +} commit 135be552d134c47c6fc71b9d8c2eeb98bdd85ede Author: Jonathan Wakely Date: Mon Apr 8 17:41:00 2024 +0100 libstdc++: Handle EMLINK and EFTYPE in std::filesystem::remove_all Although POSIX requires ELOOP, FreeBSD documents that openat with O_NOFOLLOW returns EMLINK if the last component of a filename is a symbolic link. Check for EMLINK as well as ELOOP, so that the TOCTTOU mitigation in remove_all works correctly. See https://bugs.freebsd.org/bugzilla/show_bug.cgi?id=214633 or the FreeBSD man page for reference. According to its man page, DragonFlyBSD also uses EMLINK for this error, and NetBSD uses its own EFTYPE. OpenBSD follows POSIX and uses EMLINK. This fixes these failures on FreeBSD: FAIL: 27_io/filesystem/operations/remove_all.cc -std=gnu++17 execution test FAIL: experimental/filesystem/operations/remove_all.cc -std=gnu++17 execution test libstdc++-v3/ChangeLog: * src/c++17/fs_ops.cc (remove_all) [__FreeBSD__ || __DragonFly__]: Check for EMLINK as well as ELOOP. [__NetBSD__]: Check for EFTYPE as well as ELOOP. diff --git a/libstdc++-v3/src/c++17/fs_ops.cc b/libstdc++-v3/src/c++17/fs_ops.cc index 397c46413bb..0fb7c9e5244 100644 --- a/libstdc++-v3/src/c++17/fs_ops.cc +++ b/libstdc++-v3/src/c++17/fs_ops.cc @@ -1310,7 +1310,13 @@ fs::remove_all(const path& p) // Our work here is done. return 0; case ENOTDIR: - case ELOOP: + case ELOOP: // POSIX says openat with O_NOFOLLOW sets ELOOP for a symlink. +#if defined __FreeBSD__ || defined __DragonFly__ + case EMLINK: // Used instead of ELOOP +#endif +#if defined __NetBSD__ && defined EFTYPE + case EFTYPE: // Used instead of ELOOP +#endif // Not a directory, will remove below. break; #endif @@ -1350,7 +1356,13 @@ fs::remove_all(const path& p, error_code& ec) ec.clear(); return 0; case ENOTDIR: - case ELOOP: + case ELOOP: // POSIX says openat with O_NOFOLLOW sets ELOOP for a symlink. +#if defined __FreeBSD__ || defined __DragonFly__ + case EMLINK: // Used instead of ELOOP +#endif +#if defined __NetBSD__ && defined EFTYPE + case EFTYPE: // Used instead of ELOOP +#endif // Not a directory, will remove below. break; #endif commit 595e3fa77115559343655cc0ab53cde5e4f82b86 Author: Jonathan Wakely Date: Thu Jun 20 16:13:10 2024 +0100 libstdc++: Initialize base in test allocator's constructor This fixes a warning from one of the test allocators: warning: base class 'class std::allocator<__gnu_test::copy_tracker>' should be explicitly initialized in the copy constructor [-Wextra] libstdc++-v3/ChangeLog: * testsuite/util/testsuite_allocator.h (tracker_allocator): Initialize base class in copy constructor. (cherry picked from commit e2fb245b07f489ed5bfd9a945e0053b4a3211245) diff --git a/libstdc++-v3/testsuite/util/testsuite_allocator.h b/libstdc++-v3/testsuite/util/testsuite_allocator.h index 0c41181b4a5..74eae87dbf6 100644 --- a/libstdc++-v3/testsuite/util/testsuite_allocator.h +++ b/libstdc++-v3/testsuite/util/testsuite_allocator.h @@ -154,7 +154,7 @@ namespace __gnu_test tracker_allocator() { } - tracker_allocator(const tracker_allocator&) + tracker_allocator(const tracker_allocator& a) : Alloc(a) { } ~tracker_allocator() commit 60e536d6f1682f3009c598db0f9c268db5d1749c Author: Jonathan Wakely Date: Mon Nov 28 12:16:21 2022 +0000 libstdc++: Fix std::string_view for IL32P16 targets For H8/300 with -msx -mn -mint32 the type of (_M_len - __pos) is int, because int is wider than size_t so the operands are promoted. libstdc++-v3/ChangeLog: * include/std/string_view (basic_string_view::copy) Use explicit template argument for call to std::min. (basic_string_view::substr): Likewise. diff --git a/libstdc++-v3/include/std/string_view b/libstdc++-v3/include/std/string_view index 9ee88836381..23383f578e5 100644 --- a/libstdc++-v3/include/std/string_view +++ b/libstdc++-v3/include/std/string_view @@ -299,7 +299,7 @@ _GLIBCXX_BEGIN_NAMESPACE_VERSION { __glibcxx_requires_string_len(__str, __n); __pos = std::__sv_check(size(), __pos, "basic_string_view::copy"); - const size_type __rlen = std::min(__n, _M_len - __pos); + const size_type __rlen = std::min(__n, _M_len - __pos); // _GLIBCXX_RESOLVE_LIB_DEFECTS // 2777. basic_string_view::copy should use char_traits::copy traits_type::copy(__str, data() + __pos, __rlen); @@ -310,7 +310,7 @@ _GLIBCXX_BEGIN_NAMESPACE_VERSION substr(size_type __pos = 0, size_type __n = npos) const noexcept(false) { __pos = std::__sv_check(size(), __pos, "basic_string_view::substr"); - const size_type __rlen = std::min(__n, _M_len - __pos); + const size_type __rlen = std::min(__n, _M_len - __pos); return basic_string_view{_M_str + __pos, __rlen}; } commit 5a804924b8fec76a3c2cb3557d3197ea9cd331c7 Author: GCC Administrator Date: Tue Oct 8 00:20:03 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 09c64f0d788..4eb153b60d2 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241007 +20241008 diff --git a/libstdc++-v3/ChangeLog b/libstdc++-v3/ChangeLog index 92466e991a0..176f0d3de3f 100644 --- a/libstdc++-v3/ChangeLog +++ b/libstdc++-v3/ChangeLog @@ -1,3 +1,89 @@ +2024-10-07 Jonathan Wakely + + * include/std/string_view (basic_string_view::copy) Use explicit + template argument for call to std::min. + (basic_string_view::substr): Likewise. + +2024-10-07 Jonathan Wakely + + Backported from master: + 2024-06-21 Jonathan Wakely + + * testsuite/util/testsuite_allocator.h (tracker_allocator): + Initialize base class in copy constructor. + +2024-10-07 Jonathan Wakely + + * src/c++17/fs_ops.cc (remove_all) [__FreeBSD__ || __DragonFly__]: + Check for EMLINK as well as ELOOP. + [__NetBSD__]: Check for EFTYPE as well as ELOOP. + +2024-10-07 Jonathan Wakely + + Backported from master: + 2024-06-12 Jonathan Wakely + + PR libstdc++/115399 + * include/tr2/dynamic_bitset (operator>>=): Remove redundant + call to _M_do_sanitize. + * include/tr2/dynamic_bitset.tcc (_M_do_left_shift): Zero out + low bits in words that should no longer be populated. + (_M_do_right_shift): Likewise for high bits. + * testsuite/tr2/dynamic_bitset/pr115399.cc: New test. + +2024-10-07 Kim Gräsman + + Backported from master: + 2024-08-28 Kim Gräsman + + * include/bits/cpp_type_traits.h: Improve doxygen file docs. + +2024-10-07 Kim Gräsman + + Backported from master: + 2024-08-28 Kim Gräsman + + * config/cpu/aarch64/opt/ext/opt_random.h: Improve doxygen file + docs. + * config/cpu/i486/opt/ext/opt_random.h: Likewise. + +2024-10-07 Jonathan Wakely + + Backported from master: + 2024-07-06 Jonathan Wakely + + * include/std/variant (_Variant_storage::_M_reset): Use + __unlikely__ form of attribute instead of unlikely. + +2024-10-07 Jonathan Wakely + + Backported from master: + 2024-08-28 Jonathan Wakely + + * configure.ac: Fix check for O_NONBLOCK. + * config.h.in: Regenerate. + * configure: Regenerate. + +2024-10-07 Jonathan Wakely + + Backported from master: + 2024-09-10 Jonathan Wakely + + PR libstdc++/116641 + * include/bits/basic_string.h (operator=(basic_string&&)): Call + _M_assign instead of assign. + * testsuite/21_strings/basic_string/allocator/116641.cc: New + test. + +2024-10-07 Jonathan Wakely + + Backported from master: + 2024-06-28 Jonathan Wakely + + PR libstdc++/115585 + * src/c++11/assert_fail.cc (__glibcxx_assert_fail): Add + definition for non-verbose builds. + 2024-07-09 Jonathan Wakely Backported from master: commit 1b708ef1a23c2d8a94e89602ccd7af5e3ddb3f3a Author: GCC Administrator Date: Wed Oct 9 00:21:01 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 4eb153b60d2..6d0d2e77043 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241008 +20241009 commit 37897420a2b98a0e183216fe70a46d2152158700 Author: GCC Administrator Date: Thu Oct 10 00:20:40 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 6d0d2e77043..7f0f80c6439 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241009 +20241010 commit ee5c0b815469a7d31e2f26e375575f4739e673fb Author: GCC Administrator Date: Fri Oct 11 00:20:38 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7f0f80c6439..1c5e89a4845 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241010 +20241011 commit 02dcac79d89c8baa403f27dea27e54973a7dc61b Author: GCC Administrator Date: Sat Oct 12 00:21:00 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 1c5e89a4845..4c61f9c24a3 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241011 +20241012 commit ff93befe3660e50d826d78392244130a22b95952 Author: GCC Administrator Date: Sun Oct 13 00:19:43 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 4c61f9c24a3..dcf1ca303e9 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241012 +20241013 commit 30e071ab5639ad534d0c9ea75d16359c516bbf80 Author: GCC Administrator Date: Mon Oct 14 00:19:45 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index dcf1ca303e9..d04e3ab9041 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241013 +20241014 commit fb61a7a203c5de3552e11bd633bc351463e51594 Author: Steve Baird Date: Mon Jul 8 14:45:55 2024 -0700 ada: Type conversion in instance incorrectly rejected. In some cases, a legal type conversion in a generic package is correctly accepted but the corresponding type conversion in an instance of the generic is incorrectly rejected. gcc/ada/ PR ada/114593 * sem_res.adb (Valid_Conversion): Test In_Instance instead of In_Instance_Body. diff --git a/gcc/ada/sem_res.adb b/gcc/ada/sem_res.adb index 09a2d8adb50..7dde7538734 100644 --- a/gcc/ada/sem_res.adb +++ b/gcc/ada/sem_res.adb @@ -14273,7 +14273,7 @@ package body Sem_Res is -- If it was legal in the generic, it's legal in the instance - elsif In_Instance_Body then + elsif In_Instance then return True; -- If both are tagged types, check legality of view conversions commit 3bb9418811d2ce870bd1c1c98a2ddca1bdcce998 Author: Eric Botcazou Date: Mon Oct 14 11:57:57 2024 +0200 Add regression test gcc/testsuite/ PR ada/114593 * gnat.dg/specs/generic_inst2-child2.ads: New test. * gnat.dg/specs/generic_inst2.ads: New helper. * gnat.dg/specs/generic_inst2-child1.ads: Likewise. diff --git a/gcc/testsuite/gnat.dg/specs/generic_inst2-child1.ads b/gcc/testsuite/gnat.dg/specs/generic_inst2-child1.ads new file mode 100644 index 00000000000..18e212b2e58 --- /dev/null +++ b/gcc/testsuite/gnat.dg/specs/generic_inst2-child1.ads @@ -0,0 +1,17 @@ +generic +package Generic_Inst2.Child1 is + + function Get_Custom return Custom_Type; + +private + + type Dummy is null record; + + Placeholder : constant Dummy := (null record); + + -- This type conversion fails (though only when + -- instantiated in the other package) + function Get_Custom return Custom_Type is + (Custom_Type(Placeholder'Address)); + +end Generic_Inst2.Child1; diff --git a/gcc/testsuite/gnat.dg/specs/generic_inst2-child2.ads b/gcc/testsuite/gnat.dg/specs/generic_inst2-child2.ads new file mode 100644 index 00000000000..3bb5b0aca10 --- /dev/null +++ b/gcc/testsuite/gnat.dg/specs/generic_inst2-child2.ads @@ -0,0 +1,10 @@ +-- { dg-do compile } + +with Generic_Inst2.Child1; + +generic +package Generic_Inst2.Child2 is + + package Second is new Generic_Inst2.Child1; + +end Generic_Inst2.Child2; diff --git a/gcc/testsuite/gnat.dg/specs/generic_inst2.ads b/gcc/testsuite/gnat.dg/specs/generic_inst2.ads new file mode 100644 index 00000000000..0a74e36273b --- /dev/null +++ b/gcc/testsuite/gnat.dg/specs/generic_inst2.ads @@ -0,0 +1,11 @@ +private with System; + +package Generic_Inst2 is + + type Custom_Type is private; + +private + + type Custom_Type is new System.Address; + +end Generic_Inst2; commit ea7d7818fdc7a49be3d35acd441ce122e8bb477e Author: Aldy Hernandez Date: Tue May 23 12:34:45 2023 +0200 Remove buggy special case in irange::invert [PR109934]. This patch removes a buggy special case in irange::invert which seems to have been broken for a while, and probably never triggered because the legacy code was handled elsewhere, and the non-legacy code was using an int_range_max of int_range<255> which made it extremely likely for num_ranges == 255. However, with auto-resizing ranges, int_range_max will start off at 3 and can hit this bogus code in the unswitching code. PR tree-optimization/109934 gcc/ChangeLog: * value-range.cc (irange::invert): Remove buggy special case. gcc/testsuite/ChangeLog: * gcc.dg/tree-ssa/pr109934.c: New test. (cherry picked from commit 8d5f050dabbf6dd3b992c3b46661848dbcf30d9e) diff --git a/gcc/testsuite/gcc.dg/tree-ssa/pr109934.c b/gcc/testsuite/gcc.dg/tree-ssa/pr109934.c new file mode 100644 index 00000000000..08bd5ce95c6 --- /dev/null +++ b/gcc/testsuite/gcc.dg/tree-ssa/pr109934.c @@ -0,0 +1,22 @@ +// { dg-do run } +// { dg-options "-O3" } + +int printf(const char *, ...); +short a; +long b = 3, c; +int d(int e) { + switch (e) + case 111: + case 222: + case 44: + return 0; + return e; +} +int main() { + for (; a >= 0; --a) + if (d(c + 23) - 23) + b = 0; + + if (b != 3) + __builtin_abort (); +} diff --git a/gcc/value-range.cc b/gcc/value-range.cc index 000bbcf8917..cf57e695e47 100644 --- a/gcc/value-range.cc +++ b/gcc/value-range.cc @@ -1856,14 +1856,6 @@ irange::invert () signop sign = TYPE_SIGN (ttype); wide_int type_min = wi::min_value (prec, sign); wide_int type_max = wi::max_value (prec, sign); - if (m_num_ranges == m_max_ranges - && lower_bound () != type_min - && upper_bound () != type_max) - { - m_base[1] = wide_int_to_tree (ttype, type_max); - m_num_ranges = 1; - return; - } // The algorithm is as follows. To calculate INVERT ([a,b][c,d]), we // generate [-MIN, a-1][b+1, c-1][d+1, MAX]. // commit e098149b6b2c61ffe9e758dcd3ef021a78f2d132 Author: Jan Hubicka Date: Mon Jul 22 18:05:26 2024 +0200 Fix accounting of offsets in unadjusted_ptr_and_unit_offset unadjusted_ptr_and_unit_offset accidentally throws away the offset computed by get_addr_base_and_unit_offset. Instead of passing extra_offset it passes offset. PR ipa/114207 gcc/ChangeLog: * ipa-prop.cc (unadjusted_ptr_and_unit_offset): Fix accounting of offsets in ADDR_EXPR. gcc/testsuite/ChangeLog: * gcc.c-torture/execute/pr114207.c: New test. (cherry picked from commit 391f46f10b0586c074014de82efe76787739bb0c) diff --git a/gcc/ipa-prop.cc b/gcc/ipa-prop.cc index e2e83b5f3f5..848d62e49cf 100644 --- a/gcc/ipa-prop.cc +++ b/gcc/ipa-prop.cc @@ -1263,9 +1263,9 @@ unadjusted_ptr_and_unit_offset (tree op, tree *ret, poly_int64 *offset_ret) { if (TREE_CODE (op) == ADDR_EXPR) { - poly_int64 extra_offset = 0; + poly_int64 extra_offset; tree base = get_addr_base_and_unit_offset (TREE_OPERAND (op, 0), - &offset); + &extra_offset); if (!base) { base = get_base_address (TREE_OPERAND (op, 0)); diff --git a/gcc/testsuite/gcc.c-torture/execute/pr114207.c b/gcc/testsuite/gcc.c-torture/execute/pr114207.c new file mode 100644 index 00000000000..052fa85e9fc --- /dev/null +++ b/gcc/testsuite/gcc.c-torture/execute/pr114207.c @@ -0,0 +1,23 @@ +#include +#include + +struct S { + int a, b; +}; + +__attribute__((noinline)) +void foo (struct S *s) { + struct S ss = (struct S) { + .a = s->b, + .b = s->a + }; + *s = ss; +} + +int main() { + struct S s = {6, 12}; + foo(&s); + if (s.a != 12 || s.b != 6) + __builtin_abort (); + return 0; +} commit f2686f12e9345d95a2d6b7291bd777501d197869 Author: Jan Hubicka Date: Mon Jul 22 18:08:08 2024 +0200 Fix modref_eaf_analysis::analyze_ssa_name handling of values dereferenced to function call parameters modref_eaf_analysis::analyze_ssa_name misinterprets EAF flags. If dereferenced parameter is passed (to map_iterator in the testcase) it can be returned indirectly which in turn makes it to escape into the next function call. PR ipa/115033 gcc/ChangeLog: * ipa-modref.cc (modref_eaf_analysis::analyze_ssa_name): Fix checking of EAF flags when analysing values dereferenced as function parameters. gcc/testsuite/ChangeLog: * gcc.c-torture/execute/pr115033.c: New test. (cherry picked from commit cf8ffc58aad3127031c229a75cc4b99c8ace25e0) diff --git a/gcc/ipa-modref.cc b/gcc/ipa-modref.cc index c3e3fd50d17..bcd9b49fe63 100644 --- a/gcc/ipa-modref.cc +++ b/gcc/ipa-modref.cc @@ -2568,8 +2568,10 @@ modref_eaf_analysis::analyze_ssa_name (tree name, bool deferred) int call_flags = deref_flags (gimple_call_arg_flags (call, i), ignore_stores); if (!ignore_retval && !(call_flags & EAF_UNUSED) - && !(call_flags & EAF_NOT_RETURNED_DIRECTLY) - && !(call_flags & EAF_NOT_RETURNED_INDIRECTLY)) + && (call_flags & (EAF_NOT_RETURNED_DIRECTLY + | EAF_NOT_RETURNED_INDIRECTLY)) + != (EAF_NOT_RETURNED_DIRECTLY + | EAF_NOT_RETURNED_INDIRECTLY)) merge_call_lhs_flags (call, i, name, false, true); if (ecf_flags & (ECF_CONST | ECF_NOVOPS)) m_lattice[index].merge_direct_load (); diff --git a/gcc/testsuite/gcc.c-torture/execute/pr115033.c b/gcc/testsuite/gcc.c-torture/execute/pr115033.c new file mode 100644 index 00000000000..3e79367d401 --- /dev/null +++ b/gcc/testsuite/gcc.c-torture/execute/pr115033.c @@ -0,0 +1,35 @@ + +typedef struct func +{ + int *a; +}func; +__attribute__((noinline)) +void ff(struct func *t) +{ + *(t->a) = 0; +} + + +typedef struct mapped_iterator { + func F; +}mapped_iterator; + +__attribute__((noinline)) +mapped_iterator map_iterator(func F) { + mapped_iterator t = {F}; + return t; +} + +void map_to_vector(func *F) { + mapped_iterator t = map_iterator(*F); + ff(&t.F); +} +int main() { + int resultIsStatic = 1; + func t ={&resultIsStatic}; + map_to_vector(&t); + + if (resultIsStatic) + __builtin_trap(); + __builtin_exit(0); +} commit 92889011625f16e7eed654a137b9c14a29282917 Author: Jan Hubicka Date: Thu May 16 15:33:55 2024 +0200 Fix points_to_local_or_readonly_memory_p wrt TARGET_MEM_REF TARGET_MEM_REF can be used to offset constant base into a memory object (to produce lea instruction). This confuses points_to_local_or_readonly_memory_p which treats the constant address as a base of the access. Bootstrapped/regtsted x86_64-linux, comitted. Honza gcc/ChangeLog: PR ipa/113787 * ipa-fnsummary.cc (points_to_local_or_readonly_memory_p): Do not look into TARGET_MEM_REFS with constant opreand 0. gcc/testsuite/ChangeLog: * gcc.c-torture/execute/pr113787.c: New test. (cherry picked from commit 96d53252aefcbc2fe419c4c3b4bcd3fc03d4d187) diff --git a/gcc/ipa-fnsummary.cc b/gcc/ipa-fnsummary.cc index b12e7a1124d..65e6687428e 100644 --- a/gcc/ipa-fnsummary.cc +++ b/gcc/ipa-fnsummary.cc @@ -2589,7 +2589,9 @@ points_to_local_or_readonly_memory_p (tree t) return true; return !ptr_deref_may_alias_global_p (t, false); } - if (TREE_CODE (t) == ADDR_EXPR) + if (TREE_CODE (t) == ADDR_EXPR + && (TREE_CODE (TREE_OPERAND (t, 0)) != TARGET_MEM_REF + || TREE_CODE (TREE_OPERAND (TREE_OPERAND (t, 0), 0)) != INTEGER_CST)) return refs_local_or_readonly_memory_p (TREE_OPERAND (t, 0)); return false; } diff --git a/gcc/testsuite/gcc.c-torture/execute/pr113787.c b/gcc/testsuite/gcc.c-torture/execute/pr113787.c new file mode 100644 index 00000000000..702b6c35fc6 --- /dev/null +++ b/gcc/testsuite/gcc.c-torture/execute/pr113787.c @@ -0,0 +1,38 @@ +void foo(int x, int y, int z, int d, int *buf) +{ + for(int i = z; i < y-z; ++i) + for(int j = 0; j < d; ++j) + /* buf[x(i+1) + j] = buf[x(i+1)-j-1] */ + buf[i*x+(x-z+j)] = buf[i*x+(x-z-1-j)]; +} + +void bar(int x, int y, int z, int d, int *buf) +{ + for(int i = 0; i < d; ++i) + for(int j = z; j < x-z; ++j) + /* buf[j+(y+i)*x] = buf[j+(y-1-i)*x] */ + buf[j+(y-z+i)*x] = buf[j+(y-z-1-i)*x]; +} + +__attribute__((noipa)) +void baz(int x, int y, int d, int *buf) +{ + foo(x, y, 0, d, buf); + bar(x, y, 0, d, buf); +} + +int main(void) +{ + int a[] = { 1, 2, 3 }; + baz (1, 2, 1, a); + /* foo does: + buf[1] = buf[0]; + buf[2] = buf[1]; + + bar does: + buf[2] = buf[1]; (no-op) + so we should have { 1, 1, 1 }. */ + for (int i = 0; i < 3; i++) + if (a[i] != 1) + __builtin_abort (); +} commit b454ad0e4eb6bb38a6dfe15a43acb6792315640b Author: Jan Hubicka Date: Mon Jul 22 23:01:50 2024 +0200 Fix handling of ICF_NOVOPS in ipa-modref As shown in somewhat convoluted testcase, ipa-modref is mistreating ECF_NOVOPS as "having no side effects". This come from time when modref cared only about memory accesses and thus it was possible to shortcut on it. This patch removes (hopefully) all those bad shortcuts. Bootstrapped/regtested x86_64-linux, comitted. gcc/ChangeLog: PR ipa/109985 * ipa-modref.cc (modref_summary::useful_p): Fix handling of ECF_NOVOPS. (modref_access_analysis::process_fnspec): Likevise. (modref_access_analysis::analyze_call): Likevise. (propagate_unknown_call): Likevise. (modref_propagate_in_scc): Likevise. (modref_propagate_flags_in_scc): Likewise. (ipa_merge_modref_summary_after_inlining): Likewise. (cherry picked from commit efcbe7b985e24ac002a863afd609c44a67761195) diff --git a/gcc/ipa-modref.cc b/gcc/ipa-modref.cc index bcd9b49fe63..1d954954786 100644 --- a/gcc/ipa-modref.cc +++ b/gcc/ipa-modref.cc @@ -332,7 +332,7 @@ modref_summary::useful_p (int ecf_flags, bool check_flags) if (check_flags && remove_useless_eaf_flags (static_chain_flags, ecf_flags, false)) return true; - if (ecf_flags & (ECF_CONST | ECF_NOVOPS)) + if (ecf_flags & ECF_CONST) return ((!side_effects || !nondeterministic) && (ecf_flags & ECF_LOOPING_CONST_OR_PURE)); if (loads && !loads->every_base) @@ -1261,7 +1261,7 @@ modref_access_analysis::merge_call_side_effects int flags = gimple_call_flags (call); /* Nothing to do for non-looping cont functions. */ - if ((flags & (ECF_CONST | ECF_NOVOPS)) + if ((flags & ECF_CONST) && !(flags & ECF_LOOPING_CONST_OR_PURE)) return false; @@ -1274,7 +1274,7 @@ modref_access_analysis::merge_call_side_effects /* Merge side effects and non-determinism. PURE/CONST flags makes functions deterministic and if there is no LOOPING_CONST_OR_PURE they also have no side effects. */ - if (!(flags & (ECF_CONST | ECF_NOVOPS | ECF_PURE)) + if (!(flags & (ECF_CONST | ECF_PURE)) || (flags & ECF_LOOPING_CONST_OR_PURE)) { if (!m_summary->side_effects && callee_summary->side_effects) @@ -1463,7 +1463,7 @@ modref_access_analysis::process_fnspec (gcall *call) /* PURE/CONST flags makes functions deterministic and if there is no LOOPING_CONST_OR_PURE they also have no side effects. */ - if (!(flags & (ECF_CONST | ECF_NOVOPS | ECF_PURE)) + if (!(flags & (ECF_CONST | ECF_PURE)) || (flags & ECF_LOOPING_CONST_OR_PURE) || (cfun->can_throw_non_call_exceptions && stmt_could_throw_p (cfun, call))) @@ -1602,12 +1602,12 @@ modref_access_analysis::analyze_call (gcall *stmt) print_gimple_stmt (dump_file, stmt, 0); } - if ((flags & (ECF_CONST | ECF_NOVOPS)) + if ((flags & ECF_CONST) && !(flags & ECF_LOOPING_CONST_OR_PURE)) { if (dump_file) fprintf (dump_file, - " - ECF_CONST | ECF_NOVOPS, ignoring all stores and all loads " + " - ECF_CONST, ignoring all stores and all loads " "except for args.\n"); return; } @@ -1622,7 +1622,13 @@ modref_access_analysis::analyze_call (gcall *stmt) if (dump_file) fprintf (dump_file, gimple_call_internal_p (stmt) ? " - Internal call" : " - Indirect call.\n"); - process_fnspec (stmt); + if (flags & ECF_NOVOPS) + { + set_side_effects (); + set_nondeterministic (); + } + else + process_fnspec (stmt); return; } /* We only need to handle internal calls in IPA mode. */ @@ -4510,7 +4516,7 @@ propagate_unknown_call (cgraph_node *node, return changed; } - if (!(ecf_flags & (ECF_CONST | ECF_NOVOPS | ECF_PURE)) + if (!(ecf_flags & (ECF_CONST | ECF_PURE)) || (ecf_flags & ECF_LOOPING_CONST_OR_PURE) || nontrivial_scc) { @@ -4724,7 +4730,7 @@ modref_propagate_in_scc (cgraph_node *component_node) struct cgraph_node *callee; if (!callee_edge->inline_failed - || ((flags & (ECF_CONST | ECF_NOVOPS)) + || ((flags & ECF_CONST) && !(flags & ECF_LOOPING_CONST_OR_PURE))) continue; @@ -5147,8 +5153,8 @@ modref_propagate_flags_in_scc (cgraph_node *component_node) { escape_summary *sum = escape_summaries->get (e); - if (!sum || (e->indirect_info->ecf_flags - & (ECF_CONST | ECF_NOVOPS))) + if (!sum || ((e->indirect_info->ecf_flags & ECF_CONST) + && !(e->indirect_info->ecf_flags & ECF_LOOPING_CONST_OR_PURE))) continue; changed |= modref_merge_call_site_flags @@ -5173,8 +5179,8 @@ modref_propagate_flags_in_scc (cgraph_node *component_node) modref_summary_lto *callee_summary_lto = NULL; struct cgraph_node *callee; - if (ecf_flags & (ECF_CONST | ECF_NOVOPS) - || !callee_edge->inline_failed) + if ((ecf_flags & ECF_CONST) + && !(ecf_flags & ECF_LOOPING_CONST_OR_PURE)) continue; /* Get the callee and its summary. */ @@ -5272,7 +5278,7 @@ ipa_merge_modref_summary_after_inlining (cgraph_edge *edge) if (!callee_info && to_info) { - if (!(flags & (ECF_CONST | ECF_NOVOPS))) + if (!(flags & (ECF_CONST | ECF_PURE | ECF_NOVOPS))) to_info->loads->collapse (); if (!ignore_stores) to_info->stores->collapse (); @@ -5287,7 +5293,7 @@ ipa_merge_modref_summary_after_inlining (cgraph_edge *edge) /* Merge side effects and non-determinism. PURE/CONST flags makes functions deterministic and if there is no LOOPING_CONST_OR_PURE they also have no side effects. */ - if (!(flags & (ECF_CONST | ECF_NOVOPS | ECF_PURE)) + if (!(flags & (ECF_CONST | ECF_PURE)) || (flags & ECF_LOOPING_CONST_OR_PURE)) { if (to_info) commit 65b67169fb1c641a11973988bd439e255d62131a Author: Jan Hubicka Date: Mon Jul 22 19:00:39 2024 +0200 Fix modref's iteraction with store merging Hi, this patch fixes wrong code in case store-merging introduces load of function parameter that was previously write-only (which happens for bitfields). Without this, the whole store-merged area is consdered to be killed. PR ipa/111613 gcc/ChangeLog: * ipa-modref.cc (analyze_parms): Do not preserve EAF_NO_DIRECT_READ and EAF_NO_INDIRECT_READ from past flags. gcc/testsuite/ChangeLog: * gcc.c-torture/pr111613.c: New test. (cherry picked from commit 14074773350ffed7efdebbc553adf0f23b572e87) diff --git a/gcc/ipa-modref.cc b/gcc/ipa-modref.cc index 1d954954786..ba7f21834ef 100644 --- a/gcc/ipa-modref.cc +++ b/gcc/ipa-modref.cc @@ -3008,6 +3008,9 @@ analyze_parms (modref_summary *summary, modref_summary_lto *summary_lto, (past, ecf_flags, VOID_TYPE_P (TREE_TYPE (TREE_TYPE (current_function_decl)))); + /* Store merging can produce reads when combining together multiple + bitfields. See PR111613. */ + past &= ~(EAF_NO_DIRECT_READ | EAF_NO_INDIRECT_READ); if (dump_file && (flags | past) != flags && !(flags & EAF_UNUSED)) { fprintf (dump_file, diff --git a/gcc/testsuite/gcc.c-torture/pr111613.c b/gcc/testsuite/gcc.c-torture/pr111613.c new file mode 100644 index 00000000000..1ea1c4dec07 --- /dev/null +++ b/gcc/testsuite/gcc.c-torture/pr111613.c @@ -0,0 +1,29 @@ +#include +#include + +struct bitfield { + unsigned int field1 : 1; + unsigned int field2 : 1; + unsigned int field3 : 1; +}; + +__attribute__((noinline)) static void +set_field1_and_field2(struct bitfield *b) { + b->field1 = 1; + b->field2 = 1; +} + +__attribute__((noinline)) static struct bitfield * +new_bitfield(void) { + struct bitfield *b = (struct bitfield *)malloc(sizeof(*b)); + b->field3 = 1; + set_field1_and_field2(b); + return b; +} + +int main(void) { + struct bitfield *b = new_bitfield(); + if (b->field3 != 1) + __builtin_abort (); + return 0; +} commit 0e3580191b449e599a4c8d7bfd634df2ca351d4a Author: Sam James Date: Mon Jul 29 21:47:16 2024 +0100 testsuite: fix PR111613 test PR ipa/111613 * gcc.c-torture/pr111613.c: Rename to.. * gcc.c-torture/execute/pr111613.c: ...this. (cherry picked from commit 5e5d7a88932b132437069f716160f8b20862890b) diff --git a/gcc/testsuite/gcc.c-torture/pr111613.c b/gcc/testsuite/gcc.c-torture/execute/pr111613.c similarity index 100% rename from gcc/testsuite/gcc.c-torture/pr111613.c rename to gcc/testsuite/gcc.c-torture/execute/pr111613.c commit 3dc10ba11a971f37ecdfc4b71783dce279421ede Author: GCC Administrator Date: Wed Oct 16 11:39:42 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index e66dcf4aa0d..df994cecae2 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,60 @@ +2024-10-14 Jan Hubicka + + Backported from master: + 2024-07-22 Jan Hubicka + + PR ipa/111613 + * ipa-modref.cc (analyze_parms): Do not preserve EAF_NO_DIRECT_READ and + EAF_NO_INDIRECT_READ from past flags. + +2024-10-14 Jan Hubicka + + Backported from master: + 2024-07-22 Jan Hubicka + + PR ipa/109985 + * ipa-modref.cc (modref_summary::useful_p): Fix handling of ECF_NOVOPS. + (modref_access_analysis::process_fnspec): Likevise. + (modref_access_analysis::analyze_call): Likevise. + (propagate_unknown_call): Likevise. + (modref_propagate_in_scc): Likevise. + (modref_propagate_flags_in_scc): Likewise. + (ipa_merge_modref_summary_after_inlining): Likewise. + +2024-10-14 Jan Hubicka + + Backported from master: + 2024-05-16 Jan Hubicka + + PR ipa/113787 + * ipa-fnsummary.cc (points_to_local_or_readonly_memory_p): Do not + look into TARGET_MEM_REFS with constant opreand 0. + +2024-10-14 Jan Hubicka + + Backported from master: + 2024-07-22 Jan Hubicka + + PR ipa/115033 + * ipa-modref.cc (modref_eaf_analysis::analyze_ssa_name): Fix checking of + EAF flags when analysing values dereferenced as function parameters. + +2024-10-14 Jan Hubicka + + Backported from master: + 2024-07-22 Jan Hubicka + + PR ipa/114207 + * ipa-prop.cc (unadjusted_ptr_and_unit_offset): Fix accounting of offsets in ADDR_EXPR. + +2024-10-14 Aldy Hernandez + + Backported from master: + 2023-05-23 Aldy Hernandez + + PR tree-optimization/109934 + * value-range.cc (irange::invert): Remove buggy special case. + 2024-10-05 John David Anglin * config/pa/pa.md: Fix indirect_got constraint. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d04e3ab9041..f776b88810d 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241014 +20241016 diff --git a/gcc/ada/ChangeLog b/gcc/ada/ChangeLog index 50e3c3a1892..a33b1c09495 100644 --- a/gcc/ada/ChangeLog +++ b/gcc/ada/ChangeLog @@ -1,3 +1,9 @@ +2024-10-14 Steve Baird + + PR ada/114593 + * sem_res.adb (Valid_Conversion): Test In_Instance instead of + In_Instance_Body. + 2024-06-20 Release Manager * GCC 12.4.0 released. diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 2ecd420c938..1b19624fe6c 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,58 @@ +2024-10-14 Sam James + + Backported from master: + 2024-07-29 Sam James + + PR ipa/111613 + * gcc.c-torture/pr111613.c: Rename to.. + * gcc.c-torture/execute/pr111613.c: ...this. + +2024-10-14 Jan Hubicka + + Backported from master: + 2024-07-22 Jan Hubicka + + PR ipa/111613 + * gcc.c-torture/pr111613.c: New test. + +2024-10-14 Jan Hubicka + + Backported from master: + 2024-05-16 Jan Hubicka + + * gcc.c-torture/execute/pr113787.c: New test. + +2024-10-14 Jan Hubicka + + Backported from master: + 2024-07-22 Jan Hubicka + + PR ipa/115033 + * gcc.c-torture/execute/pr115033.c: New test. + +2024-10-14 Jan Hubicka + + Backported from master: + 2024-07-22 Jan Hubicka + + PR ipa/114207 + * gcc.c-torture/execute/pr114207.c: New test. + +2024-10-14 Aldy Hernandez + + Backported from master: + 2023-05-23 Aldy Hernandez + + PR tree-optimization/109934 + * gcc.dg/tree-ssa/pr109934.c: New test. + +2024-10-14 Eric Botcazou + + PR ada/114593 + * gnat.dg/specs/generic_inst2-child2.ads: New test. + * gnat.dg/specs/generic_inst2.ads: New helper. + * gnat.dg/specs/generic_inst2-child1.ads: Likewise. + 2024-10-04 H.J. Lu Backported from master: commit a8bd38de88715fdbf0d064ff0d50e2b8734de939 Author: Uros Bizjak Date: Tue Oct 15 16:51:33 2024 +0200 i386: Fix expand_vector_set for VEC_MERGE/VEC_DUPLICATE RTX [PR117116] Middle end can generate SYMBOL_REF RTX as a value "val" in the call to expand_vector_set, but SYMBOL_REF RTX is not accepted in _pinsr insn pattern, generated via VEC_MERGE/VEC_DUPLICATE RTX path. Force the value into a register before VEC_MERGE/VEC_DUPLICATE RTX is generated if it doesn't satisfy nonimmediate_operand predicate. PR target/117116 gcc/ChangeLog: * config/i386/i386-expand.cc (expand_vector_set): Force "val" into a register before VEC_MERGE/VEC_DUPLICATE RTX is generated if it doesn't satisfy nonimmediate_operand predicate. gcc/testsuite/ChangeLog: * gcc.target/i386/pr117116.c: New test. (cherry picked from commit 80d7032067a3a5b76aecd657d9b35b0a8f5a941d) diff --git a/gcc/config/i386/i386-expand.cc b/gcc/config/i386/i386-expand.cc index c57a8f56dac..909c11e4195 100644 --- a/gcc/config/i386/i386-expand.cc +++ b/gcc/config/i386/i386-expand.cc @@ -16541,6 +16541,8 @@ quarter: else if (use_vec_merge) { do_vec_merge: + if (!nonimmediate_operand (val, inner_mode)) + val = force_reg (inner_mode, val); tmp = gen_rtx_VEC_DUPLICATE (mode, val); tmp = gen_rtx_VEC_MERGE (mode, tmp, target, GEN_INT (HOST_WIDE_INT_1U << elt)); diff --git a/gcc/testsuite/gcc.target/i386/pr117116.c b/gcc/testsuite/gcc.target/i386/pr117116.c new file mode 100644 index 00000000000..d6e28848a4b --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr117116.c @@ -0,0 +1,18 @@ +/* PR target/117116 */ +/* { dg-do compile } */ +/* { dg-options "-O2 -mavx2" } */ + +typedef void (*StmFct)(); +typedef struct { + StmFct fct_getc; + StmFct fct_putc; + StmFct fct_flush; + StmFct fct_close; +} StmInf; + +StmInf TTY_Getc_pstm; + +void TTY_Getc() { + TTY_Getc_pstm.fct_getc = TTY_Getc; + TTY_Getc_pstm.fct_putc = TTY_Getc_pstm.fct_flush = TTY_Getc_pstm.fct_close = (StmFct)1; +} commit 08f1bd1088063eba65a6e6f4b47370d5b65c97c6 Author: GCC Administrator Date: Sat Oct 19 00:21:12 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index df994cecae2..bff57f4da6c 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,13 @@ +2024-10-16 Uros Bizjak + + Backported from master: + 2024-10-15 Uros Bizjak + + PR target/117116 + * config/i386/i386-expand.cc (expand_vector_set): Force "val" + into a register before VEC_MERGE/VEC_DUPLICATE RTX is generated + if it doesn't satisfy nonimmediate_operand predicate. + 2024-10-14 Jan Hubicka Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index f776b88810d..c8d7d3442a8 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241016 +20241019 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 1b19624fe6c..236aaba929f 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-10-16 Uros Bizjak + + Backported from master: + 2024-10-15 Uros Bizjak + + PR target/117116 + * gcc.target/i386/pr117116.c: New test. + 2024-10-14 Sam James Backported from master: commit 947cb45d8d886257029813cd4c99306310ae63f9 Author: GCC Administrator Date: Sun Oct 20 00:20:23 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c8d7d3442a8..f616e4b768b 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241019 +20241020 commit 69c50a35885df4e4a5077390e8f2ee2e994513a5 Author: GCC Administrator Date: Mon Oct 21 00:20:25 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index f616e4b768b..18b2d489abc 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241020 +20241021 commit 91800a70a2af1349eefc5f3380be2b254b1db395 Author: liuhongt Date: Wed Oct 16 13:43:48 2024 +0800 Refine splitters related to "combine vpcmpuw + zero_extend to vpcmpuw" r12-6103-g1a7ce8570997eb combines vpcmpuw + zero_extend to vpcmpuw with the pre_reload splitter, but the splitter transforms the zero_extend into a subreg which make reload think the upper part is garbage, it's not correct. The patch adjusts the zero_extend define_insn_and_split to define_insn to keep zero_extend. gcc/ChangeLog: PR target/117159 * config/i386/sse.md (*_cmp3_zero_extend): Change from define_insn_and_split to define_insn. (*_cmp3_zero_extend): Ditto. (*_ucmp3_zero_extend): Ditto. (*_ucmp3_zero_extend): Ditto. (*_cmp3_zero_extend_2): Split to the zero_extend pattern. (*_cmp3_zero_extend_2): Ditto. (*_ucmp3_zero_extend_2): Ditto. (*_ucmp3_zero_extend_2): Ditto. gcc/testsuite/ChangeLog: * gcc.target/i386/pr117159.c: New test. * gcc.target/i386/avx512bw-pr103750-1.c: Remove xfail. * gcc.target/i386/avx512bw-pr103750-2.c: Remove xfail. (cherry picked from commit 5259d3927c1c8e3a15b4b844adef59b48c241233) diff --git a/gcc/config/i386/sse.md b/gcc/config/i386/sse.md index 924effce2b5..c94c8eceb33 100644 --- a/gcc/config/i386/sse.md +++ b/gcc/config/i386/sse.md @@ -3724,32 +3724,19 @@ ;; Since vpcmpd implicitly clear the upper bits of dest, transform ;; vpcmpd + zero_extend to vpcmpd since the instruction -(define_insn_and_split "*_cmp3_zero_extend" - [(set (match_operand:SWI248x 0 "register_operand") +(define_insn "*_cmp3_zero_extend" + [(set (match_operand:SWI248x 0 "register_operand" "=k") (zero_extend:SWI248x (unspec: - [(match_operand:V48H_AVX512VL 1 "nonimmediate_operand") - (match_operand:V48H_AVX512VL 2 "nonimmediate_operand") + [(match_operand:V48H_AVX512VL 1 "nonimmediate_operand" "v") + (match_operand:V48H_AVX512VL 2 "nonimmediate_operand" "vm") (match_operand:SI 3 "const_0_to_7_operand" "n")] UNSPEC_PCMP)))] "TARGET_AVX512F && (!VALID_MASK_AVX512BW_MODE (mode) || TARGET_AVX512BW) - && ix86_pre_reload_split () && (GET_MODE_NUNITS (mode) < GET_MODE_PRECISION (mode))" - "#" - "&& 1" - [(set (match_dup 0) - (unspec: - [(match_dup 1) - (match_dup 2) - (match_dup 3)] - UNSPEC_PCMP))] -{ - operands[1] = force_reg (mode, operands[1]); - operands[0] = lowpart_subreg (mode, - operands[0], mode); -} + "vcmp\t{%3, %2, %1, %0|%0, %1, %2, %3}" [(set_attr "type" "ssecmp") (set_attr "length_immediate" "1") (set_attr "prefix" "evex") @@ -3777,21 +3764,22 @@ "#" "&& 1" [(set (match_dup 0) - (unspec: - [(match_dup 1) - (match_dup 2) - (match_dup 3)] - UNSPEC_PCMP)) - (set (match_dup 4) (match_dup 0))] + (zero_extend:SWI248x + (unspec: + [(match_dup 1) + (match_dup 2) + (match_dup 3)] + UNSPEC_PCMP))) + (set (match_dup 4) (match_dup 5))] { - operands[1] = force_reg (mode, operands[1]); - operands[0] = lowpart_subreg (mode, + operands[5] = lowpart_subreg (mode, operands[0], mode); -} - [(set_attr "type" "ssecmp") - (set_attr "length_immediate" "1") - (set_attr "prefix" "evex") - (set_attr "mode" "")]) + if (SUBREG_P (operands[5])) + { + SUBREG_PROMOTED_VAR_P (operands[5]) = 1; + SUBREG_PROMOTED_SET (operands[5], 1); + } +}) (define_insn_and_split "*_cmp3" [(set (match_operand: 0 "register_operand") @@ -3826,31 +3814,18 @@ (set_attr "prefix" "evex") (set_attr "mode" "")]) -(define_insn_and_split "*_cmp3_zero_extend" - [(set (match_operand:SWI248x 0 "register_operand") +(define_insn "*_cmp3_zero_extend" + [(set (match_operand:SWI248x 0 "register_operand" "=k") (zero_extend:SWI248x (unspec: - [(match_operand:VI12_AVX512VL 1 "nonimmediate_operand") - (match_operand:VI12_AVX512VL 2 "nonimmediate_operand") - (match_operand:SI 3 "const_0_to_7_operand")] + [(match_operand:VI12_AVX512VL 1 "nonimmediate_operand" "v") + (match_operand:VI12_AVX512VL 2 "nonimmediate_operand" "vm") + (match_operand:SI 3 "const_0_to_7_operand" "n")] UNSPEC_PCMP)))] "TARGET_AVX512BW - && ix86_pre_reload_split () - && (GET_MODE_NUNITS (mode) - < GET_MODE_PRECISION (mode))" - "#" - "&& 1" - [(set (match_dup 0) - (unspec: - [(match_dup 1) - (match_dup 2) - (match_dup 3)] - UNSPEC_PCMP))] -{ - operands[1] = force_reg (mode, operands[1]); - operands[0] = lowpart_subreg (mode, - operands[0], mode); -} + && (GET_MODE_NUNITS (mode) + < GET_MODE_PRECISION (mode))" + "vpcmp\t{%3, %2, %1, %0|%0, %1, %2, %3}" [(set_attr "type" "ssecmp") (set_attr "length_immediate" "1") (set_attr "prefix" "evex") @@ -3877,16 +3852,21 @@ "#" "&& 1" [(set (match_dup 0) - (unspec: - [(match_dup 1) - (match_dup 2) - (match_dup 3)] - UNSPEC_PCMP)) - (set (match_dup 4) (match_dup 0))] + (zero_extend:SWI248x + (unspec: + [(match_dup 1) + (match_dup 2) + (match_dup 3)] + UNSPEC_PCMP))) + (set (match_dup 4) (match_dup 5))] { - operands[1] = force_reg (mode, operands[1]); - operands[0] = lowpart_subreg (mode, + operands[5] = lowpart_subreg (mode, operands[0], mode); + if (SUBREG_P (operands[5])) + { + SUBREG_PROMOTED_VAR_P (operands[5]) = 1; + SUBREG_PROMOTED_SET (operands[5], 1); + } } [(set_attr "type" "ssecmp") (set_attr "length_immediate" "1") @@ -3945,31 +3925,18 @@ (set_attr "prefix" "evex") (set_attr "mode" "")]) -(define_insn_and_split "*_ucmp3_zero_extend" - [(set (match_operand:SWI248x 0 "register_operand") +(define_insn "*_ucmp3_zero_extend" + [(set (match_operand:SWI248x 0 "register_operand" "=k") (zero_extend:SWI248x (unspec: - [(match_operand:VI12_AVX512VL 1 "nonimmediate_operand") - (match_operand:VI12_AVX512VL 2 "nonimmediate_operand") - (match_operand:SI 3 "const_0_to_7_operand")] + [(match_operand:VI12_AVX512VL 1 "nonimmediate_operand" "v") + (match_operand:VI12_AVX512VL 2 "nonimmediate_operand" "vm") + (match_operand:SI 3 "const_0_to_7_operand" "n")] UNSPEC_UNSIGNED_PCMP)))] "TARGET_AVX512BW - && ix86_pre_reload_split () && (GET_MODE_NUNITS (mode) < GET_MODE_PRECISION (mode))" - "#" - "&& 1" - [(set (match_dup 0) - (unspec: - [(match_dup 1) - (match_dup 2) - (match_dup 3)] - UNSPEC_UNSIGNED_PCMP))] -{ - operands[1] = force_reg (mode, operands[1]); - operands[0] = lowpart_subreg (mode, - operands[0], mode); -} + "vpcmpu\t{%3, %2, %1, %0|%0, %1, %2, %3}" [(set_attr "type" "ssecmp") (set_attr "length_immediate" "1") (set_attr "prefix" "evex") @@ -3997,16 +3964,21 @@ "#" "&& 1" [(set (match_dup 0) - (unspec: - [(match_dup 1) - (match_dup 2) - (match_dup 3)] - UNSPEC_UNSIGNED_PCMP)) - (set (match_dup 4) (match_dup 0))] -{ - operands[1] = force_reg (mode, operands[1]); - operands[0] = lowpart_subreg (mode, + (zero_extend:SWI248x + (unspec: + [(match_dup 1) + (match_dup 2) + (match_dup 3)] + UNSPEC_UNSIGNED_PCMP))) + (set (match_dup 4) (match_dup 5))] +{ + operands[5] = lowpart_subreg (mode, operands[0], mode); + if (SUBREG_P (operands[5])) + { + SUBREG_PROMOTED_VAR_P (operands[5]) = 1; + SUBREG_PROMOTED_SET (operands[5], 1); + } } [(set_attr "type" "ssecmp") (set_attr "length_immediate" "1") @@ -4043,32 +4015,19 @@ (set_attr "prefix" "evex") (set_attr "mode" "")]) -(define_insn_and_split "*_ucmp3_zero_extend" - [(set (match_operand:SWI248x 0 "register_operand") +(define_insn "*_ucmp3_zero_extend" + [(set (match_operand:SWI248x 0 "register_operand" "=k") (zero_extend:SWI248x (unspec: - [(match_operand:VI48_AVX512VL 1 "nonimmediate_operand") - (match_operand:VI48_AVX512VL 2 "nonimmediate_operand") - (match_operand:SI 3 "const_0_to_7_operand")] + [(match_operand:VI48_AVX512VL 1 "nonimmediate_operand" "v") + (match_operand:VI48_AVX512VL 2 "nonimmediate_operand" "vm") + (match_operand:SI 3 "const_0_to_7_operand" "n")] UNSPEC_UNSIGNED_PCMP)))] "TARGET_AVX512F && (!VALID_MASK_AVX512BW_MODE (mode) || TARGET_AVX512BW) - && ix86_pre_reload_split () && (GET_MODE_NUNITS (mode) < GET_MODE_PRECISION (mode))" - "#" - "&& 1" - [(set (match_dup 0) - (unspec: - [(match_dup 1) - (match_dup 2) - (match_dup 3)] - UNSPEC_UNSIGNED_PCMP))] -{ - operands[1] = force_reg (mode, operands[1]); - operands[0] = lowpart_subreg (mode, - operands[0], mode); -} + "vpcmpu\t{%3, %2, %1, %0|%0, %1, %2, %3}" [(set_attr "type" "ssecmp") (set_attr "length_immediate" "1") (set_attr "prefix" "evex") @@ -4096,16 +4055,21 @@ "#" "&& 1" [(set (match_dup 0) - (unspec: - [(match_dup 1) - (match_dup 2) - (match_dup 3)] - UNSPEC_UNSIGNED_PCMP)) - (set (match_dup 4) (match_dup 0))] -{ - operands[1] = force_reg (mode, operands[1]); - operands[0] = lowpart_subreg (mode, + (zero_extend:SWI248x + (unspec: + [(match_dup 1) + (match_dup 2) + (match_dup 3)] + UNSPEC_UNSIGNED_PCMP))) + (set (match_dup 4) (match_dup 5))] +{ + operands[5] = lowpart_subreg (mode, operands[0], mode); + if (SUBREG_P (operands[5])) + { + SUBREG_PROMOTED_VAR_P (operands[5]) = 1; + SUBREG_PROMOTED_SET (operands[5], 1); + } } [(set_attr "type" "ssecmp") (set_attr "length_immediate" "1") diff --git a/gcc/testsuite/gcc.target/i386/avx512bw-pr103750-1.c b/gcc/testsuite/gcc.target/i386/avx512bw-pr103750-1.c index b1165f069bb..e7d6183232b 100644 --- a/gcc/testsuite/gcc.target/i386/avx512bw-pr103750-1.c +++ b/gcc/testsuite/gcc.target/i386/avx512bw-pr103750-1.c @@ -1,8 +1,7 @@ /* PR target/103750 */ /* { dg-do compile } */ /* { dg-options "-O2 -mavx512bw -mavx512vl" } */ -/* { dg-final { scan-assembler-not "kmov" { xfail ia32 } } } */ -/* xfail need to be fixed. */ +/* { dg-final { scan-assembler-not "kmov" } } */ #include extern __m128i* pi128; diff --git a/gcc/testsuite/gcc.target/i386/avx512bw-pr103750-2.c b/gcc/testsuite/gcc.target/i386/avx512bw-pr103750-2.c index 7303f5403ba..3392e193222 100644 --- a/gcc/testsuite/gcc.target/i386/avx512bw-pr103750-2.c +++ b/gcc/testsuite/gcc.target/i386/avx512bw-pr103750-2.c @@ -1,8 +1,7 @@ /* PR target/103750 */ /* { dg-do compile } */ /* { dg-options "-O2 -mavx512dq -mavx512bw -mavx512vl" } */ -/* { dg-final { scan-assembler-not "kmov" { xfail ia32 } } } */ -/* xfail need to be fixed. */ +/* { dg-final { scan-assembler-not "kmov" } } */ #include extern __m128i* pi128; diff --git a/gcc/testsuite/gcc.target/i386/pr117159.c b/gcc/testsuite/gcc.target/i386/pr117159.c new file mode 100644 index 00000000000..b67d682ecef --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr117159.c @@ -0,0 +1,42 @@ +/* { dg-do run } */ +/* { dg-options "-Os -mavx512bw" } */ +/* { dg-require-effective-target avx512bw } */ + +typedef __attribute__((__vector_size__ (4))) unsigned char W; +typedef __attribute__((__vector_size__ (64))) int V; +typedef __attribute__((__vector_size__ (64))) long long Vq; + +W w; +V v; +Vq vq; + +static inline W +foo (short m) +{ + unsigned k = __builtin_ia32_pcmpgtq512_mask ((Vq) { }, vq, m); + W r = (W) k + w; + return r; +} + +static inline W +foo1 (short m) +{ + unsigned k = __builtin_ia32_pcmpgtd512_mask ((V) {1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1}, v, m); + W r = (W) k + w; + return r; +} + +int +main () +{ + if (!__builtin_cpu_supports ("avx512bw")) + return 0; + W y = foo1 (65535); + if (!y[0] || !y[1] || y[2] || y[3]) + __builtin_abort(); + W x = foo (65535); + if (x[0] || x[1] || x[2] || x[3]) + __builtin_abort(); + + return 0; +} commit 41377d0f4e791bcdd848e11eac172b8e81ecb6ec Author: Jeevitha Date: Mon Oct 21 04:01:46 2024 -0500 rs6000: Correct the function code for _AMO_LD_DEC_BOUNDED Corrected the function code for the Atomic Memory Operation "Fetch and Decrement Bounded", changing it from 0x1A to 0x1C. 2024-10-11 Jeevitha Palanisamy gcc/ * config/rs6000/amo.h (enum _AMO_LD): Correct the function code for _AMO_LD_DEC_BOUNDED. (cherry picked from commit 1a4c5643a5911d130dfab9a064222baeeb7f9be7) diff --git a/gcc/config/rs6000/amo.h b/gcc/config/rs6000/amo.h index ea4668e0547..47d19ee181c 100644 --- a/gcc/config/rs6000/amo.h +++ b/gcc/config/rs6000/amo.h @@ -46,7 +46,7 @@ enum _AMO_LD { _AMO_LD_CS_NE = 0x10, /* Compare and Swap Not Equal. */ _AMO_LD_INC_BOUNDED = 0x18, /* Fetch and Increment Bounded. */ _AMO_LD_INC_EQUAL = 0x19, /* Fetch and Increment Equal. */ - _AMO_LD_DEC_BOUNDED = 0x1A /* Fetch and Decrement Bounded. */ + _AMO_LD_DEC_BOUNDED = 0x1C /* Fetch and Decrement Bounded. */ }; /* Implementation of the simple LWAT/LDAT operations that take one register and commit 2db070cbf9684838643c7205bb67187369157375 Author: GCC Administrator Date: Tue Oct 22 00:23:29 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index bff57f4da6c..23eea1a512e 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,35 @@ +2024-10-21 Jeevitha + + Backported from master: + 2024-10-21 Jeevitha + + * config/rs6000/amo.h (enum _AMO_LD): Correct the function code for + _AMO_LD_DEC_BOUNDED. + +2024-10-21 liuhongt + + Backported from master: + 2024-10-21 liuhongt + + PR target/117159 + * config/i386/sse.md + (*_cmp3_zero_extend): + Change from define_insn_and_split to define_insn. + (*_cmp3_zero_extend): + Ditto. + (*_ucmp3_zero_extend): + Ditto. + (*_ucmp3_zero_extend): + Ditto. + (*_cmp3_zero_extend_2): + Split to the zero_extend pattern. + (*_cmp3_zero_extend_2): + Ditto. + (*_ucmp3_zero_extend_2): + Ditto. + (*_ucmp3_zero_extend_2): + Ditto. + 2024-10-16 Uros Bizjak Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 18b2d489abc..cf7fc14e4ea 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241021 +20241022 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 236aaba929f..52b10672a35 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,12 @@ +2024-10-21 liuhongt + + Backported from master: + 2024-10-21 liuhongt + + * gcc.target/i386/pr117159.c: New test. + * gcc.target/i386/avx512bw-pr103750-1.c: Remove xfail. + * gcc.target/i386/avx512bw-pr103750-2.c: Remove xfail. + 2024-10-16 Uros Bizjak Backported from master: commit 45bde60836d04cce4637b74ecadbb0aff90b832f Author: liuhongt Date: Tue Oct 22 11:24:23 2024 +0800 [GCC13/GCC12] Fix testcase. The optimization relies on other patterns which are only available at GCC14 and obove, so restore the xfail for GCC13/12 branch. gcc/testsuite/ChangeLog: * gcc.target/i386/avx512bw-pr103750-2.c: Add xfail for ia32. (cherry picked from commit 8b43518a01cbbbafe042b85a48fa09a32948380a) diff --git a/gcc/testsuite/gcc.target/i386/avx512bw-pr103750-2.c b/gcc/testsuite/gcc.target/i386/avx512bw-pr103750-2.c index 3392e193222..7303f5403ba 100644 --- a/gcc/testsuite/gcc.target/i386/avx512bw-pr103750-2.c +++ b/gcc/testsuite/gcc.target/i386/avx512bw-pr103750-2.c @@ -1,7 +1,8 @@ /* PR target/103750 */ /* { dg-do compile } */ /* { dg-options "-O2 -mavx512dq -mavx512bw -mavx512vl" } */ -/* { dg-final { scan-assembler-not "kmov" } } */ +/* { dg-final { scan-assembler-not "kmov" { xfail ia32 } } } */ +/* xfail need to be fixed. */ #include extern __m128i* pi128; commit fde31ceb464ca86a87e084d9793c185a4bc5a3ea Author: GCC Administrator Date: Wed Oct 23 00:21:13 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index cf7fc14e4ea..c961d1b391e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241022 +20241023 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 52b10672a35..8c4d59c30ce 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,10 @@ +2024-10-22 liuhongt + + Backported from master: + 2024-10-22 liuhongt + + * gcc.target/i386/avx512bw-pr103750-2.c: Add xfail for ia32. + 2024-10-21 liuhongt Backported from master: commit ac222968c320615b53443c5d3bce146b5556dc00 Author: GCC Administrator Date: Thu Oct 24 00:22:00 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c961d1b391e..7f899e9ca4b 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241023 +20241024 commit ab84a8a4b78990942e006e9f060dc2705f2c6d8f Author: liuhongt Date: Tue Oct 22 01:54:40 2024 -0700 Fix ICE due to isa mismatch for the builtins. gcc/ChangeLog: PR target/117240 * config/i386/i386-builtin.def: Add avx/avx512f to vaes ymm/zmm builtins. gcc/testsuite/ChangeLog: * gcc.target/i386/pr117240_avx.c: New test. * gcc.target/i386/pr117240_avx512f.c: New test. (cherry picked from commit 403e361d5aa620e77c9832578b2409a0fdd79d96) diff --git a/gcc/config/i386/i386-builtin.def b/gcc/config/i386/i386-builtin.def index 4a180cad467..64db1becca8 100644 --- a/gcc/config/i386/i386-builtin.def +++ b/gcc/config/i386/i386-builtin.def @@ -2751,18 +2751,18 @@ BDESC (0, OPTION_MASK_ISA2_AVX5124VNNIW, CODE_FOR_avx5124vnniw_vp4dpwssds_mask, BDESC (0, OPTION_MASK_ISA2_RDPID, CODE_FOR_rdpid, "__builtin_ia32_rdpid", IX86_BUILTIN_RDPID, UNKNOWN, (int) UNSIGNED_FTYPE_VOID) /* VAES. */ -BDESC (0, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesdec_v16qi, "__builtin_ia32_vaesdec_v16qi", IX86_BUILTIN_VAESDEC16, UNKNOWN, (int) V16QI_FTYPE_V16QI_V16QI) -BDESC (0, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesdec_v32qi, "__builtin_ia32_vaesdec_v32qi", IX86_BUILTIN_VAESDEC32, UNKNOWN, (int) V32QI_FTYPE_V32QI_V32QI) -BDESC (0, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesdec_v64qi, "__builtin_ia32_vaesdec_v64qi", IX86_BUILTIN_VAESDEC64, UNKNOWN, (int) V64QI_FTYPE_V64QI_V64QI) -BDESC (0, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesdeclast_v16qi, "__builtin_ia32_vaesdeclast_v16qi", IX86_BUILTIN_VAESDECLAST16, UNKNOWN, (int) V16QI_FTYPE_V16QI_V16QI) -BDESC (0, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesdeclast_v32qi, "__builtin_ia32_vaesdeclast_v32qi", IX86_BUILTIN_VAESDECLAST32, UNKNOWN, (int) V32QI_FTYPE_V32QI_V32QI) -BDESC (0, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesdeclast_v64qi, "__builtin_ia32_vaesdeclast_v64qi", IX86_BUILTIN_VAESDECLAST64, UNKNOWN, (int) V64QI_FTYPE_V64QI_V64QI) -BDESC (0, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesenc_v16qi, "__builtin_ia32_vaesenc_v16qi", IX86_BUILTIN_VAESENC16, UNKNOWN, (int) V16QI_FTYPE_V16QI_V16QI) -BDESC (0, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesenc_v32qi, "__builtin_ia32_vaesenc_v32qi", IX86_BUILTIN_VAESENC32, UNKNOWN, (int) V32QI_FTYPE_V32QI_V32QI) -BDESC (0, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesenc_v64qi, "__builtin_ia32_vaesenc_v64qi", IX86_BUILTIN_VAESENC64, UNKNOWN, (int) V64QI_FTYPE_V64QI_V64QI) -BDESC (0, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesenclast_v16qi, "__builtin_ia32_vaesenclast_v16qi", IX86_BUILTIN_VAESENCLAST16, UNKNOWN, (int) V16QI_FTYPE_V16QI_V16QI) -BDESC (0, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesenclast_v32qi, "__builtin_ia32_vaesenclast_v32qi", IX86_BUILTIN_VAESENCLAST32, UNKNOWN, (int) V32QI_FTYPE_V32QI_V32QI) -BDESC (0, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesenclast_v64qi, "__builtin_ia32_vaesenclast_v64qi", IX86_BUILTIN_VAESENCLAST64, UNKNOWN, (int) V64QI_FTYPE_V64QI_V64QI) +BDESC (OPTION_MASK_ISA_AVX512VL, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesdec_v16qi, "__builtin_ia32_vaesdec_v16qi", IX86_BUILTIN_VAESDEC16, UNKNOWN, (int) V16QI_FTYPE_V16QI_V16QI) +BDESC (OPTION_MASK_ISA_AVX, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesdec_v32qi, "__builtin_ia32_vaesdec_v32qi", IX86_BUILTIN_VAESDEC32, UNKNOWN, (int) V32QI_FTYPE_V32QI_V32QI) +BDESC (OPTION_MASK_ISA_AVX512F, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesdec_v64qi, "__builtin_ia32_vaesdec_v64qi", IX86_BUILTIN_VAESDEC64, UNKNOWN, (int) V64QI_FTYPE_V64QI_V64QI) +BDESC (OPTION_MASK_ISA_AVX512VL, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesdeclast_v16qi, "__builtin_ia32_vaesdeclast_v16qi", IX86_BUILTIN_VAESDECLAST16, UNKNOWN, (int) V16QI_FTYPE_V16QI_V16QI) +BDESC (OPTION_MASK_ISA_AVX, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesdeclast_v32qi, "__builtin_ia32_vaesdeclast_v32qi", IX86_BUILTIN_VAESDECLAST32, UNKNOWN, (int) V32QI_FTYPE_V32QI_V32QI) +BDESC (OPTION_MASK_ISA_AVX512F, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesdeclast_v64qi, "__builtin_ia32_vaesdeclast_v64qi", IX86_BUILTIN_VAESDECLAST64, UNKNOWN, (int) V64QI_FTYPE_V64QI_V64QI) +BDESC (OPTION_MASK_ISA_AVX512VL, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesenc_v16qi, "__builtin_ia32_vaesenc_v16qi", IX86_BUILTIN_VAESENC16, UNKNOWN, (int) V16QI_FTYPE_V16QI_V16QI) +BDESC (OPTION_MASK_ISA_AVX, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesenc_v32qi, "__builtin_ia32_vaesenc_v32qi", IX86_BUILTIN_VAESENC32, UNKNOWN, (int) V32QI_FTYPE_V32QI_V32QI) +BDESC (OPTION_MASK_ISA_AVX512F, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesenc_v64qi, "__builtin_ia32_vaesenc_v64qi", IX86_BUILTIN_VAESENC64, UNKNOWN, (int) V64QI_FTYPE_V64QI_V64QI) +BDESC (OPTION_MASK_ISA_AVX512VL, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesenclast_v16qi, "__builtin_ia32_vaesenclast_v16qi", IX86_BUILTIN_VAESENCLAST16, UNKNOWN, (int) V16QI_FTYPE_V16QI_V16QI) +BDESC (OPTION_MASK_ISA_AVX, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesenclast_v32qi, "__builtin_ia32_vaesenclast_v32qi", IX86_BUILTIN_VAESENCLAST32, UNKNOWN, (int) V32QI_FTYPE_V32QI_V32QI) +BDESC (OPTION_MASK_ISA_AVX512F, OPTION_MASK_ISA2_VAES, CODE_FOR_vaesenclast_v64qi, "__builtin_ia32_vaesenclast_v64qi", IX86_BUILTIN_VAESENCLAST64, UNKNOWN, (int) V64QI_FTYPE_V64QI_V64QI) /* BF16 */ BDESC (0, OPTION_MASK_ISA2_AVX512BF16, CODE_FOR_avx512f_cvtne2ps2bf16_v32hi, "__builtin_ia32_cvtne2ps2bf16_v32hi", IX86_BUILTIN_CVTNE2PS2HI16_V32HI, UNKNOWN, (int) V32HI_FTYPE_V16SF_V16SF) diff --git a/gcc/testsuite/gcc.target/i386/pr117240_avx.c b/gcc/testsuite/gcc.target/i386/pr117240_avx.c new file mode 100644 index 00000000000..24a97a9f74c --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr117240_avx.c @@ -0,0 +1,10 @@ +/* { dg-do compile } */ +/* { dg-options "-O2 -mvaes -mno-xsave -Wno-psabi -Wno-implicit-function-declaration" } */ + +typedef __attribute__((__vector_size__(32))) char V; + +V +foo(V v) +{ + return __builtin_ia32_vaesenc_v32qi(v, v);/* { dg-error "incompatible types when returning" } */ +} diff --git a/gcc/testsuite/gcc.target/i386/pr117240_avx512f.c b/gcc/testsuite/gcc.target/i386/pr117240_avx512f.c new file mode 100644 index 00000000000..1e7b5a88d7a --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr117240_avx512f.c @@ -0,0 +1,10 @@ +/* { dg-do compile } */ +/* { dg-options "-O2 -mvaes -mno-xsave -Wno-psabi -Wno-implicit-function-declaration" } */ + +typedef __attribute__((__vector_size__(64))) char V; + +V +foo(V v) +{ + return __builtin_ia32_vaesenc_v64qi(v, v);/* { dg-error "incompatible types when returning" } */ +} commit b5211c13cf2ca3576ae287b204640516de20ecff Author: Paul Thomas Date: Tue Jul 16 15:56:44 2024 +0100 Fortran: Simplify len_trim with array ref and fix mapping bug[PR84868]. 2024-07-16 Paul Thomas gcc/fortran PR fortran/84868 * simplify.cc (gfc_simplify_len_trim): If the argument is an element of a parameter array, simplify all the elements and build a new parameter array to hold the result, after checking that it doesn't already exist. * trans-expr.cc (gfc_get_interface_mapping_array) if a string length is available, use it for the typespec. (gfc_add_interface_mapping): Supply the se string length. gcc/testsuite/ PR fortran/84868 * gfortran.dg/pr84868.f90: New test. (cherry picked from commit 9f966b6a8ff0244dd6f8bf36d876799d5f9bbaee) diff --git a/gcc/fortran/simplify.cc b/gcc/fortran/simplify.cc index a10f79c4a93..b8935eb0118 100644 --- a/gcc/fortran/simplify.cc +++ b/gcc/fortran/simplify.cc @@ -4582,6 +4582,81 @@ gfc_simplify_len_trim (gfc_expr *e, gfc_expr *kind) if (k == -1) return &gfc_bad_expr; + /* If the expression is either an array element or section, an array + parameter must be built so that the reference can be applied. Constant + references should have already been simplified away. All other cases + can proceed to translation, where kind conversion will occur silently. */ + if (e->expr_type == EXPR_VARIABLE + && e->ts.type == BT_CHARACTER + && e->symtree->n.sym->attr.flavor == FL_PARAMETER + && e->ref && e->ref->type == REF_ARRAY + && e->ref->u.ar.type != AR_FULL + && e->symtree->n.sym->value) + { + char name[2*GFC_MAX_SYMBOL_LEN + 12]; + gfc_namespace *ns = e->symtree->n.sym->ns; + gfc_symtree *st; + gfc_expr *expr; + gfc_expr *p; + gfc_constructor *c; + int cnt = 0; + + sprintf (name, "_len_trim_%s_%s", e->symtree->n.sym->name, + ns->proc_name->name); + st = gfc_find_symtree (ns->sym_root, name); + if (st) + goto already_built; + + /* Recursively call this fcn to simplify the constructor elements. */ + expr = gfc_copy_expr (e->symtree->n.sym->value); + expr->ts.type = BT_INTEGER; + expr->ts.kind = k; + expr->ts.u.cl = NULL; + c = gfc_constructor_first (expr->value.constructor); + for (; c; c = gfc_constructor_next (c)) + { + if (c->iterator) + continue; + + if (c->expr && c->expr->ts.type == BT_CHARACTER) + { + p = gfc_simplify_len_trim (c->expr, kind); + if (p == NULL) + goto clean_up; + gfc_replace_expr (c->expr, p); + cnt++; + } + } + + if (cnt) + { + /* Build a new parameter to take the result. */ + st = gfc_new_symtree (&ns->sym_root, name); + st->n.sym = gfc_new_symbol (st->name, ns); + st->n.sym->value = expr; + st->n.sym->ts = expr->ts; + st->n.sym->attr.dimension = 1; + st->n.sym->attr.save = SAVE_IMPLICIT; + st->n.sym->attr.flavor = FL_PARAMETER; + st->n.sym->as = gfc_copy_array_spec (e->symtree->n.sym->as); + gfc_set_sym_referenced (st->n.sym); + st->n.sym->refs++; + gfc_commit_symbol (st->n.sym); + +already_built: + /* Build a return expression. */ + expr = gfc_copy_expr (e); + expr->ts = st->n.sym->ts; + expr->symtree = st; + gfc_expression_rank (expr); + return expr; + } + +clean_up: + gfc_free_expr (expr); + return NULL; + } + if (e->expr_type != EXPR_CONSTANT) return NULL; diff --git a/gcc/fortran/trans-expr.cc b/gcc/fortran/trans-expr.cc index e78a01003c9..54cf246fd0d 100644 --- a/gcc/fortran/trans-expr.cc +++ b/gcc/fortran/trans-expr.cc @@ -4326,12 +4326,15 @@ gfc_get_interface_mapping_charlen (gfc_interface_mapping * mapping, static tree gfc_get_interface_mapping_array (stmtblock_t * block, gfc_symbol * sym, - gfc_packed packed, tree data) + gfc_packed packed, tree data, tree len) { tree type; tree var; - type = gfc_typenode_for_spec (&sym->ts); + if (len != NULL_TREE && (TREE_CONSTANT (len) || VAR_P (len))) + type = gfc_get_character_type_len (sym->ts.kind, len); + else + type = gfc_typenode_for_spec (&sym->ts); type = gfc_get_nodesc_array_type (type, sym->as, packed, !sym->attr.target && !sym->attr.pointer && !sym->attr.proc_pointer); @@ -4478,7 +4481,8 @@ gfc_add_interface_mapping (gfc_interface_mapping * mapping, convert it to a boundless character type. */ else if (!sym->attr.dimension && sym->ts.type == BT_CHARACTER) { - tmp = gfc_get_character_type_len (sym->ts.kind, NULL); + se->string_length = gfc_evaluate_now (se->string_length, &se->pre); + tmp = gfc_get_character_type_len (sym->ts.kind, se->string_length); tmp = build_pointer_type (tmp); if (sym->attr.pointer) value = build_fold_indirect_ref_loc (input_location, @@ -4497,7 +4501,7 @@ gfc_add_interface_mapping (gfc_interface_mapping * mapping, /* For character(*), use the actual argument's descriptor. */ else if (sym->ts.type == BT_CHARACTER && !new_sym->ts.u.cl->length) value = build_fold_indirect_ref_loc (input_location, - se->expr); + se->expr); /* If the argument is an array descriptor, use it to determine information about the actual argument's shape. */ @@ -4511,7 +4515,8 @@ gfc_add_interface_mapping (gfc_interface_mapping * mapping, /* Create the replacement variable. */ tmp = gfc_conv_descriptor_data_get (desc); value = gfc_get_interface_mapping_array (&se->pre, sym, - PACKED_NO, tmp); + PACKED_NO, tmp, + se->string_length); /* Use DESC to work out the upper bounds, strides and offset. */ gfc_set_interface_mapping_bounds (&se->pre, TREE_TYPE (value), desc); @@ -4519,7 +4524,8 @@ gfc_add_interface_mapping (gfc_interface_mapping * mapping, else /* Otherwise we have a packed array. */ value = gfc_get_interface_mapping_array (&se->pre, sym, - PACKED_FULL, se->expr); + PACKED_FULL, se->expr, + se->string_length); new_sym->backend_decl = value; } diff --git a/gcc/testsuite/gfortran.dg/pr84868.f90 b/gcc/testsuite/gfortran.dg/pr84868.f90 new file mode 100644 index 00000000000..459a1c3c8b5 --- /dev/null +++ b/gcc/testsuite/gfortran.dg/pr84868.f90 @@ -0,0 +1,84 @@ +! { dg-do run } +! +! Test the fix for PR84868. Module 'orig' and the call to 'f_orig' is the +! original bug. The rest tests variants and the fix for a gimplifier ICE. +! +! Subroutine 'h' and calls to it were introduced to check the corrections +! needed to fix additional problems, noted in the review of the patch by +! Harald Anlauf +! +! Contributed by Gerhard Steinmetz +! +module orig + character(:), allocatable :: c + integer :: ans1(3,3), ans2(3), ans3(2) +contains + function f_orig(n) result(z) + character(2), parameter :: c(3) = ['x1', 'y ', 'z2'] + integer, intent(in) :: n + character(len_trim(c(n))) :: z + z = c(n) + end + function h(n) result(z) + integer, intent(in) :: n + character(2), parameter :: c(3,3) = & + reshape (['ab','c ','de','f ','gh','i ','jk','l ','mn'],[3,3]) + character(4), parameter :: chr(3) = ['ab ',' cd','e f '] + character(len_trim(c(n,n))) :: z + z = c(n,n) +! Make sure that full arrays are correctly scalarized both having been previously +! used with an array reference and not previously referenced. + ans1 = len_trim (c) + ans2 = len_trim (chr) +! Finally check a slightly more complicated array reference + ans3 = len_trim (c(1:n+1:2,n-1)) + end +end module orig + +module m + character(:), allocatable :: c +contains + function f(n, c) result(z) + character (2) :: c(:) + integer, intent(in) :: n + character(len_trim(c(n))) :: z + z = c(n) + end + subroutine foo (pc) + character(2) :: pc(:) + if (any ([(len (f(i, pc)), i = 1,3)] .ne. [2,1,2])) stop 1 + end +end +program p + use m + use orig + character (2) :: pc(3) = ['x1', 'y ', 'z2'] + integer :: i + + if (any ([(len (f_orig(i)), i = 1,3)] .ne. [2,1,2])) stop 2 ! ICE + + call foo (pc) + if (any ([(len (g(i, pc)), i = 1,3)] .ne. [2,1,2])) stop 3 + if (any ([(bar1(i), i = 1,3)] .ne. [2,1,2])) stop 4 + if (any ([(bar2(i), i = 1,3)] .ne. [2,1,2])) stop 5 + + if (h(2) .ne. 'gh') stop 6 + if (any (ans1 .ne. reshape ([2,1,2,1,2,1,2,1,2],[3,3]))) stop 7 + if (any (ans2 .ne. [2,4,3])) stop 8 + if (any (ans3 .ne. [2,2])) stop 9 +contains + function g(n, c) result(z) + character (2) :: c(:) + integer, intent(in) :: n + character(len_trim(c(n))) :: z + z = c(n) + end + integer function bar1 (i) + integer :: i + bar1 = len (f(i, pc)) ! ICE in is_gimple_min_invariant + end + integer function bar2 (i) + integer :: i + bar2 = len (g(i, pc)) + end +end commit 69a55cea0e34d9a29b51cc50ead55d9f137ef902 Author: GCC Administrator Date: Sat Oct 26 00:20:30 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 23eea1a512e..202388bf1bd 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2024-10-24 liuhongt + + Backported from master: + 2024-10-23 liuhongt + + PR target/117240 + * config/i386/i386-builtin.def: Add avx/avx512f to vaes + ymm/zmm builtins. + 2024-10-21 Jeevitha Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7f899e9ca4b..7125d77c104 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241024 +20241026 diff --git a/gcc/fortran/ChangeLog b/gcc/fortran/ChangeLog index cf1b77de38b..c1c2eccb9ab 100644 --- a/gcc/fortran/ChangeLog +++ b/gcc/fortran/ChangeLog @@ -1,3 +1,17 @@ +2024-10-25 Paul Thomas + + Backported from master: + 2024-07-16 Paul Thomas + + PR fortran/84868 + * simplify.cc (gfc_simplify_len_trim): If the argument is an + element of a parameter array, simplify all the elements and + build a new parameter array to hold the result, after checking + that it doesn't already exist. + * trans-expr.cc (gfc_get_interface_mapping_array) if a string + length is available, use it for the typespec. + (gfc_add_interface_mapping): Supply the se string length. + 2024-09-20 Harald Anlauf Backported from master: diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 8c4d59c30ce..89a7a8d5b1a 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,19 @@ +2024-10-25 Paul Thomas + + Backported from master: + 2024-07-16 Paul Thomas + + PR fortran/84868 + * gfortran.dg/pr84868.f90: New test. + +2024-10-24 liuhongt + + Backported from master: + 2024-10-23 liuhongt + + * gcc.target/i386/pr117240_avx.c: New test. + * gcc.target/i386/pr117240_avx512f.c: New test. + 2024-10-22 liuhongt Backported from master: commit 56b972db40894b85cd437a4c9b030843bad4c779 Author: GCC Administrator Date: Sun Oct 27 00:19:24 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7125d77c104..484ce9b3e42 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241026 +20241027 commit b2cd6ae7b3147ef34d7bbb16783d914e3657bbd3 Author: GCC Administrator Date: Mon Oct 28 00:19:06 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 484ce9b3e42..0dda5574ec7 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241027 +20241028 commit 7f13d0705d3b0e8f2f53aa381b4e4b6323682b88 Author: GCC Administrator Date: Tue Oct 29 00:20:10 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 0dda5574ec7..03377a986ad 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241028 +20241029 commit eeb72f26ea7e70baadf2e3b9e89e8f7055fec0a9 Author: Peter Bergner Date: Fri Aug 23 11:45:40 2024 -0500 rs6000: Fix PTImode handling in power8 swap optimization pass [PR116415] Our power8 swap optimization pass has some special handling for optimizing swaps of TImode variables. The test case reported in bugzilla uses a call to __atomic_compare_exchange, which introduces a variable of PTImode and that does not get the same treatment as TImode leading to wrong code generation. The simple fix is to treat PTImode identically to TImode. 2024-08-23 Peter Bergner gcc/ PR target/116415 * config/rs6000/rs6000.h (TI_OR_PTI_MODE): New define. * config/rs6000/rs6000-p8swap.cc (rs6000_analyze_swaps): Use it to handle PTImode identically to TImode. gcc/testsuite/ PR target/116415 * gcc.target/powerpc/pr116415.c: New test. (cherry picked from commit 6e68c3df1540c5bafbb47343698bf4e270333fdb) diff --git a/gcc/config/rs6000/rs6000-p8swap.cc b/gcc/config/rs6000/rs6000-p8swap.cc index 62f5ca5bff4..e97a32b4c23 100644 --- a/gcc/config/rs6000/rs6000-p8swap.cc +++ b/gcc/config/rs6000/rs6000-p8swap.cc @@ -2467,10 +2467,10 @@ rs6000_analyze_swaps (function *fun) mode = V4SImode; } - if (ALTIVEC_OR_VSX_VECTOR_MODE (mode) || mode == TImode) + if (ALTIVEC_OR_VSX_VECTOR_MODE (mode) || TI_OR_PTI_MODE (mode)) { insn_entry[uid].is_relevant = 1; - if (mode == TImode || mode == V1TImode + if (TI_OR_PTI_MODE (mode) || mode == V1TImode || FLOAT128_VECTOR_P (mode)) insn_entry[uid].is_128_int = 1; if (DF_REF_INSN_INFO (mention)) @@ -2495,10 +2495,10 @@ rs6000_analyze_swaps (function *fun) && ALTIVEC_OR_VSX_VECTOR_MODE (GET_MODE (SET_DEST (insn)))) mode = GET_MODE (SET_DEST (insn)); - if (ALTIVEC_OR_VSX_VECTOR_MODE (mode) || mode == TImode) + if (ALTIVEC_OR_VSX_VECTOR_MODE (mode) || TI_OR_PTI_MODE (mode)) { insn_entry[uid].is_relevant = 1; - if (mode == TImode || mode == V1TImode + if (TI_OR_PTI_MODE (mode) || mode == V1TImode || FLOAT128_VECTOR_P (mode)) insn_entry[uid].is_128_int = 1; if (DF_REF_INSN_INFO (mention)) diff --git a/gcc/config/rs6000/rs6000.h b/gcc/config/rs6000/rs6000.h index 5588a4bae02..48ba4df1da5 100644 --- a/gcc/config/rs6000/rs6000.h +++ b/gcc/config/rs6000/rs6000.h @@ -1051,6 +1051,8 @@ enum data_align { align_abi, align_opt, align_both }; (ALTIVEC_VECTOR_MODE (MODE) || VSX_VECTOR_MODE (MODE) \ || (MODE) == V2DImode || (MODE) == V1TImode) +#define TI_OR_PTI_MODE(mode) ((mode) == TImode || (mode) == PTImode) + /* Post-reload, we can't use any new AltiVec registers, as we already emitted the vrsave mask. */ diff --git a/gcc/testsuite/gcc.target/powerpc/pr116415.c b/gcc/testsuite/gcc.target/powerpc/pr116415.c new file mode 100644 index 00000000000..08cc282e2c2 --- /dev/null +++ b/gcc/testsuite/gcc.target/powerpc/pr116415.c @@ -0,0 +1,42 @@ +/* { dg-do run } */ +/* { dg-require-effective-target p8vector_hw } */ +/* { dg-require-effective-target int128 } */ +/* { dg-options "-O2 -mdejagnu-cpu=power8" } */ + +/* PR 116415: Verify our Power8 swap optimization pass doesn't incorrectly swap + PTImode values. They should be handled identically to TImode values. */ + +#include +#include +#include + +typedef union { + struct { + uint64_t a; + uint64_t b; + } t; + __uint128_t data; +} Value; +Value value, next; + +void +bug (Value *val, Value *nxt) +{ + for (;;) { + nxt->t.a = val->t.a + 1; + nxt->t.b = val->t.b + 2; + if (__atomic_compare_exchange (&val->data, &val->data, &nxt->data, + 0, __ATOMIC_SEQ_CST, __ATOMIC_ACQUIRE)) + break; + } +} + +int +main (void) +{ + bug (&value, &next); + printf ("%lu %lu\n", value.t.a, value.t.b); + if (value.t.a != 1 || value.t.b != 2) + abort (); + return 0; +} commit 0711e018b77eac34efc6d2e1e66cdf16e01b47c0 Author: Eric Botcazou Date: Tue Oct 29 21:40:34 2024 +0100 Fix miscompilation of function containing __builtin_unreachable This is a wrong-code generation on the SPARC for a function containing a call to __builtin_unreachable caused by the delay slot scheduling pass, and more specifically the find_end_label function which has these lines: /* Otherwise, see if there is a label at the end of the function. If there is, it must be that RETURN insns aren't needed, so that is our return label and we don't have to do anything else. */ The comment was correct 20 years ago but no longer is nowadays in the presence of RTL epilogues and calls to __builtin_unreachable, so the patch just removes the associated two lines of code: else if (LABEL_P (insn)) *plabel = as_a (insn); and otherwise contains just adjustments to the commentary. gcc/ PR rtl-optimization/117327 * reorg.cc (find_end_label): Do not return a dangling label at the end of the function and adjust commentary. gcc/testsuite/ * gcc.c-torture/execute/20241029-1.c: New test. diff --git a/gcc/reorg.cc b/gcc/reorg.cc index 7624f514906..df5d57fc0af 100644 --- a/gcc/reorg.cc +++ b/gcc/reorg.cc @@ -336,13 +336,14 @@ insn_sets_resource_p (rtx insn, struct resources *res, return resource_conflicts_p (&insn_sets, res); } -/* Find a label at the end of the function or before a RETURN. If there - is none, try to make one. If that fails, returns 0. +/* Find a label before a RETURN. If there is none, try to make one; if this + fails, return 0. KIND is either ret_rtx or simple_return_rtx, indicating + which type of RETURN we're looking for. - The property of such a label is that it is placed just before the - epilogue or a bare RETURN insn, so that another bare RETURN can be - turned into a jump to the label unconditionally. In particular, the - label cannot be placed before a RETURN insn with a filled delay slot. + The property of the label is that it is placed just before a bare RETURN + insn, so that another bare RETURN can be turned into a jump to the label + unconditionally. In particular, the label cannot be placed before a + RETURN insn with a filled delay slot. ??? There may be a problem with the current implementation. Suppose we start with a bare RETURN insn and call find_end_label. It may set @@ -353,9 +354,7 @@ insn_sets_resource_p (rtx insn, struct resources *res, Note that this is probably mitigated by the following observation: once function_return_label is made, it is very likely the target of a jump, so filling the delay slot of the RETURN will be much more - difficult. - KIND is either simple_return_rtx or ret_rtx, indicating which type of - return we're looking for. */ + difficult. */ static rtx_code_label * find_end_label (rtx kind) @@ -375,10 +374,7 @@ find_end_label (rtx kind) if (*plabel) return *plabel; - /* Otherwise, see if there is a label at the end of the function. If there - is, it must be that RETURN insns aren't needed, so that is our return - label and we don't have to do anything else. */ - + /* Otherwise, scan the insns backward from the end of the function. */ insn = get_last_insn (); while (NOTE_P (insn) || (NONJUMP_INSN_P (insn) @@ -386,9 +382,8 @@ find_end_label (rtx kind) || GET_CODE (PATTERN (insn)) == CLOBBER))) insn = PREV_INSN (insn); - /* When a target threads its epilogue we might already have a - suitable return insn. If so put a label before it for the - function_return_label. */ + /* First, see if there is a RETURN at the end of the function. If so, + put the label before it. */ if (BARRIER_P (insn) && JUMP_P (PREV_INSN (insn)) && PATTERN (PREV_INSN (insn)) == kind) @@ -397,8 +392,8 @@ find_end_label (rtx kind) rtx_code_label *label = gen_label_rtx (); LABEL_NUSES (label) = 0; - /* Put the label before any USE insns that may precede the RETURN - insn. */ + /* Put the label before any USE insns that may precede the + RETURN insn. */ while (GET_CODE (temp) == USE) temp = PREV_INSN (temp); @@ -406,15 +401,12 @@ find_end_label (rtx kind) *plabel = label; } - else if (LABEL_P (insn)) - *plabel = as_a (insn); + /* If the basic block reordering pass has moved the return insn to some + other place, try to locate it again and put the label there. */ else { rtx_code_label *label = gen_label_rtx (); LABEL_NUSES (label) = 0; - /* If the basic block reorder pass moves the return insn to - some other place try to locate it again and put our - function_return_label there. */ while (insn && ! (JUMP_P (insn) && (PATTERN (insn) == kind))) insn = PREV_INSN (insn); if (insn) diff --git a/gcc/testsuite/gcc.c-torture/execute/20241029-1.c b/gcc/testsuite/gcc.c-torture/execute/20241029-1.c new file mode 100644 index 00000000000..1090edd9c7c --- /dev/null +++ b/gcc/testsuite/gcc.c-torture/execute/20241029-1.c @@ -0,0 +1,23 @@ +/* PR rtl-optimization/117327 */ +/* Testcase by Brad Moody */ + +__attribute__((noinline)) +void foo(int *self, int *x) +{ + __builtin_puts ("foo\n"); + + if (x) { + while (1) { + ++*self; + if (*self == 6) break; + if (*self == 7) __builtin_unreachable(); + } + } +} + +int main (void) +{ + int y = 0; + foo (&y, 0); + return 0; +} commit 0bc4a352e25a4fb17dd1f6759e044ee02e230adb Author: GCC Administrator Date: Wed Oct 30 00:20:15 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 202388bf1bd..be7a3250ef9 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,19 @@ +2024-10-29 Eric Botcazou + + PR rtl-optimization/117327 + * reorg.cc (find_end_label): Do not return a dangling label at the + end of the function and adjust commentary. + +2024-10-29 Peter Bergner + + Backported from master: + 2024-08-23 Peter Bergner + + PR target/116415 + * config/rs6000/rs6000.h (TI_OR_PTI_MODE): New define. + * config/rs6000/rs6000-p8swap.cc (rs6000_analyze_swaps): Use it to + handle PTImode identically to TImode. + 2024-10-24 liuhongt Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 03377a986ad..d334905e040 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241029 +20241030 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 89a7a8d5b1a..fb53c99dbda 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,15 @@ +2024-10-29 Eric Botcazou + + * gcc.c-torture/execute/20241029-1.c: New test. + +2024-10-29 Peter Bergner + + Backported from master: + 2024-08-23 Peter Bergner + + PR target/116415 + * gcc.target/powerpc/pr116415.c: New test. + 2024-10-25 Paul Thomas Backported from master: commit d0a932fb53ccdf5155d1111b90632901c55446b8 Author: liuhongt Date: Tue Oct 29 02:09:39 2024 -0700 Fix ICE due to subreg:us_truncate. Force_operand issues an ICE when input is (subreg:DI (us_truncate:V8QI)), it's probably because it's an invalid rtx, So refine backend patterns for that. gcc/ChangeLog: PR target/117318 * config/i386/sse.md (*avx512vl_v2div2qi2_mask_store_1): Rename to .. (avx512vl_v2div2qi2_mask_store_1): .. this. (avx512vl_v2div2qi2_mask_store_2): Change to define_expand. (*avx512vl_v4qi2_mask_store_1): Rename to .. (avx512vl_v4qi2_mask_store_1): .. this. (avx512vl_v4qi2_mask_store_2): Change to define_expand. (*avx512vl_v8qi2_mask_store_1): Rename to .. (avx512vl_v8qi2_mask_store_1): .. this. (avx512vl_v8qi2_mask_store_2): Change to define_expand. (*avx512vl_v4hi2_mask_store_1): Rename to .. (avx512vl_v4hi2_mask_store_1): .. this. (avx512vl_v4hi2_mask_store_2): Change to define_expand. (*avx512vl_v2div2hi2_mask_store_1): Rename to .. (avx512vl_v2div2hi2_mask_store_1): .. this. (avx512vl_v2div2hi2_mask_store_2): Change to define_expand. (*avx512vl_v2div2si2_mask_store_1): Rename to .. (avx512vl_v2div2si2_mask_store_1): .. this. (avx512vl_v2div2si2_mask_store_2): Change to define_expand. (*avx512f_v8div16qi2_mask_store_1): Rename to .. (avx512f_v8div16qi2_mask_store_1): .. this. (avx512f_v8div16qi2_mask_store_2): Change to define_expand. gcc/testsuite/ChangeLog: * gcc.target/i386/pr117318.c: New test. (cherry picked from commit bc0eeccf27a084461a2d5661e23468350acb43da) diff --git a/gcc/config/i386/sse.md b/gcc/config/i386/sse.md index c94c8eceb33..3ad96f321a6 100644 --- a/gcc/config/i386/sse.md +++ b/gcc/config/i386/sse.md @@ -13853,7 +13853,7 @@ (set_attr "prefix" "evex") (set_attr "mode" "TI")]) -(define_insn "*avx512vl_v2div2qi2_mask_store_1" +(define_insn "avx512vl_v2div2qi2_mask_store_1" [(set (match_operand:V2QI 0 "memory_operand" "=m") (vec_merge:V2QI (any_truncate:V2QI @@ -13867,28 +13867,19 @@ (set_attr "prefix" "evex") (set_attr "mode" "TI")]) -(define_insn_and_split "avx512vl_v2div2qi2_mask_store_2" - [(set (match_operand:HI 0 "memory_operand") - (subreg:HI - (vec_merge:V2QI - (any_truncate:V2QI - (match_operand:V2DI 1 "register_operand")) - (vec_select:V2QI - (subreg:V4QI - (vec_concat:V2HI - (match_dup 0) - (const_int 0)) 0) - (parallel [(const_int 0) (const_int 1)])) - (match_operand:QI 2 "register_operand")) 0))] - "TARGET_AVX512VL && ix86_pre_reload_split ()" - "#" - "&& 1" - [(set (match_dup 0) - (vec_merge:V2QI - (any_truncate:V2QI (match_dup 1)) - (match_dup 0) - (match_dup 2)))] - "operands[0] = adjust_address_nv (operands[0], V2QImode, 0);") +(define_expand "avx512vl_v2div2qi2_mask_store_2" + [(match_operand:HI 0 "memory_operand") + (any_truncate:V2QI + (match_operand:V2DI 1 "register_operand")) + (match_operand:QI 2 "register_operand")] + "TARGET_AVX512VL" +{ + operands[0] = adjust_address_nv (operands[0], V2QImode, 0); + emit_insn (gen_avx512vl_v2div2qi2_mask_store_1 (operands[0], + operands[1], + operands[2])); + DONE; +}) (define_insn "*avx512vl_v4qi2_store_1" [(set (match_operand:V4QI 0 "memory_operand" "=m") @@ -13957,7 +13948,7 @@ (set_attr "prefix" "evex") (set_attr "mode" "TI")]) -(define_insn "*avx512vl_v4qi2_mask_store_1" +(define_insn "avx512vl_v4qi2_mask_store_1" [(set (match_operand:V4QI 0 "memory_operand" "=m") (vec_merge:V4QI (any_truncate:V4QI @@ -13971,29 +13962,19 @@ (set_attr "prefix" "evex") (set_attr "mode" "TI")]) -(define_insn_and_split "avx512vl_v4qi2_mask_store_2" - [(set (match_operand:SI 0 "memory_operand") - (subreg:SI - (vec_merge:V4QI - (any_truncate:V4QI - (match_operand:VI4_128_8_256 1 "register_operand")) - (vec_select:V4QI - (subreg:V8QI - (vec_concat:V2SI - (match_dup 0) - (const_int 0)) 0) - (parallel [(const_int 0) (const_int 1) - (const_int 2) (const_int 3)])) - (match_operand:QI 2 "register_operand")) 0))] - "TARGET_AVX512VL && ix86_pre_reload_split ()" - "#" - "&& 1" - [(set (match_dup 0) - (vec_merge:V4QI - (any_truncate:V4QI (match_dup 1)) - (match_dup 0) - (match_dup 2)))] - "operands[0] = adjust_address_nv (operands[0], V4QImode, 0);") +(define_expand "avx512vl_v4qi2_mask_store_2" + [(match_operand:SI 0 "memory_operand") + (any_truncate:V4QI + (match_operand:VI4_128_8_256 1 "register_operand")) + (match_operand:QI 2 "register_operand")] + "TARGET_AVX512VL" +{ + operands[0] = adjust_address_nv (operands[0], V4QImode, 0); + emit_insn (gen_avx512vl_v4qi2_mask_store_1 (operands[0], + operands[1], + operands[2])); + DONE; +}) (define_mode_iterator VI2_128_BW_4_256 [(V8HI "TARGET_AVX512BW") V8SI]) @@ -14065,7 +14046,7 @@ (set_attr "prefix" "evex") (set_attr "mode" "TI")]) -(define_insn "*avx512vl_v8qi2_mask_store_1" +(define_insn "avx512vl_v8qi2_mask_store_1" [(set (match_operand:V8QI 0 "memory_operand" "=m") (vec_merge:V8QI (any_truncate:V8QI @@ -14079,31 +14060,19 @@ (set_attr "prefix" "evex") (set_attr "mode" "TI")]) -(define_insn_and_split "avx512vl_v8qi2_mask_store_2" - [(set (match_operand:DI 0 "memory_operand") - (subreg:DI - (vec_merge:V8QI - (any_truncate:V8QI - (match_operand:VI2_128_BW_4_256 1 "register_operand")) - (vec_select:V8QI - (subreg:V16QI - (vec_concat:V2DI - (match_dup 0) - (const_int 0)) 0) - (parallel [(const_int 0) (const_int 1) - (const_int 2) (const_int 3) - (const_int 4) (const_int 5) - (const_int 6) (const_int 7)])) - (match_operand:QI 2 "register_operand")) 0))] - "TARGET_AVX512VL && ix86_pre_reload_split ()" - "#" - "&& 1" - [(set (match_dup 0) - (vec_merge:V8QI - (any_truncate:V8QI (match_dup 1)) - (match_dup 0) - (match_dup 2)))] - "operands[0] = adjust_address_nv (operands[0], V8QImode, 0);") +(define_expand "avx512vl_v8qi2_mask_store_2" + [(match_operand:DI 0 "memory_operand") + (any_truncate:V8QI + (match_operand:VI2_128_BW_4_256 1 "register_operand")) + (match_operand:QI 2 "register_operand")] + "TARGET_AVX512VL" +{ + operands[0] = adjust_address_nv (operands[0], V8QImode, 0); + emit_insn (gen_avx512vl_v8qi2_mask_store_1 (operands[0], + operands[1], + operands[2])); + DONE; +}) (define_mode_iterator PMOV_SRC_MODE_4 [V4DI V2DI V4SI]) (define_mode_attr pmov_dst_4 @@ -14225,7 +14194,7 @@ (set_attr "prefix" "evex") (set_attr "mode" "TI")]) -(define_insn "*avx512vl_v4hi2_mask_store_1" +(define_insn "avx512vl_v4hi2_mask_store_1" [(set (match_operand:V4HI 0 "memory_operand" "=m") (vec_merge:V4HI (any_truncate:V4HI @@ -14243,30 +14212,19 @@ (set_attr "prefix" "evex") (set_attr "mode" "TI")]) -(define_insn_and_split "avx512vl_v4hi2_mask_store_2" - [(set (match_operand:DI 0 "memory_operand") - (subreg:DI - (vec_merge:V4HI - (any_truncate:V4HI - (match_operand:VI4_128_8_256 1 "register_operand")) - (vec_select:V4HI - (subreg:V8HI - (vec_concat:V2DI - (match_dup 0) - (const_int 0)) 0) - (parallel [(const_int 0) (const_int 1) - (const_int 2) (const_int 3)])) - (match_operand:QI 2 "register_operand")) 0))] - "TARGET_AVX512VL && ix86_pre_reload_split ()" - "#" - "&& 1" - [(set (match_dup 0) - (vec_merge:V4HI - (any_truncate:V4HI (match_dup 1)) - (match_dup 0) - (match_dup 2)))] - "operands[0] = adjust_address_nv (operands[0], V4HImode, 0);") - +(define_expand "avx512vl_v4hi2_mask_store_2" + [(match_operand:DI 0 "memory_operand") + (any_truncate:V4HI + (match_operand:VI4_128_8_256 1 "register_operand")) + (match_operand:QI 2 "register_operand")] + "TARGET_AVX512VL" +{ + operands[0] = adjust_address_nv (operands[0], V4HImode, 0); + emit_insn (gen_avx512vl_v4hi2_mask_store_1 (operands[0], + operands[1], + operands[2])); + DONE; +}) (define_insn "*avx512vl_v2div2hi2_store_1" [(set (match_operand:V2HI 0 "memory_operand" "=m") @@ -14327,7 +14285,7 @@ (set_attr "prefix" "evex") (set_attr "mode" "TI")]) -(define_insn "*avx512vl_v2div2hi2_mask_store_1" +(define_insn "avx512vl_v2div2hi2_mask_store_1" [(set (match_operand:V2HI 0 "memory_operand" "=m") (vec_merge:V2HI (any_truncate:V2HI @@ -14341,28 +14299,19 @@ (set_attr "prefix" "evex") (set_attr "mode" "TI")]) -(define_insn_and_split "avx512vl_v2div2hi2_mask_store_2" - [(set (match_operand:SI 0 "memory_operand") - (subreg:SI - (vec_merge:V2HI - (any_truncate:V2HI - (match_operand:V2DI 1 "register_operand")) - (vec_select:V2HI - (subreg:V4HI - (vec_concat:V2SI - (match_dup 0) - (const_int 0)) 0) - (parallel [(const_int 0) (const_int 1)])) - (match_operand:QI 2 "register_operand")) 0))] - "TARGET_AVX512VL && ix86_pre_reload_split ()" - "#" - "&& 1" - [(set (match_dup 0) - (vec_merge:V2HI - (any_truncate:V2HI (match_dup 1)) - (match_dup 0) - (match_dup 2)))] - "operands[0] = adjust_address_nv (operands[0], V2HImode, 0);") +(define_expand "avx512vl_v2div2hi2_mask_store_2" + [(match_operand:SI 0 "memory_operand") + (any_truncate:V2HI + (match_operand:V2DI 1 "register_operand")) + (match_operand:QI 2 "register_operand")] + "TARGET_AVX512VL" +{ + operands[0] = adjust_address_nv (operands[0], V2HImode, 0); + emit_insn (gen_avx512vl_v2div2hi2_mask_store_1 (operands[0], + operands[1], + operands[2])); + DONE; +}) (define_expand "truncv2div2si2" [(set (match_operand:V2SI 0 "register_operand") @@ -14470,7 +14419,7 @@ (set_attr "prefix" "evex") (set_attr "mode" "TI")]) -(define_insn "*avx512vl_v2div2si2_mask_store_1" +(define_insn "avx512vl_v2div2si2_mask_store_1" [(set (match_operand:V2SI 0 "memory_operand" "=m") (vec_merge:V2SI (any_truncate:V2SI @@ -14484,28 +14433,19 @@ (set_attr "prefix" "evex") (set_attr "mode" "TI")]) -(define_insn_and_split "avx512vl_v2div2si2_mask_store_2" - [(set (match_operand:DI 0 "memory_operand") - (subreg:DI - (vec_merge:V2SI - (any_truncate:V2SI - (match_operand:V2DI 1 "register_operand")) - (vec_select:V2SI - (subreg:V4SI - (vec_concat:V2DI - (match_dup 0) - (const_int 0)) 0) - (parallel [(const_int 0) (const_int 1)])) - (match_operand:QI 2 "register_operand")) 0))] - "TARGET_AVX512VL && ix86_pre_reload_split ()" - "#" - "&& 1" - [(set (match_dup 0) - (vec_merge:V2SI - (any_truncate:V2SI (match_dup 1)) - (match_dup 0) - (match_dup 2)))] - "operands[0] = adjust_address_nv (operands[0], V2SImode, 0);") +(define_expand "avx512vl_v2div2si2_mask_store_2" + [(match_operand:DI 0 "memory_operand") + (any_truncate:V2SI + (match_operand:V2DI 1 "register_operand")) + (match_operand:QI 2 "register_operand")] + "TARGET_AVX512VL" +{ + operands[0] = adjust_address_nv (operands[0], V2SImode, 0); + emit_insn (gen_avx512vl_v2div2si2_mask_store_1 (operands[0], + operands[1], + operands[2])); + DONE; +}) (define_expand "truncv8div8qi2" [(set (match_operand:V8QI 0 "register_operand") @@ -14604,7 +14544,7 @@ (set_attr "prefix" "evex") (set_attr "mode" "TI")]) -(define_insn "*avx512f_v8div16qi2_mask_store_1" +(define_insn "avx512f_v8div16qi2_mask_store_1" [(set (match_operand:V8QI 0 "memory_operand" "=m") (vec_merge:V8QI (any_truncate:V8QI @@ -14618,31 +14558,19 @@ (set_attr "prefix" "evex") (set_attr "mode" "TI")]) -(define_insn_and_split "avx512f_v8div16qi2_mask_store_2" - [(set (match_operand:DI 0 "memory_operand") - (subreg:DI - (vec_merge:V8QI - (any_truncate:V8QI - (match_operand:V8DI 1 "register_operand")) - (vec_select:V8QI - (subreg:V16QI - (vec_concat:V2DI - (match_dup 0) - (const_int 0)) 0) - (parallel [(const_int 0) (const_int 1) - (const_int 2) (const_int 3) - (const_int 4) (const_int 5) - (const_int 6) (const_int 7)])) - (match_operand:QI 2 "register_operand")) 0))] - "TARGET_AVX512F && ix86_pre_reload_split ()" - "#" - "&& 1" - [(set (match_dup 0) - (vec_merge:V8QI - (any_truncate:V8QI (match_dup 1)) - (match_dup 0) - (match_dup 2)))] - "operands[0] = adjust_address_nv (operands[0], V8QImode, 0);") +(define_expand "avx512f_v8div16qi2_mask_store_2" + [(match_operand:DI 0 "memory_operand") + (any_truncate:V8QI + (match_operand:V8DI 1 "register_operand")) + (match_operand:QI 2 "register_operand")] + "TARGET_AVX512F" +{ + operands[0] = adjust_address_nv (operands[0], V8QImode, 0); + emit_insn (gen_avx512f_v8div16qi2_mask_store_1 (operands[0], + operands[1], + operands[2])); + DONE; +}) ;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;; ;; diff --git a/gcc/testsuite/gcc.target/i386/pr117318.c b/gcc/testsuite/gcc.target/i386/pr117318.c new file mode 100644 index 00000000000..3d316ad04cf --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr117318.c @@ -0,0 +1,12 @@ +/* { dg-do compile } */ +/* { dg-options "-mavx512f -O" } */ + +typedef __attribute__((__vector_size__ (64))) long long V; +unsigned long long x; + +unsigned long long +foo() +{ + __builtin_ia32_pmovusqb512mem_mask (&x, (V){8000000000000000}, 255); + return x; +} commit 49971d35ef30c3f2cde2636f9e0c072ca016db8a Author: GCC Administrator Date: Thu Oct 31 00:19:44 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index be7a3250ef9..78593f16aa8 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,39 @@ +2024-10-30 liuhongt + + Backported from master: + 2024-10-30 liuhongt + + PR target/117318 + * config/i386/sse.md (*avx512vl_v2div2qi2_mask_store_1): + Rename to .. + (avx512vl_v2div2qi2_mask_store_1): .. this. + (avx512vl_v2div2qi2_mask_store_2): Change to + define_expand. + (*avx512vl_v4qi2_mask_store_1): Rename to .. + (avx512vl_v4qi2_mask_store_1): .. this. + (avx512vl_v4qi2_mask_store_2): Change to + define_expand. + (*avx512vl_v8qi2_mask_store_1): Rename to .. + (avx512vl_v8qi2_mask_store_1): .. this. + (avx512vl_v8qi2_mask_store_2): Change to + define_expand. + (*avx512vl_v4hi2_mask_store_1): Rename to .. + (avx512vl_v4hi2_mask_store_1): .. this. + (avx512vl_v4hi2_mask_store_2): Change to + define_expand. + (*avx512vl_v2div2hi2_mask_store_1): Rename to .. + (avx512vl_v2div2hi2_mask_store_1): .. this. + (avx512vl_v2div2hi2_mask_store_2): Change to + define_expand. + (*avx512vl_v2div2si2_mask_store_1): Rename to .. + (avx512vl_v2div2si2_mask_store_1): .. this. + (avx512vl_v2div2si2_mask_store_2): Change to + define_expand. + (*avx512f_v8div16qi2_mask_store_1): Rename to .. + (avx512f_v8div16qi2_mask_store_1): .. this. + (avx512f_v8div16qi2_mask_store_2): Change to + define_expand. + 2024-10-29 Eric Botcazou PR rtl-optimization/117327 diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d334905e040..bf9a843b29c 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241030 +20241031 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index fb53c99dbda..a504ec57e18 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,10 @@ +2024-10-30 liuhongt + + Backported from master: + 2024-10-30 liuhongt + + * gcc.target/i386/pr117318.c: New test. + 2024-10-29 Eric Botcazou * gcc.c-torture/execute/20241029-1.c: New test. commit 94334ab83b02ac3a43f231174a3cced9920d1550 Author: GCC Administrator Date: Fri Nov 1 00:22:13 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index bf9a843b29c..c47a8c6250a 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241031 +20241101 commit 5210bf4d494d6ea60722193c7eb97827e73f5295 Author: Paul Thomas Date: Fri Oct 25 17:59:03 2024 +0100 Fortran: Fix ICE with structure constructor in data statement [PR79685] 2024-10-25 Paul Thomas gcc/fortran PR fortran/79685 * decl.cc (match_data_constant): Find the symtree instead of the symbol so the use renamed symbols are found. Pass this and the derived type to gfc_match_structure_constructor. * match.h: Update prototype of gfc_match_structure_contructor. * primary.cc (gfc_match_structure_constructor): Remove call to gfc_get_ha_sym_tree and use caller supplied symtree instead. gcc/testsuite/ PR fortran/79685 * gfortran.dg/use_rename_13.f90: New test. (cherry picked from commit 6cb1da72cac166bd3b005c0430557b68b9761da5) diff --git a/gcc/fortran/decl.cc b/gcc/fortran/decl.cc index d98f98d7ec6..8c101b0daf7 100644 --- a/gcc/fortran/decl.cc +++ b/gcc/fortran/decl.cc @@ -376,6 +376,7 @@ match_data_constant (gfc_expr **result) gfc_expr *expr; match m; locus old_loc; + gfc_symtree *symtree; m = gfc_match_literal_constant (&expr, 1); if (m == MATCH_YES) @@ -435,9 +436,11 @@ match_data_constant (gfc_expr **result) if (m != MATCH_YES) return m; - if (gfc_find_symbol (name, NULL, 1, &sym)) + if (gfc_find_sym_tree (name, NULL, 1, &symtree)) return MATCH_ERROR; + sym = symtree->n.sym; + if (sym && sym->attr.generic) dt_sym = gfc_find_dt_in_generic (sym); @@ -451,7 +454,7 @@ match_data_constant (gfc_expr **result) return MATCH_ERROR; } else if (dt_sym && gfc_fl_struct (dt_sym->attr.flavor)) - return gfc_match_structure_constructor (dt_sym, result); + return gfc_match_structure_constructor (dt_sym, symtree, result); /* Check to see if the value is an initialization array expression. */ if (sym->value->expr_type == EXPR_ARRAY) diff --git a/gcc/fortran/match.h b/gcc/fortran/match.h index 495c93e0b5c..d2273356858 100644 --- a/gcc/fortran/match.h +++ b/gcc/fortran/match.h @@ -296,7 +296,7 @@ match gfc_match_bind_c_stmt (void); match gfc_match_bind_c (gfc_symbol *, bool); /* primary.cc. */ -match gfc_match_structure_constructor (gfc_symbol *, gfc_expr **); +match gfc_match_structure_constructor (gfc_symbol *, gfc_symtree *, gfc_expr **); match gfc_match_variable (gfc_expr **, int); match gfc_match_equiv_variable (gfc_expr **); match gfc_match_actual_arglist (int, gfc_actual_arglist **, bool = false); diff --git a/gcc/fortran/primary.cc b/gcc/fortran/primary.cc index 4f8bd129ee9..328e92b5aef 100644 --- a/gcc/fortran/primary.cc +++ b/gcc/fortran/primary.cc @@ -3380,18 +3380,16 @@ gfc_convert_to_structure_constructor (gfc_expr *e, gfc_symbol *sym, gfc_expr **c match -gfc_match_structure_constructor (gfc_symbol *sym, gfc_expr **result) +gfc_match_structure_constructor (gfc_symbol *sym, gfc_symtree *symtree, + gfc_expr **result) { match m; gfc_expr *e; - gfc_symtree *symtree; bool t = true; - gfc_get_ha_sym_tree (sym->name, &symtree); - e = gfc_get_expr (); - e->symtree = symtree; e->expr_type = EXPR_FUNCTION; + e->symtree = symtree; e->where = gfc_current_locus; gcc_assert (gfc_fl_struct (sym->attr.flavor) diff --git a/gcc/testsuite/gfortran.dg/use_rename_13.f90 b/gcc/testsuite/gfortran.dg/use_rename_13.f90 new file mode 100644 index 00000000000..97f26f42f76 --- /dev/null +++ b/gcc/testsuite/gfortran.dg/use_rename_13.f90 @@ -0,0 +1,37 @@ +! { dg-do compile } +! +! Test the fix for pr79685, which failed as in the comments below. +! +! Contributed by Juergen Reuter +! +module omega_color + implicit none + + type omega_color_factor + integer :: i + end type + + type(omega_color_factor), parameter :: op = omega_color_factor (199) + +end module + +module foo + use omega_color, ocf => omega_color_factor, ocfp => op + implicit none + + type(ocf) :: table_color_factors1 = ocf(42) + type(ocf) :: table_color_factors2 + type(ocf) :: table_color_factors3 (2) + type(ocf) :: table_color_factors4 + data table_color_factors2 / ocf(99) / ! This failed in gfc_match_structure_constructor. + data table_color_factors3 / ocf(1), ocf(2) / ! ditto. + data table_color_factors4 / ocfp / +end module + + use foo + if (table_color_factors1%i .ne. 42) stop 1 + if (table_color_factors2%i .ne. 99) stop 2 + if (any (table_color_factors3%i .ne. [1,2])) stop 3 + if (table_color_factors4%i .ne. 199) stop 4 +end + commit 11af155ac0cb44001e1824f7aa0eeb993833ba7c Author: GCC Administrator Date: Sat Nov 2 00:22:10 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c47a8c6250a..0ffe82a8a48 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241101 +20241102 diff --git a/gcc/fortran/ChangeLog b/gcc/fortran/ChangeLog index c1c2eccb9ab..7ebf8cb86b8 100644 --- a/gcc/fortran/ChangeLog +++ b/gcc/fortran/ChangeLog @@ -1,3 +1,16 @@ +2024-11-01 Paul Thomas + + Backported from master: + 2024-10-25 Paul Thomas + + PR fortran/79685 + * decl.cc (match_data_constant): Find the symtree instead of + the symbol so the use renamed symbols are found. Pass this and + the derived type to gfc_match_structure_constructor. + * match.h: Update prototype of gfc_match_structure_contructor. + * primary.cc (gfc_match_structure_constructor): Remove call to + gfc_get_ha_sym_tree and use caller supplied symtree instead. + 2024-10-25 Paul Thomas Backported from master: diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index a504ec57e18..7910065fb88 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-11-01 Paul Thomas + + Backported from master: + 2024-10-25 Paul Thomas + + PR fortran/79685 + * gfortran.dg/use_rename_13.f90: New test. + 2024-10-30 liuhongt Backported from master: commit fd1fb86d6268d655df31f00fbb1f05da8450881e Author: GCC Administrator Date: Sun Nov 3 00:19:49 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 0ffe82a8a48..e99b9656d62 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241102 +20241103 commit ea81f223274eb9bb55d0684298360032addc83e1 Author: GCC Administrator Date: Mon Nov 4 00:20:25 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e99b9656d62..9a7f1aceb27 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241103 +20241104 commit d2061812ea9ff4dd862f32aed81b127300b29ee2 Author: Eric Botcazou Date: Fri Aug 16 16:03:30 2024 +0200 ada: Fix internal error on concatenation of discriminant-dependent component This only occurs with optimization enabled, but the expanded code is always wrong because it reuses the formal parameter of an initialization procedure associated with a discriminant (a discriminal in GNAT parlance) outside of the initialization procedure. gcc/ada/ * checks.adb (Selected_Length_Checks.Get_E_Length): For a component of a record with discriminants and if the expression is a selected component, try to build an actual subtype from its prefix instead of from the discriminal. diff --git a/gcc/ada/checks.adb b/gcc/ada/checks.adb index e1a1b0cfec1..449aa6578ac 100644 --- a/gcc/ada/checks.adb +++ b/gcc/ada/checks.adb @@ -9936,7 +9936,15 @@ package body Checks is if Ekind (Scope (E)) = E_Record_Type and then Has_Discriminants (Scope (E)) then - N := Build_Discriminal_Subtype_Of_Component (E); + -- If the expression is a selected component, in other words, + -- has a prefix, then build an actual subtype from the prefix. + -- Otherwise, build an actual subtype from the discriminal. + + if Nkind (Expr) = N_Selected_Component then + N := Build_Actual_Subtype_Of_Component (E, Expr); + else + N := Build_Discriminal_Subtype_Of_Component (E); + end if; if Present (N) then Insert_Action (Expr, N); commit db2611e392a28f6b4969d8d469ad47f307367ab0 Author: Eric Botcazou Date: Mon Nov 4 11:15:15 2024 +0100 Add regression test This is for the latest fix made to Selected_Length_Checks in Checks. gcc/testsuite * gnat.dg/specs/array7.ads: New test. diff --git a/gcc/testsuite/gnat.dg/specs/array7.ads b/gcc/testsuite/gnat.dg/specs/array7.ads new file mode 100644 index 00000000000..643ec3bb9ce --- /dev/null +++ b/gcc/testsuite/gnat.dg/specs/array7.ads @@ -0,0 +1,14 @@ +-- { dg-do compile } +-- { dg-options "-O" } + +package Array7 is + + type I is interface; + + type Rec (Name_Len : Natural) is new I with record + Input : String (1 .. Name_Len); + end record; + + function Image (R : Rec) return String is ("I" & String (R.Input)); + +end Array7; commit 78d3156d53567cb2794c3df21dc0d914abe43543 Author: Andrew MacLeod Date: Mon Nov 4 10:02:35 2024 -0500 Don't call invert on VARYING. When all cases go to one label and resul in a VARYING value, we can't invert that value to remove all values from the default case. Simply check for this case and set the default to UNDEFINED. PR tree-optimization/117398 gcc/ * gimple-range-edge.cc (gimple_outgoing_range::calc_switch_ranges): Check for VARYING and don't call invert () on it. gcc/testsuite/ * gcc.dg/pr117398.c: New. diff --git a/gcc/gimple-range-edge.cc b/gcc/gimple-range-edge.cc index 6caa07c8f02..a0cc1383a88 100644 --- a/gcc/gimple-range-edge.cc +++ b/gcc/gimple-range-edge.cc @@ -145,8 +145,14 @@ gimple_outgoing_range::calc_switch_ranges (gswitch *sw) // Remove the case range from the default case. int_range_max def_range (low, high); range_cast (def_range, type); - def_range.invert (); - default_range.intersect (def_range); + // If all possible values are taken, set default_range to UNDEFINED. + if (def_range.varying_p ()) + default_range.set_undefined (); + else + { + def_range.invert (); + default_range.intersect (def_range); + } // Create/union this case with anything on else on the edge. int_range_max case_range (low, high); diff --git a/gcc/testsuite/gcc.dg/pr117398.c b/gcc/testsuite/gcc.dg/pr117398.c new file mode 100644 index 00000000000..c43f2a3ed6b --- /dev/null +++ b/gcc/testsuite/gcc.dg/pr117398.c @@ -0,0 +1,17 @@ +/* { dg-do compile } */ +/* { dg-options "-O2" } */ + +int a; +void c(void); +int d(_Bool b) { + switch (b+0) { + case 0: + break; + case 1: + break; + default: + c(); + } + if (b) + return a; +} commit a48d9c7fa3d23bf0c252040a47c484a44f1ab9ca Author: GCC Administrator Date: Tue Nov 5 00:21:42 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 78593f16aa8..6e3dbc9f013 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,9 @@ +2024-11-04 Andrew MacLeod + + PR tree-optimization/117398 + * gimple-range-edge.cc (gimple_outgoing_range::calc_switch_ranges): + Check for VARYING and don't call invert () on it. + 2024-10-30 liuhongt Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 9a7f1aceb27..30cad970f74 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241104 +20241105 diff --git a/gcc/ada/ChangeLog b/gcc/ada/ChangeLog index a33b1c09495..3f47bc1fab8 100644 --- a/gcc/ada/ChangeLog +++ b/gcc/ada/ChangeLog @@ -1,3 +1,10 @@ +2024-11-04 Eric Botcazou + + * checks.adb (Selected_Length_Checks.Get_E_Length): For a + component of a record with discriminants and if the expression is + a selected component, try to build an actual subtype from its + prefix instead of from the discriminal. + 2024-10-14 Steve Baird PR ada/114593 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 7910065fb88..64d6c718ac9 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,12 @@ +2024-11-04 Andrew MacLeod + + PR tree-optimization/117398 + * gcc.dg/pr117398.c: New. + +2024-11-04 Eric Botcazou + + * gnat.dg/specs/array7.ads: New test. + 2024-11-01 Paul Thomas Backported from master: commit 5fdd38d576c20d5a337b5c7d14108981d0751434 Author: Simon Martin Date: Tue Nov 5 10:07:42 2024 +0100 c++: Defer -fstrong-eval-order processing to template instantiation time [PR117158] Since r10-3793-g1a37b6d9a7e57c, we ICE upon the following valid code with -std=c++17 and above === cut here === struct Base { unsigned int *intarray; }; template struct Sub : public Base { bool Get(int i) { return (Base::intarray[++i] == 0); } }; === cut here === The problem is that from c++17 on, we use -fstrong-eval-order and need to wrap the array access expression into a SAVE_EXPR. We do so at template declaration time, and end up calling contains_placeholder_p with a SCOPE_REF, that it does not handle well. This patch fixes this by deferring the wrapping into SAVE_EXPR to instantiation time for templates, when the SCOPE_REF will have been turned into a COMPONENT_REF. PR c++/117158 gcc/cp/ChangeLog: * typeck.cc (cp_build_array_ref): Only wrap array expression into a SAVE_EXPR at template instantiation time. gcc/testsuite/ChangeLog: * g++.dg/cpp1z/eval-order13.C: New test. * g++.dg/parse/crash77.C: New test. (cherry picked from commit b1d92aeb8583c8d1491c97703680c5fb88ed1fe4) diff --git a/gcc/cp/typeck.cc b/gcc/cp/typeck.cc index ff8f0ef0083..df819701d4a 100644 --- a/gcc/cp/typeck.cc +++ b/gcc/cp/typeck.cc @@ -3919,7 +3919,8 @@ cp_build_array_ref (location_t loc, tree array, tree idx, tree ar = cp_default_conversion (array, complain); tree ind = cp_default_conversion (idx, complain); - if (!first && flag_strong_eval_order == 2 && TREE_SIDE_EFFECTS (ind)) + if (!processing_template_decl + && !first && flag_strong_eval_order == 2 && TREE_SIDE_EFFECTS (ind)) ar = first = save_expr (ar); /* Put the integer in IND to simplify error checking. */ diff --git a/gcc/testsuite/g++.dg/cpp1z/eval-order13.C b/gcc/testsuite/g++.dg/cpp1z/eval-order13.C new file mode 100644 index 00000000000..6e8ebeb3096 --- /dev/null +++ b/gcc/testsuite/g++.dg/cpp1z/eval-order13.C @@ -0,0 +1,29 @@ +// PR c++/117158 - Similar to eval-order7.C, only with templates. +// { dg-do run { target c++11 } } +// { dg-options "-fstrong-eval-order" } + +int a[4] = { 1, 2, 3, 4 }; +int b[4] = { 5, 6, 7, 8 }; + +struct Base { + int *intarray; +}; + +template +struct Sub : public Base { + int Get(int i) { + Base::intarray = a; + int r = Base::intarray[(Base::intarray = b, i)]; + if (Base::intarray != b) + __builtin_abort (); + return r; + } +}; + +int +main () +{ + Sub s; + if (s.Get (3) != 4) + __builtin_abort (); +} diff --git a/gcc/testsuite/g++.dg/parse/crash77.C b/gcc/testsuite/g++.dg/parse/crash77.C new file mode 100644 index 00000000000..729362eb599 --- /dev/null +++ b/gcc/testsuite/g++.dg/parse/crash77.C @@ -0,0 +1,13 @@ +// PR c++/117158 +// { dg-do "compile" } + +struct Base { + unsigned int *intarray; +}; + +template +struct Sub : public Base { + bool Get(int i) { + return (Base::intarray[++i] == 0); + } +}; commit 7cf09486969d314c885f837697e60802f5e124e4 Author: GCC Administrator Date: Wed Nov 6 00:21:21 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 30cad970f74..e7bef9850f2 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241105 +20241106 diff --git a/gcc/cp/ChangeLog b/gcc/cp/ChangeLog index b2f55e13d7d..bb52869e0ad 100644 --- a/gcc/cp/ChangeLog +++ b/gcc/cp/ChangeLog @@ -1,3 +1,12 @@ +2024-11-05 Simon Martin + + Backported from master: + 2024-11-05 Simon Martin + + PR c++/117158 + * typeck.cc (cp_build_array_ref): Only wrap array expression + into a SAVE_EXPR at template instantiation time. + 2024-09-17 Marek Polacek Backported from master: diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 64d6c718ac9..1c0f3289eff 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,12 @@ +2024-11-05 Simon Martin + + Backported from master: + 2024-11-05 Simon Martin + + PR c++/117158 + * g++.dg/cpp1z/eval-order13.C: New test. + * g++.dg/parse/crash77.C: New test. + 2024-11-04 Andrew MacLeod PR tree-optimization/117398 commit 6235b5806ceffaf7152cbaaa7bda3aa8e5adafe7 Author: GCC Administrator Date: Thu Nov 7 00:20:45 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e7bef9850f2..6d8a676e763 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241106 +20241107 commit 9336065d8c1a60fa02cfab3ca8806ce80b112fc5 Author: GCC Administrator Date: Sat Nov 9 16:05:14 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 6d8a676e763..7731736334d 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241107 +20241109 commit dfea9abfb4260798c99354173c38c06a35ce5a7a Author: GCC Administrator Date: Sun Nov 10 00:19:35 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7731736334d..ac00abcae8e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241109 +20241110 commit 6f01cab3240b3e0c7c7b0ea3de820a60f3bb812a Author: GCC Administrator Date: Mon Nov 11 00:19:12 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index ac00abcae8e..31b8ebf7ebe 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241110 +20241111 commit a1500fe4a12dd74bd3a331839e34baa438338bbd Author: GCC Administrator Date: Tue Nov 12 00:21:19 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 31b8ebf7ebe..353869c3cba 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241111 +20241112 commit 571a5103a1e195dbfac3447065ccfb6c4e69aa2f Author: John David Anglin Date: Tue Nov 12 14:26:08 2024 -0500 hppa: Fix decrement_and_branch_until_zero constraint The third alternative for argument 4 needs to be an early clobber constraint. Noticed testing LRA. 2024-11-12 John David Anglin gcc/ChangeLog: * config/pa/pa.md (decrement_and_branch_until_zero): Fix constraint. diff --git a/gcc/config/pa/pa.md b/gcc/config/pa/pa.md index 1d97cb6f486..63335c2480c 100644 --- a/gcc/config/pa/pa.md +++ b/gcc/config/pa/pa.md @@ -9451,7 +9451,7 @@ add,l %2,%3,%3\;bv,n %%r0(%3)" (pc))) (set (match_dup 0) (plus:SI (match_dup 0) (match_dup 1))) - (clobber (match_scratch:SI 4 "=X,r,r"))] + (clobber (match_scratch:SI 4 "=X,r,&r"))] "" "* return pa_output_dbra (operands, insn, which_alternative); " ;; Do not expect to understand this the first time through. commit 3fd7c034003e754431158936c6e71bedece0b96c Author: GCC Administrator Date: Wed Nov 13 00:20:50 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 6e3dbc9f013..4b3f299c4f6 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,8 @@ +2024-11-12 John David Anglin + + * config/pa/pa.md (decrement_and_branch_until_zero): Fix + constraint. + 2024-11-04 Andrew MacLeod PR tree-optimization/117398 diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 353869c3cba..8a225a1f705 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241112 +20241113 commit e41fdca8a290c4d72b1972af8cdfd1dd60af31df Author: Hu, Lin1 Date: Wed Nov 6 15:42:13 2024 +0800 i386: Zero extend 32-bit address to 64-bit with option -mx32 -maddress-mode=long. [PR 117418] -maddress-mode=long let Pmode = DI_mode, so zero extend 32-bit address to 64-bit and uses a 64-bit register as a pointer for avoid raise an ICE. gcc/ChangeLog: PR target/117418 * config/i386/i386-expand.cc (ix86_expand_builtin): Convert pointer's mode according to Pmode. gcc/testsuite/ChangeLog: PR target/117418 * gcc.target/i386/pr117418-1.c: New test. (cherry picked from commit 2272cd2508f1854c880082f792de15e76ec09a99) diff --git a/gcc/config/i386/i386-expand.cc b/gcc/config/i386/i386-expand.cc index 909c11e4195..5c8d9c556af 100644 --- a/gcc/config/i386/i386-expand.cc +++ b/gcc/config/i386/i386-expand.cc @@ -12747,6 +12747,9 @@ ix86_expand_builtin (tree exp, rtx target, rtx subtarget, op1 = expand_normal (arg1); op2 = expand_normal (arg2); + if (GET_MODE (op1) != Pmode) + op1 = convert_to_mode (Pmode, op1, 1); + if (!address_operand (op2, VOIDmode)) { op2 = convert_memory_address (Pmode, op2); @@ -12782,6 +12785,9 @@ ix86_expand_builtin (tree exp, rtx target, rtx subtarget, emit_label (ok_label); emit_insn (gen_rtx_SET (target, pat)); + if (GET_MODE (op0) != Pmode) + op0 = convert_to_mode (Pmode, op0, 1); + for (i = 0; i < 8; i++) { op = gen_rtx_MEM (V2DImode, @@ -12806,6 +12812,9 @@ ix86_expand_builtin (tree exp, rtx target, rtx subtarget, if (!REG_P (op0)) op0 = copy_to_mode_reg (SImode, op0); + if (GET_MODE (op2) != Pmode) + op2 = convert_to_mode (Pmode, op2, 1); + op = gen_rtx_REG (V2DImode, GET_SSE_REGNO (0)); emit_move_insn (op, op1); @@ -12843,6 +12852,9 @@ ix86_expand_builtin (tree exp, rtx target, rtx subtarget, if (!REG_P (op0)) op0 = copy_to_mode_reg (SImode, op0); + if (GET_MODE (op3) != Pmode) + op3 = convert_to_mode (Pmode, op3, 1); + /* Force to use xmm0, xmm1 for keylow, keyhi*/ op = gen_rtx_REG (V2DImode, GET_SSE_REGNO (0)); emit_move_insn (op, op1); diff --git a/gcc/testsuite/gcc.target/i386/pr117418-1.c b/gcc/testsuite/gcc.target/i386/pr117418-1.c new file mode 100644 index 00000000000..4839b139b79 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr117418-1.c @@ -0,0 +1,24 @@ +/* PR target/117418 */ +/* { dg-do compile { target { ! ia32 } } } */ +/* { dg-options "-maddress-mode=long -mwidekl -mx32" } */ +/* { dg-require-effective-target maybe_x32 } */ +/* { dg-final { scan-assembler-times "aesdec128kl" 1 } } */ +/* { dg-final { scan-assembler-times "aesdec256kl" 1 } } */ +/* { dg-final { scan-assembler-times "aesenc128kl" 1 } } */ +/* { dg-final { scan-assembler-times "aesenc256kl" 1 } } */ +/* { dg-final { scan-assembler-times "encodekey128" 1 } } */ +/* { dg-final { scan-assembler-times "encodekey256" 1 } } */ + +typedef __attribute__((__vector_size__(16))) long long V; +V a; + +void +foo() +{ + __builtin_ia32_aesdec128kl_u8 (&a, a, &a); + __builtin_ia32_aesdec256kl_u8 (&a, a, &a); + __builtin_ia32_aesenc128kl_u8 (&a, a, &a); + __builtin_ia32_aesenc256kl_u8 (&a, a, &a); + __builtin_ia32_encodekey128_u32 (0, a, &a); + __builtin_ia32_encodekey256_u32 (0, a, a, &a); +} commit 424bd4c3d14473c18850e7366804e761fc0b03de Author: John David Anglin Date: Wed Nov 13 09:40:42 2024 -0500 hppa: Remove inner `fix:SF/DF` from fixed-point patterns 2024-11-13 John David Anglin gcc/ChangeLog: PR target/117525 * config/pa/pa.md (fix_truncsfsi2): Remove inner `fix:SF`. (fix_truncdfsi2, fix_truncsfdi2, fix_truncdfdi2, fixuns_truncsfsi2, fixuns_truncdfsi2, fixuns_truncsfdi2, fixuns_truncdfdi2): Likewise. diff --git a/gcc/config/pa/pa.md b/gcc/config/pa/pa.md index 63335c2480c..2b1ee3d32a3 100644 --- a/gcc/config/pa/pa.md +++ b/gcc/config/pa/pa.md @@ -4908,7 +4908,7 @@ (define_insn "fix_truncsfsi2" [(set (match_operand:SI 0 "register_operand" "=f") - (fix:SI (fix:SF (match_operand:SF 1 "register_operand" "f"))))] + (fix:SI (match_operand:SF 1 "register_operand" "f")))] "! TARGET_SOFT_FLOAT" "{fcnvfxt,sgl,sgl|fcnv,t,sgl,w} %1,%0" [(set_attr "type" "fpalu") @@ -4916,7 +4916,7 @@ (define_insn "fix_truncdfsi2" [(set (match_operand:SI 0 "register_operand" "=f") - (fix:SI (fix:DF (match_operand:DF 1 "register_operand" "f"))))] + (fix:SI (match_operand:DF 1 "register_operand" "f")))] "! TARGET_SOFT_FLOAT" "{fcnvfxt,dbl,sgl|fcnv,t,dbl,w} %1,%0" [(set_attr "type" "fpalu") @@ -4924,7 +4924,7 @@ (define_insn "fix_truncsfdi2" [(set (match_operand:DI 0 "register_operand" "=f") - (fix:DI (fix:SF (match_operand:SF 1 "register_operand" "f"))))] + (fix:DI (match_operand:SF 1 "register_operand" "f")))] "TARGET_PA_11 && ! TARGET_SOFT_FLOAT" "{fcnvfxt,sgl,dbl|fcnv,t,sgl,dw} %1,%0" [(set_attr "type" "fpalu") @@ -4932,7 +4932,7 @@ (define_insn "fix_truncdfdi2" [(set (match_operand:DI 0 "register_operand" "=f") - (fix:DI (fix:DF (match_operand:DF 1 "register_operand" "f"))))] + (fix:DI (match_operand:DF 1 "register_operand" "f")))] "TARGET_PA_11 && ! TARGET_SOFT_FLOAT" "{fcnvfxt,dbl,dbl|fcnv,t,dbl,dw} %1,%0" [(set_attr "type" "fpalu") @@ -4972,7 +4972,7 @@ (define_insn "fixuns_truncsfsi2" [(set (match_operand:SI 0 "register_operand" "=f") - (unsigned_fix:SI (fix:SF (match_operand:SF 1 "register_operand" "f"))))] + (unsigned_fix:SI (match_operand:SF 1 "register_operand" "f")))] "! TARGET_SOFT_FLOAT && TARGET_PA_20" "fcnv,t,sgl,uw %1,%0" [(set_attr "type" "fpalu") @@ -4980,7 +4980,7 @@ (define_insn "fixuns_truncdfsi2" [(set (match_operand:SI 0 "register_operand" "=f") - (unsigned_fix:SI (fix:DF (match_operand:DF 1 "register_operand" "f"))))] + (unsigned_fix:SI (match_operand:DF 1 "register_operand" "f")))] "! TARGET_SOFT_FLOAT && TARGET_PA_20" "fcnv,t,dbl,uw %1,%0" [(set_attr "type" "fpalu") @@ -4988,7 +4988,7 @@ (define_insn "fixuns_truncsfdi2" [(set (match_operand:DI 0 "register_operand" "=f") - (unsigned_fix:DI (fix:SF (match_operand:SF 1 "register_operand" "f"))))] + (unsigned_fix:DI (match_operand:SF 1 "register_operand" "f")))] "! TARGET_SOFT_FLOAT && TARGET_PA_20" "fcnv,t,sgl,udw %1,%0" [(set_attr "type" "fpalu") @@ -4996,7 +4996,7 @@ (define_insn "fixuns_truncdfdi2" [(set (match_operand:DI 0 "register_operand" "=f") - (unsigned_fix:DI (fix:DF (match_operand:DF 1 "register_operand" "f"))))] + (unsigned_fix:DI (match_operand:DF 1 "register_operand" "f")))] "! TARGET_SOFT_FLOAT && TARGET_PA_20" "fcnv,t,dbl,udw %1,%0" [(set_attr "type" "fpalu") commit 169f70a693cf35dfee6086351bdda0621be2b832 Author: GCC Administrator Date: Thu Nov 14 17:21:45 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 4b3f299c4f6..cc7f6cb5b89 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,20 @@ +2024-11-13 John David Anglin + + PR target/117525 + * config/pa/pa.md (fix_truncsfsi2): Remove inner `fix:SF`. + (fix_truncdfsi2, fix_truncsfdi2, fix_truncdfdi2, + fixuns_truncsfsi2, fixuns_truncdfsi2, fixuns_truncsfdi2, + fixuns_truncdfdi2): Likewise. + +2024-11-13 Hu, Lin1 + + Backported from master: + 2024-11-13 Hu, Lin1 + + PR target/117418 + * config/i386/i386-expand.cc (ix86_expand_builtin): Convert + pointer's mode according to Pmode. + 2024-11-12 John David Anglin * config/pa/pa.md (decrement_and_branch_until_zero): Fix diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 8a225a1f705..97867f47cfd 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241113 +20241114 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 1c0f3289eff..3ecf3c20def 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-11-13 Hu, Lin1 + + Backported from master: + 2024-11-13 Hu, Lin1 + + PR target/117418 + * gcc.target/i386/pr117418-1.c: New test. + 2024-11-05 Simon Martin Backported from master: commit d01a31607a1a54710f7719228e211ca709734f48 Author: GCC Administrator Date: Fri Nov 15 00:20:23 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 97867f47cfd..1bcb5dd5f30 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241114 +20241115 commit 178b89f76ec179169af206be5aca6674c7ad4473 Author: GCC Administrator Date: Sat Nov 16 00:20:38 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 1bcb5dd5f30..af4dbe513ab 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241115 +20241116 commit d0bb7a14ff99148386d7d5845f3aa4993b8113fe Author: GCC Administrator Date: Sun Nov 17 00:24:56 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index af4dbe513ab..b57dfce74e2 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241116 +20241117 commit c6646b5a74e4f56a6e16fe69683c2a529b51741e Author: GCC Administrator Date: Mon Nov 18 00:21:34 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index b57dfce74e2..7d15800c874 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241117 +20241118 commit a2725b4ed6f248e1239aace322a78ee4d45db110 Author: Georg-Johann Lay Date: Mon Nov 18 18:12:38 2024 +0100 AVR: target/117659 - Fix wrong code for u24 << 16. gcc/ PR target/117659 * config/avr/avr.cc (avr_out_ashlpsi3) [case 16]: Use %A1 as input (instead of bogus %A0). (cherry picked from commit bba27015f2815a8fa6fae46a29a70644e868341c) diff --git a/gcc/config/avr/avr.cc b/gcc/config/avr/avr.cc index f355146f992..c0f5242c0b2 100644 --- a/gcc/config/avr/avr.cc +++ b/gcc/config/avr/avr.cc @@ -7048,7 +7048,7 @@ avr_out_ashlpsi3 (rtx_insn *insn, rtx *op, int *plen) int reg1 = REGNO (op[1]); if (reg0 + 2 != reg1) - avr_asm_len ("mov %C0,%A0", op, plen, 1); + avr_asm_len ("mov %C0,%A1", op, plen, 1); return avr_asm_len ("clr %B0" CR_TAB "clr %A0", op, plen, 2); commit 2bca9b85f45542795cb8a9ee2dd9561e9cbffea8 Author: GCC Administrator Date: Tue Nov 19 00:20:33 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index cc7f6cb5b89..fa57b96a725 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2024-11-18 Georg-Johann Lay + + Backported from master: + 2024-11-18 Georg-Johann Lay + + PR target/117659 + * config/avr/avr.cc (avr_out_ashlpsi3) [case 16]: Use %A1 as + input (instead of bogus %A0). + 2024-11-13 John David Anglin PR target/117525 diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7d15800c874..6d3c9325db8 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241118 +20241119 commit 540c0c7c424a43c1d99dd22f6db020cc0cd6eaea Author: Uros Bizjak Date: Mon Nov 18 22:38:46 2024 +0100 i386: Enable *rsqrtsf2_sse without TARGET_SSE_MATH [PR117357] __builtin_ia32_rsqrtsf2 expander generates UNSPEC_RSQRT insn pattern also when TARGET_SSE_MATH is not set. Enable *rsqrtsf2_sse without TARGET_SSE_MATH to avoid ICE with unrecognizable insn. PR target/117357 gcc/ChangeLog: * config/i386/i386.md (*rsqrtsf2_sse): Also enable for !TARGET_SSE_MATH. gcc/testsuite/ChangeLog: * gcc.target/i386/pr117357.c: New test. (cherry picked from commit 344356f781ddb7bf0abb11edf9bdd13f6802dea8) diff --git a/gcc/config/i386/i386.md b/gcc/config/i386/i386.md index b04edb9201b..4142d7fec4d 100644 --- a/gcc/config/i386/i386.md +++ b/gcc/config/i386/i386.md @@ -18008,7 +18008,7 @@ [(set (match_operand:SF 0 "register_operand" "=x,x,x") (unspec:SF [(match_operand:SF 1 "nonimmediate_operand" "0,x,m")] UNSPEC_RSQRT))] - "TARGET_SSE && TARGET_SSE_MATH" + "TARGET_SSE" "@ %vrsqrtss\t{%d1, %0|%0, %d1} %vrsqrtss\t{%d1, %0|%0, %d1} diff --git a/gcc/testsuite/gcc.target/i386/pr117357.c b/gcc/testsuite/gcc.target/i386/pr117357.c new file mode 100644 index 00000000000..7a27691a977 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr117357.c @@ -0,0 +1,7 @@ +/* { dg-do compile } */ +/* { dg-options "-msse -mfpmath=387" } */ + +float foo (float f) +{ + return __builtin_ia32_rsqrtf (f); +} commit 61ce58768ad56ae1c60d8d6639dd91903e3dcdaa Author: GCC Administrator Date: Wed Nov 20 00:21:13 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index fa57b96a725..1e0c794232c 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2024-11-19 Uros Bizjak + + Backported from master: + 2024-11-18 Uros Bizjak + + PR target/117357 + * config/i386/i386.md (*rsqrtsf2_sse): + Also enable for !TARGET_SSE_MATH. + 2024-11-18 Georg-Johann Lay Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 6d3c9325db8..837c3849695 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241119 +20241120 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 3ecf3c20def..33f9fea5b97 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-11-19 Uros Bizjak + + Backported from master: + 2024-11-18 Uros Bizjak + + PR target/117357 + * gcc.target/i386/pr117357.c: New test. + 2024-11-13 Hu, Lin1 Backported from master: commit 63d679a1410ccea447a0c4961446187092332285 Author: GCC Administrator Date: Thu Nov 21 00:22:28 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 837c3849695..5116abdc66f 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241120 +20241121 commit f5d64b323fce27997ba7267948e96e3fbd29cd44 Author: GCC Administrator Date: Fri Nov 22 00:22:04 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 5116abdc66f..e1e2985b305 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241121 +20241122 commit 5ed7b592d869d34c7b8702255f634b39fcc46b02 Author: GCC Administrator Date: Sat Nov 23 00:24:30 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e1e2985b305..f69d3317a0a 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241122 +20241123 commit 7f04fddec3bc33b6fb418f3980995b9b7697e6b1 Author: Georg-Johann Lay Date: Sat Nov 23 12:51:32 2024 +0100 AVR: target/117744 - Fix asm for partial clobber of address reg, gcc/ PR target/117744 * config/avr/avr.cc (out_movqi_r_mr): Fix code when a load only partially clobbers an address register due to changing the address register temporally to accommodate for faked addressing modes. (cherry picked from commit ee8e6784876aa050d2e01f54d1da4acf758b635a) diff --git a/gcc/config/avr/avr.cc b/gcc/config/avr/avr.cc index c0f5242c0b2..5d63c7153d6 100644 --- a/gcc/config/avr/avr.cc +++ b/gcc/config/avr/avr.cc @@ -4241,11 +4241,11 @@ avr_out_movqi_r_mr_reg_disp_tiny (rtx_insn *insn, rtx op[], int *plen) rtx base2 = all_regs_rtx[1 ^ REGNO (dest)]; if (!reg_unused_after (insn, base2)) - avr_asm_len ("mov __tmp_reg__,%0" , &base2, plen, 1); + avr_asm_len ("mov __tmp_reg__,%0", &base2, plen, 1); avr_asm_len (TINY_ADIW (%I1, %J1, %o1) CR_TAB "ld %0,%b1", op, plen, 3); if (!reg_unused_after (insn, base2)) - avr_asm_len ("mov %0,__tmp_reg__" , &base2, plen, 1); + avr_asm_len ("mov %0,__tmp_reg__", &base2, plen, 1); } return ""; @@ -4272,40 +4272,66 @@ out_movqi_r_mr (rtx_insn *insn, rtx op[], int *plen) { /* memory access by reg+disp */ - int disp = INTVAL (XEXP (x, 1)); - if (AVR_TINY) return avr_out_movqi_r_mr_reg_disp_tiny (insn, op, plen); + if (plen) + *plen = 0; + + int disp = INTVAL (XEXP (x, 1)); + rtx base = XEXP (x, 0); + rtx base2 = all_regs_rtx[1 ^ REGNO (dest)]; + bool partial_clobber = (reg_overlap_mentioned_p (dest, base) + && ! reg_unused_after (insn, base2)); + if (disp - GET_MODE_SIZE (GET_MODE (src)) >= 63) - { - if (REGNO (XEXP (x, 0)) != REG_Y) - fatal_insn ("incorrect insn:",insn); + { + // PR117744: The base register overlaps dest and is + // only partially clobbered. + if (partial_clobber) + avr_asm_len ("mov __tmp_reg__,%0", &base2, plen, 1); - if (disp <= 63 + MAX_LD_OFFSET (GET_MODE (src))) - return avr_asm_len ("adiw r28,%o1-63" CR_TAB - "ldd %0,Y+63" CR_TAB - "sbiw r28,%o1-63", op, plen, -3); + if (REGNO (XEXP (x, 0)) != REG_Y) + fatal_insn ("incorrect insn:",insn); - return avr_asm_len ("subi r28,lo8(-%o1)" CR_TAB - "sbci r29,hi8(-%o1)" CR_TAB - "ld %0,Y" CR_TAB - "subi r28,lo8(%o1)" CR_TAB - "sbci r29,hi8(%o1)", op, plen, -5); - } + if (disp <= 63 + MAX_LD_OFFSET (GET_MODE (src))) + avr_asm_len ("adiw r28,%o1-63" CR_TAB + "ldd %0,Y+63" CR_TAB + "sbiw r28,%o1-63", op, plen, 3); + else + avr_asm_len ("subi r28,lo8(-%o1)" CR_TAB + "sbci r29,hi8(-%o1)" CR_TAB + "ld %0,Y" CR_TAB + "subi r28,lo8(%o1)" CR_TAB + "sbci r29,hi8(%o1)", op, plen, 5); + + if (partial_clobber) + avr_asm_len ("mov __tmp_reg__,%0", &base2, plen, 1); + + return ""; + } else if (REGNO (XEXP (x, 0)) == REG_X) { /* This is a paranoid case LEGITIMIZE_RELOAD_ADDRESS must exclude it but I have this situation with extremal optimizing options. */ - avr_asm_len ("adiw r26,%o1" CR_TAB - "ld %0,X", op, plen, -2); - - if (!reg_overlap_mentioned_p (dest, XEXP (x, 0)) - && !reg_unused_after (insn, XEXP (x, 0))) - { - avr_asm_len ("sbiw r26,%o1", op, plen, 1); - } + // PR117744: The base register overlaps dest and is + // only partially clobbered. + bool clobber_r26 = (partial_clobber + && REGNO (base) == (REGNO (base) & ~1)); + if (partial_clobber + && ! clobber_r26) + avr_asm_len ("mov __tmp_reg__,%0", &base2, plen, 1); + + avr_asm_len ("adiw r26,%o1" CR_TAB + "ld %0,X", op, plen, 2); + + if (clobber_r26) + avr_asm_len ("subi r26,lo8(%o1)", op, plen, 1); + else if (partial_clobber) + avr_asm_len ("mov %0,__tmp_reg__", &base2, plen, 1); + else if (! reg_unused_after (insn, base)) + avr_asm_len ("sbiw r26,%o1", op, plen, 1); return ""; } commit 8a4cb2a758938df6e51d88e48340f5fede0061d1 Author: Paul Thomas Date: Mon Nov 11 12:21:57 2024 +0000 Fortran: Fix elemental array refs in SELECT TYPE [PR109345] 2024-11-10 Paul Thomas gcc/fortran PR fortran/109345 * trans-array.cc (gfc_get_array_span): Unlimited polymorphic expressions are now treated separately since the span need not be the same as the element size. gcc/testsuite/ PR fortran/109345 * gfortran.dg/character_workout_1.f90: Cut trailing whitespace. * gfortran.dg/pr109345.f90: New test. (cherry picked from commit e22d80d4f0f8d33f538c1a4bad07b2c819a6d55c) diff --git a/gcc/fortran/trans-array.cc b/gcc/fortran/trans-array.cc index 59668177bbf..504b1bb07f0 100644 --- a/gcc/fortran/trans-array.cc +++ b/gcc/fortran/trans-array.cc @@ -943,6 +943,8 @@ tree gfc_get_array_span (tree desc, gfc_expr *expr) { tree tmp; + gfc_symbol *sym = expr->expr_type == EXPR_VARIABLE + ? expr->symtree->n.sym : NULL; if (is_pointer_array (desc) || (get_CFI_desc (NULL, expr, &desc, NULL) @@ -964,25 +966,43 @@ gfc_get_array_span (tree desc, gfc_expr *expr) desc = build_fold_indirect_ref_loc (input_location, desc); tmp = gfc_conv_descriptor_span_get (desc); } + else if (UNLIMITED_POLY (expr) + || (sym && UNLIMITED_POLY (sym))) + { + /* Treat unlimited polymorphic expressions separately because + the element size need not be the same as the span. Obtain + the class container, which is simplified here by their being + no component references. */ + if (sym && sym->attr.dummy) + { + tmp = gfc_get_symbol_decl (sym); + tmp = GFC_DECL_SAVED_DESCRIPTOR (tmp); + if (INDIRECT_REF_P (tmp)) + tmp = TREE_OPERAND (tmp, 0); + } + else + { + gcc_assert (GFC_DESCRIPTOR_TYPE_P (TREE_TYPE (desc))); + tmp = TREE_OPERAND (desc, 0); + } + tmp = gfc_class_data_get (tmp); + tmp = gfc_conv_descriptor_span_get (tmp); + } else if (TREE_CODE (desc) == COMPONENT_REF && GFC_DESCRIPTOR_TYPE_P (TREE_TYPE (desc)) && GFC_CLASS_TYPE_P (TREE_TYPE (TREE_OPERAND (desc, 0)))) { - /* The descriptor is a class _data field and so use the vtable - size for the receiving span field. */ - tmp = gfc_get_vptr_from_expr (desc); + /* The descriptor is a class _data field. Use the vtable size + since it is guaranteed to have been set and is always OK for + class array descriptors that are not unlimited. */ + tmp = gfc_class_vptr_get (TREE_OPERAND (desc, 0)); tmp = gfc_vptr_size_get (tmp); } - else if (expr && expr->expr_type == EXPR_VARIABLE - && expr->symtree->n.sym->ts.type == BT_CLASS - && expr->ref->type == REF_COMPONENT - && expr->ref->next->type == REF_ARRAY - && expr->ref->next->next == NULL - && CLASS_DATA (expr->symtree->n.sym)->attr.dimension) + else if (sym && sym->ts.type == BT_CLASS && sym->attr.dummy) { - /* Dummys come in sometimes with the descriptor detached from - the class field or declaration. */ - tmp = gfc_class_vptr_get (expr->symtree->n.sym->backend_decl); + /* Class dummys usually requires extraction from the saved + descriptor, which gfc_class_vptr_get does for us. */ + tmp = gfc_class_vptr_get (sym->backend_decl); tmp = gfc_vptr_size_get (tmp); } else diff --git a/gcc/testsuite/gfortran.dg/character_workout_1.f90 b/gcc/testsuite/gfortran.dg/character_workout_1.f90 index 98133b48960..8f8bdbf0069 100644 --- a/gcc/testsuite/gfortran.dg/character_workout_1.f90 +++ b/gcc/testsuite/gfortran.dg/character_workout_1.f90 @@ -1,7 +1,7 @@ ! { dg-do run } ! ! Tests fix for PR100120/100816/100818/100819/100821 -! +! program main_p @@ -27,10 +27,10 @@ program main_p character(len=m, kind=k), pointer :: pm(:) character(len=e, kind=k), pointer :: pe(:) character(len=:, kind=k), pointer :: pd(:) - + class(*), pointer :: su class(*), pointer :: pu(:) - + integer :: i, j nullify(s1, sm, se, sd, su) @@ -41,7 +41,7 @@ program main_p cm(i)(j:j) = char(i*m+j+c-m, kind=k) end do end do - + s1 => c1(n) if(.not.associated(s1)) stop 1 if(.not.associated(s1, c1(n))) stop 2 diff --git a/gcc/testsuite/gfortran.dg/pr109345.f90 b/gcc/testsuite/gfortran.dg/pr109345.f90 new file mode 100644 index 00000000000..cff9aaa987a --- /dev/null +++ b/gcc/testsuite/gfortran.dg/pr109345.f90 @@ -0,0 +1,77 @@ +! { dg-do run } +! +! Test the fix for PR109345 in which array references in the SELECT TYPE +! block below failed because the descriptor span was not set correctly. +! +! Contributed by Lauren Chilutti +! +program test + implicit none + type :: t + character(len=12, kind=4) :: str_array(4) + integer :: i + end type + character(len=12, kind=1), target :: str_array(4) + character(len=12, kind=4), target :: str_array4(4) + type(t) :: str_t (4) + integer :: i + + str_array(:) = "" + str_array(1) = "12345678" + str_array(2) = "@ABCDEFG" +! Original failing test + call foo (str_array) + + str_array4(:) = "" + str_array4(1) = "12345678" + str_array4(2) = "@ABCDEFG" + str_t = [(t(str_array4, i), i = 1, 4)] +! Test character(kind=4) + call foo (str_t(2)%str_array) +! Test component references + call foo (str_t%str_array(1), .true.) +! Test component references and that array offset is correct. + call foo (str_t(2:3)%i) + +contains + subroutine foo (var, flag) + class(*), intent(in) :: var(:) + integer(kind=4) :: i + logical, optional :: flag + select type (var) + type is (character(len=*, kind=1)) + if (len (var) /= 12) stop 1 +! Scalarised array references worked. + if (any (var /= str_array)) stop 2 + do i = 1, size(var) +! Elemental array references did not work. + if (trim (var(i)) /= trim (str_array(i))) stop 3 + enddo + + type is (character(len=*, kind=4)) + if (len (var) /= 12) stop 4 +! Scalarised array references worked. + if (any (var /= var(1))) then + if (any (var /= str_array4)) stop 5 + else + if (any (var /= str_array4(1))) stop 6 + end if + do i = 1, size(var) +! Elemental array references did not work. + if (var(i) /= var(1)) then + if (present (flag)) stop 7 + if (trim (var(i)) /= trim (str_array4(i))) stop 8 + else + if (trim (var(i)) /= trim (str_array4(1))) stop 9 + end if + enddo + + type is (integer(kind=4)) + if (any(var /= [2,3])) stop 10 + do i = 1, size (var) + if (var(i) /= i+1) stop 11 + end do + end select + end +end + commit bfe6f62ab602cbee32911c55a336ae778eb20084 Author: GCC Administrator Date: Sun Nov 24 00:21:09 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 1e0c794232c..f09eeaa2232 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,14 @@ +2024-11-23 Georg-Johann Lay + + Backported from master: + 2024-11-23 Georg-Johann Lay + + PR target/117744 + * config/avr/avr.cc (out_movqi_r_mr): Fix code when a load + only partially clobbers an address register due to + changing the address register temporally to accommodate for + faked addressing modes. + 2024-11-19 Uros Bizjak Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index f69d3317a0a..eba8e1cb787 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241123 +20241124 diff --git a/gcc/fortran/ChangeLog b/gcc/fortran/ChangeLog index 7ebf8cb86b8..b7e213a131e 100644 --- a/gcc/fortran/ChangeLog +++ b/gcc/fortran/ChangeLog @@ -1,3 +1,13 @@ +2024-11-23 Paul Thomas + + Backported from master: + 2024-11-11 Paul Thomas + + PR fortran/109345 + * trans-array.cc (gfc_get_array_span): Unlimited polymorphic + expressions are now treated separately since the span need not + be the same as the element size. + 2024-11-01 Paul Thomas Backported from master: diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 33f9fea5b97..0fee413c5fa 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,12 @@ +2024-11-23 Paul Thomas + + Backported from master: + 2024-11-11 Paul Thomas + + PR fortran/109345 + * gfortran.dg/character_workout_1.f90: Cut trailing whitespace. + * gfortran.dg/pr109345.f90: New test. + 2024-11-19 Uros Bizjak Backported from master: commit 433fcc4accc7fcec18f4996370b05f055da365a6 Author: GCC Administrator Date: Mon Nov 25 00:20:04 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index eba8e1cb787..19047096d89 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241124 +20241125 commit 1f7aaeb9c7e7244959d8262134fe28dd40172086 Author: GCC Administrator Date: Tue Nov 26 00:20:05 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 19047096d89..866e998b5b7 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241125 +20241126 commit 89a27cf6b1354cc80d834d71f7a3aa137d605e94 Author: liuhongt Date: Thu Nov 21 23:57:38 2024 -0800 Fix uninitialized operands[2] in vec_unpacks_hi_v4sf. It could cause weired spill in RA when register pressure is high. gcc/ChangeLog: PR target/117562 * config/i386/sse.md (vec_unpacks_hi_v4sf): Initialize operands[2] with CONST0_RTX. (cherry picked from commit ba4cf2e296d8d5950c3d356fa6b6efcad00d0189) diff --git a/gcc/config/i386/sse.md b/gcc/config/i386/sse.md index 3ad96f321a6..09b308e03c7 100644 --- a/gcc/config/i386/sse.md +++ b/gcc/config/i386/sse.md @@ -9129,7 +9129,10 @@ (match_dup 2) (parallel [(const_int 0) (const_int 1)]))))] "TARGET_SSE2" - "operands[2] = gen_reg_rtx (V4SFmode);") +{ + operands[2] = gen_reg_rtx (V4SFmode); + emit_move_insn (operands[2], CONST0_RTX (V4SFmode)); +}) (define_expand "vec_unpacks_hi_v8sf" [(set (match_dup 2) commit 2897bfe8f9efdf58a61e955b51821cc49d2a9cfc Author: Arsen Arsenović Date: Thu Aug 15 19:17:41 2024 +0200 gnat: fix lto-type-mismatch between C_Version_String and gnat_version_string [PR115917] gcc/ada/ChangeLog: PR ada/115917 * gnatvsn.ads: Add note about the duplication of this value in version.c. * version.c (VER_LEN_MAX): Define to the same value as Gnatvsn.Ver_Len_Max. (gnat_version_string): Use VER_LEN_MAX as bound. (cherry picked from commit 9cbcf8d1de159e6113fafb5dc2feb4a7e467a302) diff --git a/gcc/ada/gnatvsn.ads b/gcc/ada/gnatvsn.ads index 47a06b96c3c..99d06c7e5aa 100644 --- a/gcc/ada/gnatvsn.ads +++ b/gcc/ada/gnatvsn.ads @@ -83,7 +83,8 @@ package Gnatvsn is -- space to store any possible version string value for checks. This -- value should never be decreased in the future, but it would be -- OK to increase it if absolutely necessary. If it is increased, - -- be sure to increase GNAT.Compiler.Version.Ver_Len_Max as well. + -- be sure to increase GNAT.Compiler.Version.Ver_Len_Max, and to update + -- the VER_LEN_MAX define in version.c as well. Ver_Prefix : constant String := "GNAT Version: "; -- Prefix generated by binder. If it is changed, be sure to change diff --git a/gcc/ada/version.c b/gcc/ada/version.c index 5e64edd0b17..2fa9b8c2c85 100644 --- a/gcc/ada/version.c +++ b/gcc/ada/version.c @@ -31,4 +31,7 @@ #include "version.h" -char gnat_version_string[] = version_string; +/* Logically a reference to Gnatvsn.Ver_Len_Max. Please keep in sync. */ +#define VER_LEN_MAX 256 + +char gnat_version_string[VER_LEN_MAX] = version_string; commit ca8df391602593b5a9be6220f0337767d8ed8837 Author: GCC Administrator Date: Wed Nov 27 00:21:35 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index f09eeaa2232..15817d5d62d 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2024-11-26 liuhongt + + Backported from master: + 2024-11-25 liuhongt + + PR target/117562 + * config/i386/sse.md (vec_unpacks_hi_v4sf): Initialize + operands[2] with CONST0_RTX. + 2024-11-23 Georg-Johann Lay Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 866e998b5b7..2073d1c3902 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241126 +20241127 diff --git a/gcc/ada/ChangeLog b/gcc/ada/ChangeLog index 3f47bc1fab8..f75addf32ed 100644 --- a/gcc/ada/ChangeLog +++ b/gcc/ada/ChangeLog @@ -1,3 +1,15 @@ +2024-11-26 Arsen Arsenović + + Backported from master: + 2024-08-19 Arsen Arsenović + + PR ada/115917 + * gnatvsn.ads: Add note about the duplication of this value in + version.c. + * version.c (VER_LEN_MAX): Define to the same value as + Gnatvsn.Ver_Len_Max. + (gnat_version_string): Use VER_LEN_MAX as bound. + 2024-11-04 Eric Botcazou * checks.adb (Selected_Length_Checks.Get_E_Length): For a commit 2ae871b71512f77cc6857bf0ecbf80dd1253e18c Author: Paul Thomas Date: Tue Nov 26 08:58:21 2024 +0000 Fortran: Partial reversion of r15-5083 [PR117763] 2024-11-26 Paul Thomas gcc/fortran PR fortran/117763 * trans-array.cc (gfc_get_array_span): Guard against derefences of 'expr'. Clean up some typos. Use 'gfc_get_vptr_from_expr' for clarity and apply a functional reversion of last section that deals with class dummies. gcc/testsuite/ PR fortran/117763 * gfortran.dg/pr117763.f90: New test. (cherry picked from commit 8278d9551df610179fca114808a7e6e62bab3d82) diff --git a/gcc/fortran/trans-array.cc b/gcc/fortran/trans-array.cc index 504b1bb07f0..ddaedf9604e 100644 --- a/gcc/fortran/trans-array.cc +++ b/gcc/fortran/trans-array.cc @@ -943,8 +943,8 @@ tree gfc_get_array_span (tree desc, gfc_expr *expr) { tree tmp; - gfc_symbol *sym = expr->expr_type == EXPR_VARIABLE - ? expr->symtree->n.sym : NULL; + gfc_symbol *sym = (expr && expr->expr_type == EXPR_VARIABLE) ? + expr->symtree->n.sym : NULL; if (is_pointer_array (desc) || (get_CFI_desc (NULL, expr, &desc, NULL) @@ -971,7 +971,7 @@ gfc_get_array_span (tree desc, gfc_expr *expr) { /* Treat unlimited polymorphic expressions separately because the element size need not be the same as the span. Obtain - the class container, which is simplified here by their being + the class container, which is simplified here by there being no component references. */ if (sym && sym->attr.dummy) { @@ -995,12 +995,16 @@ gfc_get_array_span (tree desc, gfc_expr *expr) /* The descriptor is a class _data field. Use the vtable size since it is guaranteed to have been set and is always OK for class array descriptors that are not unlimited. */ - tmp = gfc_class_vptr_get (TREE_OPERAND (desc, 0)); + tmp = gfc_get_vptr_from_expr (desc); tmp = gfc_vptr_size_get (tmp); } - else if (sym && sym->ts.type == BT_CLASS && sym->attr.dummy) + else if (sym && sym->ts.type == BT_CLASS + && expr->ref->type == REF_COMPONENT + && expr->ref->next->type == REF_ARRAY + && expr->ref->next->next == NULL + && CLASS_DATA (sym)->attr.dimension) { - /* Class dummys usually requires extraction from the saved + /* Class dummys usually require extraction from the saved descriptor, which gfc_class_vptr_get does for us. */ tmp = gfc_class_vptr_get (sym->backend_decl); tmp = gfc_vptr_size_get (tmp); diff --git a/gcc/testsuite/gfortran.dg/pr117763.f90 b/gcc/testsuite/gfortran.dg/pr117763.f90 new file mode 100644 index 00000000000..5f7b36c0269 --- /dev/null +++ b/gcc/testsuite/gfortran.dg/pr117763.f90 @@ -0,0 +1,279 @@ +! { dg-do compile } +! { dg-options "-fdump-tree-original" } +! +! Test the fix for PR117763, which was a regression caused by the patch for +! PR109345. +! +! Contributed by Juergen Reuter +! +module iso_varying_string + implicit none + integer, parameter, private :: GET_BUFFER_LEN = 1 + + type, public :: varying_string + private + character(LEN=1), dimension(:), allocatable :: chars + end type varying_string + + interface assignment(=) + module procedure op_assign_CH_VS + module procedure op_assign_VS_CH + end interface assignment(=) + + interface char + module procedure char_auto + module procedure char_fixed + end interface char + + interface len + module procedure len_ + end interface len + + interface var_str + module procedure var_str_ + end interface var_str + + public :: assignment(=) + public :: char + public :: len + public :: var_str + + private :: op_assign_CH_VS + private :: op_assign_VS_CH + private :: char_auto + private :: char_fixed + private :: len_ + private :: var_str_ + +contains + + elemental function len_ (string) result (length) + type(varying_string), intent(in) :: string + integer :: length + if(ALLOCATED(string%chars)) then + length = SIZE(string%chars) + else + length = 0 + endif + end function len_ + + elemental subroutine op_assign_CH_VS (var, exp) + character(LEN=*), intent(out) :: var + type(varying_string), intent(in) :: exp + var = char(exp) + end subroutine op_assign_CH_VS + + elemental subroutine op_assign_VS_CH (var, exp) + type(varying_string), intent(out) :: var + character(LEN=*), intent(in) :: exp + var = var_str(exp) + end subroutine op_assign_VS_CH + + pure function char_auto (string) result (char_string) + type(varying_string), intent(in) :: string + character(LEN=len(string)) :: char_string + integer :: i_char + forall(i_char = 1:len(string)) + char_string(i_char:i_char) = string%chars(i_char) + end forall + end function char_auto + + pure function char_fixed (string, length) result (char_string) + type(varying_string), intent(in) :: string + integer, intent(in) :: length + character(LEN=length) :: char_string + char_string = char(string) + end function char_fixed + + elemental function var_str_ (char) result (string) + character(LEN=*), intent(in) :: char + type(varying_string) :: string + integer :: length + integer :: i_char + length = LEN(char) + ALLOCATE(string%chars(length)) + forall(i_char = 1:length) + string%chars(i_char) = char(i_char:i_char) + end forall + end function var_str_ + +end module iso_varying_string + +module model_data + use, intrinsic :: iso_c_binding !NODEP! + use iso_varying_string, string_t => varying_string + + implicit none + private + + public :: field_data_t + public :: model_data_t + + type :: field_data_t + private + type(string_t) :: longname + integer :: pdg = 0 + logical :: has_anti = .false. + type(string_t), dimension(:), allocatable :: name, anti + type(string_t) :: tex_name + integer :: multiplicity = 1 + contains + procedure :: init => field_data_init + procedure :: set => field_data_set + procedure :: get_longname => field_data_get_longname + procedure :: get_name_array => field_data_get_name_array + end type field_data_t + + type :: model_data_t + private + type(field_data_t), dimension(:), allocatable :: field + contains + generic :: init => model_data_init + procedure, private :: model_data_init + procedure :: get_field_array_ptr => model_data_get_field_array_ptr + procedure :: get_field_ptr_by_index => model_data_get_field_ptr_index + procedure :: init_sm_test => model_data_init_sm_test + end type model_data_t + + +contains + + subroutine field_data_init (prt, longname, pdg) + class(field_data_t), intent(out) :: prt + type(string_t), intent(in) :: longname + integer, intent(in) :: pdg + prt%longname = longname + prt%pdg = pdg + prt%tex_name = "" + end subroutine field_data_init + + subroutine field_data_set (prt, & + name, anti, tex_name) + class(field_data_t), intent(inout) :: prt + type(string_t), dimension(:), intent(in), optional :: name, anti + type(string_t), intent(in), optional :: tex_name + if (present (name)) then + if (allocated (prt%name)) deallocate (prt%name) + allocate (prt%name (size (name)), source = name) + end if + if (present (anti)) then + if (allocated (prt%anti)) deallocate (prt%anti) + allocate (prt%anti (size (anti)), source = anti) + prt%has_anti = .true. + end if + if (present (tex_name)) prt%tex_name = tex_name + end subroutine field_data_set + + pure function field_data_get_longname (prt) result (name) + type(string_t) :: name + class(field_data_t), intent(in) :: prt + name = prt%longname + end function field_data_get_longname + + subroutine field_data_get_name_array (prt, is_antiparticle, name) + class(field_data_t), intent(in) :: prt + logical, intent(in) :: is_antiparticle + type(string_t), dimension(:), allocatable, intent(inout) :: name + if (allocated (name)) deallocate (name) + if (is_antiparticle) then + if (prt%has_anti) then + allocate (name (size (prt%anti))) + name = prt%anti + else + allocate (name (0)) + end if + else + allocate (name (size (prt%name))) + name = prt%name + end if + end subroutine field_data_get_name_array + + subroutine model_data_init (model, n_field) + class(model_data_t), intent(out) :: model + integer, intent(in) :: n_field + allocate (model%field (n_field)) + end subroutine model_data_init + + function model_data_get_field_array_ptr (model) result (ptr) + class(model_data_t), intent(in), target :: model + type(field_data_t), dimension(:), pointer :: ptr + ptr => model%field + end function model_data_get_field_array_ptr + + function model_data_get_field_ptr_index (model, i) result (ptr) + class(model_data_t), intent(in), target :: model + integer, intent(in) :: i + type(field_data_t), pointer :: ptr + ptr => model%field(i) + end function model_data_get_field_ptr_index + + subroutine model_data_init_sm_test (model) + class(model_data_t), intent(out) :: model + type(field_data_t), pointer :: field + integer :: i + call model%init (2) + i = 0 + i = i + 1 + field => model%get_field_ptr_by_index (i) + call field%init (var_str ("W_BOSON"), 24) + call field%set (name = [var_str ("W+")], anti = [var_str ("W-")]) + i = i + 1 + field => model%get_field_ptr_by_index (i) + call field%init (var_str ("HIGGS"), 25) + call field%set (name = [var_str ("H")]) + end subroutine model_data_init_sm_test + +end module model_data + + +module models + use, intrinsic :: iso_c_binding !NODEP! + use iso_varying_string, string_t => varying_string + use model_data +! use parser +! use variables + implicit none + private + public :: model_t + + type, extends (model_data_t) :: model_t + private + contains + procedure :: append_field_vars => model_append_field_vars + end type model_t + +contains + + subroutine model_append_field_vars (model) + class(model_t), intent(inout) :: model + type(field_data_t), dimension(:), pointer :: field_array + type(field_data_t), pointer :: field + type(string_t) :: name + type(string_t), dimension(:), allocatable :: name_array + integer :: i, j + field_array => model%get_field_array_ptr () + do i = 1, size (field_array) + name = field_array(i)%get_longname () + call field_array(i)%get_name_array (.false., name_array) + end do + end subroutine model_append_field_vars + +end module models + + +program main_ut + use iso_varying_string, string_t => varying_string + use model_data + use models + implicit none + + class(model_data_t), pointer :: model + model => null () + allocate (model_t :: model) + select type (model) + type is (model_t) + call model%init_sm_test () + call model%append_field_vars () + end select +end program main_ut +! { dg-final { scan-tree-dump-times "__result->span = \[12\].." 1 "original" } } commit dc0e962ea18667bc3cdabcafef85b241a4f2c678 Author: Martin Jambor Date: Fri Nov 15 14:37:06 2024 +0100 tree-sra: Avoid SRAing arguments to a function returning_twice (PR 117142) This is a manual bacport of commit 29d8f1f0b7ad3c69b3bdb130325300d5f73aa784 which must be done slightly elsewhere for gcc 13 and 12 because function build_access_from_call_arg was added only in gcc 14. But the gist of the patch is the same. The commit message of the original fix says: PR 117142 shows that the current SRA probably never worked reliably with arguments passed to a function returning twice, because it then creates statements before the call which however needs to be at the beginning of a basic block. While it should be possible to make at least the case of passing arguments by value work with SRA (the statements would need to be put just on the non-abnormal edges leading to the BB), this would mean large surgery of function sra_modify_expr and I guess the time would better be spent re-organizing the whole pass. gcc/ChangeLog: 2024-11-14 Martin Jambor PR tree-optimization/117142 * tree-sra.cc (scan_function): Disqualify any candidate passed to a function returning twice. gcc/testsuite/ChangeLog: 2024-11-14 Martin Jambor * gcc.dg/tree-ssa/pr117142.c: New test. (cherry picked from commit 6244de432a5ba9807c6f0065e70a8025af7b1bd6) diff --git a/gcc/testsuite/gcc.dg/tree-ssa/pr117142.c b/gcc/testsuite/gcc.dg/tree-ssa/pr117142.c new file mode 100644 index 00000000000..fc62c1e58f2 --- /dev/null +++ b/gcc/testsuite/gcc.dg/tree-ssa/pr117142.c @@ -0,0 +1,14 @@ +/* { dg-do compile } */ +/* { dg-options "-O1" } */ + +struct a { + int b; +}; +void c(int, int); +void __attribute__((returns_twice)) +bar1(struct a); +void bar(struct a) { + struct a d; + bar1(d); + c(d.b, d.b); +} diff --git a/gcc/tree-sra.cc b/gcc/tree-sra.cc index 47eee5add12..5a9eaf31b6e 100644 --- a/gcc/tree-sra.cc +++ b/gcc/tree-sra.cc @@ -1392,9 +1392,16 @@ scan_function (void) break; case GIMPLE_CALL: - for (i = 0; i < gimple_call_num_args (stmt); i++) - ret |= build_access_from_expr (gimple_call_arg (stmt, i), - stmt, false); + if (gimple_call_flags (stmt) & ECF_RETURNS_TWICE) + { + for (i = 0; i < gimple_call_num_args (stmt); i++) + disqualify_base_of_expr (gimple_call_arg (stmt, i), + "Passed to a returns_twice call."); + } + else + for (i = 0; i < gimple_call_num_args (stmt); i++) + ret |= build_access_from_expr (gimple_call_arg (stmt, i), + stmt, false); t = gimple_call_lhs (stmt); if (t && !disqualify_if_bad_bb_terminating_stmt (stmt, t, NULL)) commit aaa935f14c87eecde2c2bc545e01d500db7a51ad Author: GCC Administrator Date: Thu Nov 28 00:22:14 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 15817d5d62d..ea331f065d8 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2024-11-27 Martin Jambor + + Backported from master: + 2024-11-15 Martin Jambor + + PR tree-optimization/117142 + * tree-sra.cc (scan_function): Disqualify any candidate passed to + a function returning twice. + 2024-11-26 liuhongt Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 2073d1c3902..7c44885e94f 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241127 +20241128 diff --git a/gcc/fortran/ChangeLog b/gcc/fortran/ChangeLog index b7e213a131e..74ad47f71d0 100644 --- a/gcc/fortran/ChangeLog +++ b/gcc/fortran/ChangeLog @@ -1,3 +1,14 @@ +2024-11-27 Paul Thomas + + Backported from master: + 2024-11-26 Paul Thomas + + PR fortran/117763 + * trans-array.cc (gfc_get_array_span): Guard against derefences + of 'expr'. Clean up some typos. Use 'gfc_get_vptr_from_expr' + for clarity and apply a functional reversion of last section + that deals with class dummies. + 2024-11-23 Paul Thomas Backported from master: diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 0fee413c5fa..9a0a6e33149 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,18 @@ +2024-11-27 Martin Jambor + + Backported from master: + 2024-11-15 Martin Jambor + + * gcc.dg/tree-ssa/pr117142.c: New test. + +2024-11-27 Paul Thomas + + Backported from master: + 2024-11-26 Paul Thomas + + PR fortran/117763 + * gfortran.dg/pr117763.f90: New test. + 2024-11-23 Paul Thomas Backported from master: commit 255b9662a78f4ad940abcc4b0fd035e046fb48a6 Author: GCC Administrator Date: Fri Nov 29 00:21:42 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7c44885e94f..c38b4a0a46e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241128 +20241129 commit b44e2edb60ca7af37e5ba74ff9b254b30fb893b1 Author: GCC Administrator Date: Sat Nov 30 00:22:00 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c38b4a0a46e..fd179292ccd 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241129 +20241130 commit da848c3b9396456c85d8c8055af8158148cbc1a6 Author: Paul Thomas Date: Wed Nov 13 08:57:55 2024 +0000 Fortran: Fix failing character pointer fcn assignment [PR105054] 2024-11-14 Paul Thomas gcc/fortran PR fortran/105054 * resolve.cc (get_temp_from_expr): If the pointer function has a deferred character length, generate a new deferred charlen for the temporary. gcc/testsuite/ PR fortran/105054 * gfortran.dg/ptr_func_assign_6.f08: New test. (cherry picked from commit f530a8c61383b174a476b64f46d56adeedf49dc4) diff --git a/gcc/fortran/resolve.cc b/gcc/fortran/resolve.cc index 6a7325e15e7..0ce41941edd 100644 --- a/gcc/fortran/resolve.cc +++ b/gcc/fortran/resolve.cc @@ -11800,6 +11800,17 @@ resolve_ptr_fcn_assign (gfc_code **code, gfc_namespace *ns) tmp_ptr_expr->symtree->n.sym->attr.allocatable = 0; tmp_ptr_expr->where = (*code)->loc; + /* A new charlen is required to ensure that the variable string length + is different to that of the original lhs for deferred results. */ + if (s->result->ts.deferred && tmp_ptr_expr->ts.type == BT_CHARACTER) + { + tmp_ptr_expr->ts.u.cl = gfc_get_charlen(); + tmp_ptr_expr->ts.deferred = 1; + tmp_ptr_expr->ts.u.cl->next = gfc_current_ns->cl_list; + gfc_current_ns->cl_list = tmp_ptr_expr->ts.u.cl; + tmp_ptr_expr->symtree->n.sym->ts.u.cl = tmp_ptr_expr->ts.u.cl; + } + this_code = build_assignment (EXEC_ASSIGN, tmp_ptr_expr, (*code)->expr2, NULL, NULL, (*code)->loc); diff --git a/gcc/testsuite/gfortran.dg/ptr_func_assign_6.f08 b/gcc/testsuite/gfortran.dg/ptr_func_assign_6.f08 new file mode 100644 index 00000000000..d62815d7afa --- /dev/null +++ b/gcc/testsuite/gfortran.dg/ptr_func_assign_6.f08 @@ -0,0 +1,89 @@ +! { dg-do run } +! +! Test the fix for PR105054. +! +! Contributed by Arjen Markus +! +module string_pointers + implicit none + character(len=20), dimension(10), target :: array_strings + character(len=:), dimension(:), target, allocatable :: array_strings2 + +contains + +function pointer_to_string( i , flag) + integer, intent(in) :: i, flag + + character(len=:), pointer :: pointer_to_string + + if (flag == 1) then + pointer_to_string => array_strings(i) + return + endif + + if (.not.allocated (array_strings2)) allocate (array_strings2(4), & + mold = ' ') + pointer_to_string => array_strings2(i) +end function pointer_to_string + +function pointer_to_string2( i , flag) result (res) + integer, intent(in) :: i, flag + + character(len=:), pointer :: res + + if (flag == 1) then + res => array_strings(i) + return + endif + + if (.not.allocated (array_strings2)) allocate (array_strings2(4), & + mold = ' ') + res => array_strings2(i) +end function pointer_to_string2 + +end module string_pointers + +program chk_string_pointer + use string_pointers + implicit none + integer :: i + character(*), parameter :: chr(4) = ['1234 ','ABCDefgh ', & + '12345678 ',' '] + + pointer_to_string(1, 1) = '1234567890' + pointer_to_string(2, 1) = '12345678901234567890' + + if (len(pointer_to_string(3, 1)) /= 20) stop 1 + + array_strings(1) = array_strings(1)(1:4) // 'ABC' + if (pointer_to_string(1, 1) /= '1234ABC') stop 2 + + pointer_to_string(1, 2) = '1234' + pointer_to_string(2, 2) = 'ABCDefgh' + pointer_to_string(3, 2) = '12345678' + + do i = 1, 3 + if (trim (array_strings2(i)) /= trim(chr(i))) stop 3 + enddo + +! Clear the target arrays + array_strings = repeat (' ', 20) + deallocate (array_strings2) + +! Repeat with an explicit result. + pointer_to_string2(1, 1) = '1234567890' + pointer_to_string2(2, 1) = '12345678901234567890' + + if (len(pointer_to_string(3, 1)) /= 20) stop 4 + + array_strings(1) = array_strings(1)(1:4) // 'ABC' + if (pointer_to_string(1, 1) /= '1234ABC') stop 5 + + pointer_to_string2(1, 2) = '1234' + pointer_to_string2(2, 2) = 'ABCDefgh' + pointer_to_string2(3, 2) = '12345678' + + do i = 1, 3 + if (trim (array_strings2(i)) /= trim(chr(i))) stop 6 + enddo +end program chk_string_pointer commit c669cfd702ae477b9272d571115eaad322860a7d Author: GCC Administrator Date: Sun Dec 1 00:20:38 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index fd179292ccd..8b81eb36155 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241130 +20241201 diff --git a/gcc/fortran/ChangeLog b/gcc/fortran/ChangeLog index 74ad47f71d0..170af6090ac 100644 --- a/gcc/fortran/ChangeLog +++ b/gcc/fortran/ChangeLog @@ -1,3 +1,13 @@ +2024-11-30 Paul Thomas + + Backported from master: + 2024-11-13 Paul Thomas + + PR fortran/105054 + * resolve.cc (get_temp_from_expr): If the pointer function has + a deferred character length, generate a new deferred charlen + for the temporary. + 2024-11-27 Paul Thomas Backported from master: diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 9a0a6e33149..ee202c4478c 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-11-30 Paul Thomas + + Backported from master: + 2024-11-13 Paul Thomas + + PR fortran/105054 + * gfortran.dg/ptr_func_assign_6.f08: New test. + 2024-11-27 Martin Jambor Backported from master: commit 2ab4bee84dfdd743a4399e6bb06f213a4118b4d1 Author: GCC Administrator Date: Mon Dec 2 00:21:00 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 8b81eb36155..7942c684a26 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241201 +20241202 commit ddfc04188bca888f1cbdadd8a2457ed7d7031f32 Author: Andre Vieira Date: Mon Dec 2 13:35:03 2024 +0000 arm, mve: Adding missing Runtime Library Exception to header files Add missing Runtime Library Exception to mve header files to bring them into line with other similar headers. Not adding it in the first place was an oversight. gcc/ChangeLog: * config/arm/arm_mve.h: Add Runtime Library Exception. * config/arm/arm_mve_types.h: Likewise. (cherry picked from commit cde7ce0628f66a5d03cc97c70d4695e6f2acd4db) diff --git a/gcc/config/arm/arm_mve.h b/gcc/config/arm/arm_mve.h index c359e9c6336..9727c538463 100644 --- a/gcc/config/arm/arm_mve.h +++ b/gcc/config/arm/arm_mve.h @@ -15,6 +15,10 @@ or FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for more details. + Under Section 7 of GPL version 3, you are granted additional + permissions described in the GCC Runtime Library Exception, version + 3.1, as published by the Free Software Foundation. + You should have received a copy of the GNU General Public License along with GCC; see the file COPYING3. If not see . */ diff --git a/gcc/config/arm/arm_mve_types.h b/gcc/config/arm/arm_mve_types.h index 0b2d6422545..e1193d52c0f 100644 --- a/gcc/config/arm/arm_mve_types.h +++ b/gcc/config/arm/arm_mve_types.h @@ -15,6 +15,10 @@ or FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for more details. + Under Section 7 of GPL version 3, you are granted additional + permissions described in the GCC Runtime Library Exception, version + 3.1, as published by the Free Software Foundation. + You should have received a copy of the GNU General Public License along with GCC; see the file COPYING3. If not see . */ commit 4dc02ed98536b610bd8a00c60c1d8ec97d6c7434 Author: GCC Administrator Date: Tue Dec 3 00:22:55 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index ea331f065d8..e196ddd6fdd 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,11 @@ +2024-12-02 Andre Vieira + + Backported from master: + 2024-12-02 Andre Vieira + + * config/arm/arm_mve.h: Add Runtime Library Exception. + * config/arm/arm_mve_types.h: Likewise. + 2024-11-27 Martin Jambor Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7942c684a26..ba184185f0b 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241202 +20241203 commit 45bb94e6e6ed50f142b434199f9a8b3d592ea1ce Author: GCC Administrator Date: Wed Dec 4 00:23:59 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index ba184185f0b..74f6fd4e4e4 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241203 +20241204 commit d3fb3dbe7581663bf25ec5e8c15667c76dd1db9e Author: GCC Administrator Date: Thu Dec 5 00:22:28 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 74f6fd4e4e4..5f2357c99fc 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241204 +20241205 commit ae8d9d2b40aa7fd6a455beda38ff1b3c21728c31 Author: Simon Martin Date: Tue Dec 3 14:30:43 2024 +0100 c++: Don't reject pointer to virtual method during constant evaluation [PR117615] We currently reject the following valid code: === cut here === struct Base { virtual void doit (int v) const {} }; struct Derived : Base { void doit (int v) const {} }; using fn_t = void (Base::*)(int) const; struct Helper { fn_t mFn; constexpr Helper (auto && fn) : mFn(static_cast(fn)) {} }; void foo () { constexpr Helper h (&Derived::doit); } === cut here === The problem is that since r6-4014-gdcdbc004d531b4, &Derived::doit is represented with an expression with type pointer to method and using an INTEGER_CST (here 1), and that cxx_eval_constant_expression rejects any such expression with a non-null INTEGER_CST. This patch uses the same strategy as r12-4491-gf45610a45236e9 (fix for PR c++/102786), and simply lets such expressions go through. PR c++/117615 gcc/cp/ChangeLog: * constexpr.cc (cxx_eval_constant_expression): Don't reject INTEGER_CSTs with type POINTER_TYPE to METHOD_TYPE. gcc/testsuite/ChangeLog: * g++.dg/cpp2a/constexpr-virtual22.C: New test. (cherry picked from commit 72a2380a306a1c3883cb7e4f99253522bc265af0) diff --git a/gcc/cp/constexpr.cc b/gcc/cp/constexpr.cc index 20abbee3600..6c8d8ab17f2 100644 --- a/gcc/cp/constexpr.cc +++ b/gcc/cp/constexpr.cc @@ -7353,6 +7353,12 @@ cxx_eval_constant_expression (const constexpr_ctx *ctx, tree t, return t; } } + else if (TYPE_PTR_P (type) + && TREE_CODE (TREE_TYPE (type)) == METHOD_TYPE) + /* INTEGER_CST with pointer-to-method type is only used + for a virtual method in a pointer to member function. + Don't reject those. */ + ; else { /* This detects for example: diff --git a/gcc/testsuite/g++.dg/cpp2a/constexpr-virtual22.C b/gcc/testsuite/g++.dg/cpp2a/constexpr-virtual22.C new file mode 100644 index 00000000000..89330bf8620 --- /dev/null +++ b/gcc/testsuite/g++.dg/cpp2a/constexpr-virtual22.C @@ -0,0 +1,22 @@ +// PR c++/117615 +// { dg-do "compile" { target c++20 } } + +struct Base { + virtual void doit (int v) const {} +}; + +struct Derived : Base { + void doit (int v) const {} +}; + +using fn_t = void (Base::*)(int) const; + +struct Helper { + fn_t mFn; + constexpr Helper (auto && fn) : mFn(static_cast(fn)) {} +}; + +void foo () { + constexpr Helper h (&Derived::doit); + constexpr Helper h2 (&Base::doit); +} commit 499d3dc84e40849f607154bd76ed07d37d744cc1 Author: Georg-Johann Lay Date: Wed Dec 4 20:56:50 2024 +0100 AVR: target/64242 - Copy FP to a local reg in nonlocal_goto. In nonlocal_goto sets, change hard_frame_pointer_rtx only after emit_stack_restore() restored SP. This is needed because SP my be stored in some frame location. gcc/ PR target/64242 * config/avr/avr.md (nonlocal_goto): Don't restore hard_frame_pointer_rtx directly, but copy it to local register, and only set hard_frame_pointer_rtx from it after emit_stack_restore(). (cherry picked from commit f7b5527d1b48b33d8ab633c1e9dcb9883667492a) diff --git a/gcc/config/avr/avr.md b/gcc/config/avr/avr.md index f76249340b8..90ba2d0400e 100644 --- a/gcc/config/avr/avr.md +++ b/gcc/config/avr/avr.md @@ -381,9 +381,14 @@ emit_clobber (gen_rtx_MEM (BLKmode, hard_frame_pointer_rtx)); - emit_move_insn (hard_frame_pointer_rtx, r_fp); + // PR64242: When r_sp is located in the frame, we must not + // change FP prior to reading r_sp. Hence copy r_fp to a + // local register (and hope that reload won't spill it). + rtx r_fp_reg = copy_to_reg (r_fp); emit_stack_restore (SAVE_NONLOCAL, r_sp); + emit_move_insn (hard_frame_pointer_rtx, r_fp_reg); + emit_use (hard_frame_pointer_rtx); emit_use (stack_pointer_rtx); commit 090cd9636a2732ada078de160b97c3c6e8a6d93f Author: GCC Administrator Date: Fri Dec 6 00:20:44 2024 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index e196ddd6fdd..0892d146c17 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,14 @@ +2024-12-05 Georg-Johann Lay + + Backported from master: + 2024-12-05 Georg-Johann Lay + + PR target/64242 + * config/avr/avr.md (nonlocal_goto): Don't restore + hard_frame_pointer_rtx directly, but copy it to local + register, and only set hard_frame_pointer_rtx from it + after emit_stack_restore(). + 2024-12-02 Andre Vieira Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 5f2357c99fc..9093ce495f9 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241205 +20241206 diff --git a/gcc/cp/ChangeLog b/gcc/cp/ChangeLog index bb52869e0ad..f7830dd119a 100644 --- a/gcc/cp/ChangeLog +++ b/gcc/cp/ChangeLog @@ -1,3 +1,12 @@ +2024-12-05 Simon Martin + + Backported from master: + 2024-12-04 Simon Martin + + PR c++/117615 + * constexpr.cc (cxx_eval_constant_expression): Don't reject + INTEGER_CSTs with type POINTER_TYPE to METHOD_TYPE. + 2024-11-05 Simon Martin Backported from master: diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index ee202c4478c..8f9960dd206 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2024-12-05 Simon Martin + + Backported from master: + 2024-12-04 Simon Martin + + PR c++/117615 + * g++.dg/cpp2a/constexpr-virtual22.C: New test. + 2024-11-30 Paul Thomas Backported from master: commit 06fab719eba2c0c2a5a0bffc4eb2c4e280ee967c Author: GCC Administrator Date: Sat Dec 7 00:20:54 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 9093ce495f9..a611dd48945 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241206 +20241207 commit 6e9f95d973fb0bf4337d7443f8df905be9183057 Author: GCC Administrator Date: Sun Dec 8 00:19:57 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index a611dd48945..f4039d392c2 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241207 +20241208 commit 20fc5d11ea6435de9e3e5bd172b02815d4a3f76f Author: GCC Administrator Date: Mon Dec 9 00:19:24 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index f4039d392c2..775d8f2db22 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241208 +20241209 commit 60c51f868a05e05f2e14dbc9b535d12bcd253ec2 Author: GCC Administrator Date: Tue Dec 10 00:20:45 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 775d8f2db22..61bea039942 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241209 +20241210 commit 1cfbb1736e7d75c46715688a7ad490c62d0da242 Author: GCC Administrator Date: Wed Dec 11 00:20:06 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 61bea039942..43fbb1840bf 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241210 +20241211 commit 0e2518c7544430d65090ee64e2330cd73b0397e3 Author: GCC Administrator Date: Thu Dec 12 00:20:00 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 43fbb1840bf..3489d2d624c 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241211 +20241212 commit 62c1d98b870f84bd511deba7b93e8c49e38f4335 Author: Eric Botcazou Date: Thu Dec 12 16:25:09 2024 +0100 Fix precondition failure with Ada.Numerics.Generic_Real_Arrays.Eigenvalues This fixes a precondition failure triggered when the Eigenvalues routine of Ada.Numerics.Generic_Real_Arrays is instantiated with -gnata, beause it calls Sort_Eigensystem on an empty vector. gcc/ada PR ada/117996 * libgnat/a-ngrear.adb (Jacobi): Remove default value for Compute_Vectors formal parameter. (Sort_Eigensystem): Add Compute_Vectors formal parameter. Do not modify the Vectors if Compute_Vectors is False. (Eigensystem): Pass True as Compute_Vectors to Sort_Eigensystem. (Eigenvalues): Pass False as Compute_Vectors to Sort_Eigensystem. gcc/testsuite * gnat.dg/matrix1.adb: New test. diff --git a/gcc/ada/libgnat/a-ngrear.adb b/gcc/ada/libgnat/a-ngrear.adb index 9cfd9562955..844d6264ee7 100644 --- a/gcc/ada/libgnat/a-ngrear.adb +++ b/gcc/ada/libgnat/a-ngrear.adb @@ -96,7 +96,7 @@ package body Ada.Numerics.Generic_Real_Arrays is (A : Real_Matrix; Values : out Real_Vector; Vectors : out Real_Matrix; - Compute_Vectors : Boolean := True); + Compute_Vectors : Boolean); -- Perform Jacobi's eigensystem algorithm on real symmetric matrix A function Length is new Square_Matrix_Length (Real'Base, Real_Matrix); @@ -107,8 +107,9 @@ package body Ada.Numerics.Generic_Real_Arrays is -- Perform a Givens rotation procedure Sort_Eigensystem - (Values : in out Real_Vector; - Vectors : in out Real_Matrix); + (Values : in out Real_Vector; + Vectors : in out Real_Matrix; + Compute_Vectors : Boolean); -- Sort Values and associated Vectors by decreasing absolute value procedure Swap (Left, Right : in out Real); @@ -486,7 +487,7 @@ package body Ada.Numerics.Generic_Real_Arrays is is begin Jacobi (A, Values, Vectors, Compute_Vectors => True); - Sort_Eigensystem (Values, Vectors); + Sort_Eigensystem (Values, Vectors, Compute_Vectors => True); end Eigensystem; ----------------- @@ -500,7 +501,7 @@ package body Ada.Numerics.Generic_Real_Arrays is Vectors : Real_Matrix (1 .. 0, 1 .. 0); begin Jacobi (A, Values, Vectors, Compute_Vectors => False); - Sort_Eigensystem (Values, Vectors); + Sort_Eigensystem (Values, Vectors, Compute_Vectors => False); end; end return; end Eigenvalues; @@ -522,7 +523,7 @@ package body Ada.Numerics.Generic_Real_Arrays is (A : Real_Matrix; Values : out Real_Vector; Vectors : out Real_Matrix; - Compute_Vectors : Boolean := True) + Compute_Vectors : Boolean) is -- This subprogram uses Carl Gustav Jacob Jacobi's iterative method -- for computing eigenvalues and eigenvectors and is based on @@ -731,8 +732,9 @@ package body Ada.Numerics.Generic_Real_Arrays is ---------------------- procedure Sort_Eigensystem - (Values : in out Real_Vector; - Vectors : in out Real_Matrix) + (Values : in out Real_Vector; + Vectors : in out Real_Matrix; + Compute_Vectors : Boolean) is procedure Swap (Left, Right : Integer); -- Swap Values (Left) with Values (Right), and also swap the @@ -748,8 +750,10 @@ package body Ada.Numerics.Generic_Real_Arrays is procedure Swap (Left, Right : Integer) is begin Swap (Values (Left), Values (Right)); - Swap_Column (Vectors, Left - Values'First + Vectors'First (2), - Right - Values'First + Vectors'First (2)); + if Compute_Vectors then + Swap_Column (Vectors, Left - Values'First + Vectors'First (2), + Right - Values'First + Vectors'First (2)); + end if; end Swap; begin diff --git a/gcc/testsuite/gnat.dg/matrix1.adb b/gcc/testsuite/gnat.dg/matrix1.adb new file mode 100644 index 00000000000..2a920e27f0e --- /dev/null +++ b/gcc/testsuite/gnat.dg/matrix1.adb @@ -0,0 +1,16 @@ +-- { dg-do run } +-- { dg-options "-gnata" } + +with Ada.Numerics.Generic_Real_Arrays; + +procedure Matrix1 is + + package GRA is new Ada.Numerics.Generic_Real_Arrays (real => float); + use GRA; + + M : constant Real_Matrix (1..2, 1..2) := ((1.0, 0.0), (0.0, 2.0)); + E : constant Real_Vector := Eigenvalues (M); + +begin + null; +end; commit c817ba13dfbebc03c9b293b8d8bcbea04f52d885 Author: GCC Administrator Date: Fri Dec 13 00:21:07 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 3489d2d624c..dba749d567d 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241212 +20241213 diff --git a/gcc/ada/ChangeLog b/gcc/ada/ChangeLog index f75addf32ed..f4404c7b010 100644 --- a/gcc/ada/ChangeLog +++ b/gcc/ada/ChangeLog @@ -1,3 +1,13 @@ +2024-12-12 Eric Botcazou + + PR ada/117996 + * libgnat/a-ngrear.adb (Jacobi): Remove default value for + Compute_Vectors formal parameter. + (Sort_Eigensystem): Add Compute_Vectors formal parameter. Do not + modify the Vectors if Compute_Vectors is False. + (Eigensystem): Pass True as Compute_Vectors to Sort_Eigensystem. + (Eigenvalues): Pass False as Compute_Vectors to Sort_Eigensystem. + 2024-11-26 Arsen Arsenović Backported from master: diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 8f9960dd206..5b58977a7a1 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,7 @@ +2024-12-12 Eric Botcazou + + * gnat.dg/matrix1.adb: New test. + 2024-12-05 Simon Martin Backported from master: commit 86c5dbc158736c494535bd087a96844fd1b213ac Author: GCC Administrator Date: Sat Dec 14 00:22:01 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index dba749d567d..ae0d5cf829e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241213 +20241214 commit 5f6a72b68f75cac8012bd3b8e40b33194b19074d Author: GCC Administrator Date: Sun Dec 15 00:20:26 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index ae0d5cf829e..565e340ccfe 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241214 +20241215 commit bdc37599e44fa062f2fa41d46f4f04507940e210 Author: GCC Administrator Date: Mon Dec 16 00:20:41 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 565e340ccfe..ccdbb249195 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241215 +20241216 commit 1c6e31ad81cb9cbc39d16e544f075d29fb425f2d Author: GCC Administrator Date: Tue Dec 17 00:21:46 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index ccdbb249195..cbf8d2dfdf1 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241216 +20241217 commit faf432ffedc3c9e3c3d28244c36c341ec7a9e38c Author: GCC Administrator Date: Wed Dec 18 00:21:46 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index cbf8d2dfdf1..781eeaca5bc 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241217 +20241218 commit 1851587d42aad36c849e666700d48cb665d14d8b Author: GCC Administrator Date: Thu Dec 19 00:21:58 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 781eeaca5bc..fcb1ceaa6b9 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241218 +20241219 commit 3b2c5e8e87f1a5796aa5068ce9af99fd4ee3c99e Author: GCC Administrator Date: Fri Dec 20 00:21:42 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index fcb1ceaa6b9..938c9799fa1 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241219 +20241220 commit 5a996b52f350a6030131f9019274bb0d1d35efd6 Author: GCC Administrator Date: Sat Dec 21 00:21:43 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 938c9799fa1..d4528d6fc4f 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241220 +20241221 commit a03684e480af72094f7bdd43d9ef6355bd95a28f Author: GCC Administrator Date: Sun Dec 22 00:21:42 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d4528d6fc4f..6f2cf7ace4e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241221 +20241222 commit 7456ecb0fe9051ebcb3960ae87963c037a83046e Author: GCC Administrator Date: Mon Dec 23 00:20:44 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 6f2cf7ace4e..8f5ae44e831 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241222 +20241223 commit 1eff0e26c5ed9538753b933700bfe9ea7b2f85c6 Author: GCC Administrator Date: Tue Dec 24 00:22:02 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 8f5ae44e831..556dfeee7c0 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241223 +20241224 commit 2a46fb456ecbaf250163737d2d29f0bd0a18abd6 Author: GCC Administrator Date: Wed Dec 25 00:20:54 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 556dfeee7c0..fa34f54b062 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241224 +20241225 commit 421e27ecc6611fe8fbb2cdf639deac1f07b729fa Author: GCC Administrator Date: Thu Dec 26 00:21:47 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index fa34f54b062..66ccea7c37b 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241225 +20241226 commit bf7987c2f748e58df379770167ec4f2086d13d69 Author: GCC Administrator Date: Fri Dec 27 00:20:00 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 66ccea7c37b..c880c6c1d36 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241226 +20241227 commit 8ddbec586be4589d8ae865e9d7ab72e1d7c6dcd2 Author: GCC Administrator Date: Sat Dec 28 00:21:02 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c880c6c1d36..922c136a4f6 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241227 +20241228 commit afeeda00a9ee638791cab4b69bce4cf547cc00f1 Author: GCC Administrator Date: Sun Dec 29 00:20:56 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 922c136a4f6..4f5a98435b4 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241228 +20241229 commit 8c9f8cd327a0a4c38ca14dfa30244e399880de50 Author: GCC Administrator Date: Mon Dec 30 00:20:30 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 4f5a98435b4..71153cc35f6 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241229 +20241230 commit 0268510414748be9c1ac8e1f6ab338dfc8083a74 Author: GCC Administrator Date: Tue Dec 31 00:21:33 2024 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 71153cc35f6..adb1f5f8c2f 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241230 +20241231 commit 36a977eb4933d41d2cda7a534aa4ea992eb19d48 Author: GCC Administrator Date: Wed Jan 1 00:21:55 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index adb1f5f8c2f..fc6c4b42e85 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20241231 +20250101 commit df5e5d0ad7764b159848c95acaff848f67f5f574 Author: GCC Administrator Date: Thu Jan 2 00:23:24 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index fc6c4b42e85..dc5f7519d5d 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250101 +20250102 commit 2eabd17593fc92dc3670af056a0f2bea6812f8d5 Author: GCC Administrator Date: Fri Jan 3 00:20:03 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index dc5f7519d5d..39a15df0cdc 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250102 +20250103 commit 0b7969984d67ad3f3e20446d9ebc3cc4c3863205 Author: GCC Administrator Date: Sat Jan 4 00:20:40 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 39a15df0cdc..d01f0cb4cc9 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250103 +20250104 commit 91e5cb07f92915315bb3f5b74a46cec4afabba07 Author: GCC Administrator Date: Sun Jan 5 00:19:05 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d01f0cb4cc9..224468dbaf5 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250104 +20250105 commit e8b53a5263946be717169e2549eabf67e6a227ad Author: Marc Poulhiès Date: Mon Mar 27 16:47:04 2023 +0200 ada: Fix crash on vector initialization Initializing a vector using Vec : V.Vector := [Some_Type'(Some_Abstract_Type with F => 0)]; may crash the compiler. The expander marks the N_Extension_Aggregate for delayed expansion which never happens and incorrectly ends up in gigi. The delayed expansion is needed for nested aggregates, which the original code is testing for, but container aggregates are handled differently. Such assignments to container aggregates are later transformed into procedure calls to the procedures named in the Aggregate aspect definition, for which the delayed expansion is not required/expected. gcc/ada/ PR ada/118234 * exp_aggr.adb (Convert_To_Assignments): Do not mark node for delayed expansion if parent type has the Aggregate aspect. * sem_util.adb (Is_Container_Aggregate): Move... * sem_util.ads (Is_Container_Aggregate): ... here and make it public. diff --git a/gcc/ada/exp_aggr.adb b/gcc/ada/exp_aggr.adb index e360baa6683..a97882dba73 100644 --- a/gcc/ada/exp_aggr.adb +++ b/gcc/ada/exp_aggr.adb @@ -4930,9 +4930,12 @@ package body Exp_Aggr is if -- Internal aggregate (transformed when expanding the parent) + -- excluding the Container aggregate as these are transformed to + -- procedure call later. - Parent_Kind in - N_Aggregate | N_Extension_Aggregate | N_Component_Association + (Parent_Kind in + N_Component_Association | N_Aggregate | N_Extension_Aggregate + and then not Is_Container_Aggregate (Parent_Node)) -- Allocator (see Convert_Aggr_In_Allocator) diff --git a/gcc/ada/sem_util.adb b/gcc/ada/sem_util.adb index 3c55dda5a85..cbe19cbeeb2 100644 --- a/gcc/ada/sem_util.adb +++ b/gcc/ada/sem_util.adb @@ -130,9 +130,6 @@ package body Sem_Util is -- Determine whether arbitrary entity Id denotes an atomic object as per -- RM C.6(7). - function Is_Container_Aggregate (Exp : Node_Id) return Boolean; - -- Is the given expression a container aggregate? - generic with function Is_Effectively_Volatile_Entity (Id : Entity_Id) return Boolean; diff --git a/gcc/ada/sem_util.ads b/gcc/ada/sem_util.ads index 695158a34f3..28dcefd71fb 100644 --- a/gcc/ada/sem_util.ads +++ b/gcc/ada/sem_util.ads @@ -1531,6 +1531,9 @@ package Sem_Util is -- integer for use in compile-time checking. Note: Level is restricted to -- be non-dynamic. + function Is_Container_Aggregate (Exp : Node_Id) return Boolean; + -- Is the given expression a container aggregate? + function Is_Newly_Constructed (Exp : Node_Id; Context_Requires_NC : Boolean) return Boolean; -- Indicates whether a given expression is "newly constructed" (RM 4.4). commit 9b618e2e883ffada66f6c0c6a6a672f35286413c Author: Eric Botcazou Date: Fri May 26 00:09:14 2023 +0200 ada: Fix internal error on aggregate within container aggregate This just applies the same fix to Expand_Array_Aggregate as the one that was recently applied to Convert_To_Assignments. gcc/ada/ PR ada/118234 * exp_aggr.adb (Convert_To_Assignments): Tweak comment. (Expand_Array_Aggregate): Do not delay the expansion if the parent node is a container aggregate. diff --git a/gcc/ada/exp_aggr.adb b/gcc/ada/exp_aggr.adb index a97882dba73..cda5e66943e 100644 --- a/gcc/ada/exp_aggr.adb +++ b/gcc/ada/exp_aggr.adb @@ -4929,9 +4929,9 @@ package body Exp_Aggr is -- done top down from above. if - -- Internal aggregate (transformed when expanding the parent) - -- excluding the Container aggregate as these are transformed to - -- procedure call later. + -- Internal aggregates (transformed when expanding the parent), + -- excluding container aggregates as these are transformed into + -- subprogram calls later. (Parent_Kind in N_Component_Association | N_Aggregate | N_Extension_Aggregate @@ -6788,7 +6788,8 @@ package body Exp_Aggr is -- STEP 3 -- Delay expansion for nested aggregates: it will be taken care of when - -- the parent aggregate is expanded. + -- the parent aggregate is expanded, excluding container aggregates as + -- these are transformed into subprogram calls later. Parent_Node := Parent (N); Parent_Kind := Nkind (Parent_Node); @@ -6798,9 +6799,10 @@ package body Exp_Aggr is Parent_Kind := Nkind (Parent_Node); end if; - if Parent_Kind = N_Aggregate - or else Parent_Kind = N_Extension_Aggregate - or else Parent_Kind = N_Component_Association + if ((Parent_Kind = N_Component_Association + or else Parent_Kind = N_Aggregate + or else Parent_Kind = N_Extension_Aggregate) + and then not Is_Container_Aggregate (Parent_Node)) or else (Parent_Kind = N_Object_Declaration and then Needs_Finalization (Typ)) or else (Parent_Kind = N_Assignment_Statement commit 6c7c99f95661db80b6e1b52c10c34e9ce9c5eb6a Author: Eric Botcazou Date: Wed Sep 6 09:37:29 2023 +0200 ada: Fix internal error on aggregate nested in container aggregate This handles the case where a component association is present. gcc/ada/ PR ada/118234 * exp_aggr.adb (Convert_To_Assignments): In the case of a component association, call Is_Container_Aggregate on the parent's parent. (Expand_Array_Aggregate): Likewise. diff --git a/gcc/ada/exp_aggr.adb b/gcc/ada/exp_aggr.adb index cda5e66943e..479db647f3c 100644 --- a/gcc/ada/exp_aggr.adb +++ b/gcc/ada/exp_aggr.adb @@ -4933,9 +4933,11 @@ package body Exp_Aggr is -- excluding container aggregates as these are transformed into -- subprogram calls later. - (Parent_Kind in - N_Component_Association | N_Aggregate | N_Extension_Aggregate - and then not Is_Container_Aggregate (Parent_Node)) + (Parent_Kind = N_Component_Association + and then not Is_Container_Aggregate (Parent (Parent_Node))) + + or else (Parent_Kind in N_Aggregate | N_Extension_Aggregate + and then not Is_Container_Aggregate (Parent_Node)) -- Allocator (see Convert_Aggr_In_Allocator) @@ -6799,10 +6801,10 @@ package body Exp_Aggr is Parent_Kind := Nkind (Parent_Node); end if; - if ((Parent_Kind = N_Component_Association - or else Parent_Kind = N_Aggregate - or else Parent_Kind = N_Extension_Aggregate) - and then not Is_Container_Aggregate (Parent_Node)) + if (Parent_Kind = N_Component_Association + and then not Is_Container_Aggregate (Parent (Parent_Node))) + or else (Parent_Kind in N_Aggregate | N_Extension_Aggregate + and then not Is_Container_Aggregate (Parent_Node)) or else (Parent_Kind = N_Object_Declaration and then Needs_Finalization (Typ)) or else (Parent_Kind = N_Assignment_Statement commit 85319f4637648691882a79ebac3d63c4d0204735 Author: Estevan Castilho (Tevo) Date: Sat Dec 28 20:37:37 2024 +0000 Ada: Fix build for dummy s-taprop gcc/ada * libgnarl/s-taprop__dummy.adb: Remove use clause for System.Parameters. (Unlock): Remove Global_Lock formal parameter. (Write_Lock): Likewise. diff --git a/gcc/ada/libgnarl/s-taprop__dummy.adb b/gcc/ada/libgnarl/s-taprop__dummy.adb index 5ab41b1b3c1..e5360a63f3d 100644 --- a/gcc/ada/libgnarl/s-taprop__dummy.adb +++ b/gcc/ada/libgnarl/s-taprop__dummy.adb @@ -37,7 +37,6 @@ package body System.Task_Primitives.Operations is use System.Tasking; - use System.Parameters; pragma Warnings (Off); -- Turn off warnings since so many unreferenced parameters @@ -480,10 +479,7 @@ package body System.Task_Primitives.Operations is null; end Unlock; - procedure Unlock - (L : not null access RTS_Lock; - Global_Lock : Boolean := False) - is + procedure Unlock (L : not null access RTS_Lock) is begin null; end Unlock; @@ -522,10 +518,7 @@ package body System.Task_Primitives.Operations is Ceiling_Violation := False; end Write_Lock; - procedure Write_Lock - (L : not null access RTS_Lock; - Global_Lock : Boolean := False) - is + procedure Write_Lock (L : not null access RTS_Lock) is begin null; end Write_Lock; commit 465e7aca6d3c0a0f243eb1145d7de935915dd469 Author: GCC Administrator Date: Mon Jan 6 00:19:13 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 224468dbaf5..197d498df29 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250105 +20250106 diff --git a/gcc/ada/ChangeLog b/gcc/ada/ChangeLog index f4404c7b010..b0085cf918b 100644 --- a/gcc/ada/ChangeLog +++ b/gcc/ada/ChangeLog @@ -1,3 +1,34 @@ +2025-01-05 Estevan Castilho (Tevo) + + * libgnarl/s-taprop__dummy.adb: Remove use clause for + System.Parameters. + (Unlock): Remove Global_Lock formal parameter. + (Write_Lock): Likewise. + +2025-01-05 Eric Botcazou + + PR ada/118234 + * exp_aggr.adb (Convert_To_Assignments): In the case of a + component association, call Is_Container_Aggregate on the parent's + parent. + (Expand_Array_Aggregate): Likewise. + +2025-01-05 Eric Botcazou + + PR ada/118234 + * exp_aggr.adb (Convert_To_Assignments): Tweak comment. + (Expand_Array_Aggregate): Do not delay the expansion if the parent + node is a container aggregate. + +2025-01-05 Marc Poulhiès + + PR ada/118234 + * exp_aggr.adb (Convert_To_Assignments): Do not mark node for + delayed expansion if parent type has the Aggregate aspect. + * sem_util.adb (Is_Container_Aggregate): Move... + * sem_util.ads (Is_Container_Aggregate): ... here and make it + public. + 2024-12-12 Eric Botcazou PR ada/117996 commit 5a49cbe2db9869a9d3164d686243977406f02af7 Author: GCC Administrator Date: Tue Jan 7 00:21:31 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 197d498df29..0bf88b7cb92 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250106 +20250107 commit f0718f1d7815c7845243a182c66f4a454efbfb72 Author: Jan Hubicka Date: Tue Sep 3 16:26:16 2024 +0200 Zen5 tuning part 3: scheduler tweaks this patch adds support for new fussion in znver5 documented in the optimization manual: The Zen5 microarchitecture adds support to fuse reg-reg MOV Instructions with certain ALU instructions. The following conditions need to be met for fusion to happen: - The MOV should be reg-reg mov with Opcode 0x89 or 0x8B - The MOV is followed by an ALU instruction where the MOV and ALU destination register match. - The ALU instruction may source only registers or immediate data. There cannot be any memory source. - The ALU instruction sources either the source or dest of MOV instruction. - If ALU instruction has 2 reg sources, they should be different. - The following ALU instructions can fuse with an older qualified MOV instruction: ADD ADC AND XOR OP SUB SBB INC DEC NOT SAL / SHL SHR SAR (I assume OP is OR) I also increased issue rate from 4 to 6. Theoretically znver5 can do more, but with our model we can't realy use it. Increasing issue rate to 8 leads to infinite loop in scheduler. Finally, I also enabled fuse_alu_and_branch since it is supported by znver5 (I think by earlier zens too). New fussion pattern moves quite few instructions around in common code: @@ -2210,13 +2210,13 @@ .cfi_offset 3, -32 leaq 63(%rsi), %rbx movq %rbx, %rbp + shrq $6, %rbp + salq $3, %rbp subq $16, %rsp .cfi_def_cfa_offset 48 movq %rdi, %r12 - shrq $6, %rbp - movq %rsi, 8(%rsp) - salq $3, %rbp movq %rbp, %rdi + movq %rsi, 8(%rsp) call _Znwm movq 8(%rsp), %rsi movl $0, 8(%r12) @@ -2224,8 +2224,8 @@ movq %rax, (%r12) movq %rbp, 32(%r12) testq %rsi, %rsi - movq %rsi, %rdx cmovns %rsi, %rbx + movq %rsi, %rdx sarq $63, %rdx shrq $58, %rdx sarq $6, %rbx which should help decoder bandwidth and perhaps also cache, though I was not able to measure off-noise effect on SPEC. gcc/ChangeLog: * config/i386/i386.h (TARGET_FUSE_MOV_AND_ALU): New tune. * config/i386/x86-tune-sched.cc (ix86_issue_rate): Updat for znver5. (ix86_adjust_cost): Add TODO about znver5 memory latency. (ix86_fuse_mov_alu_p): New. (ix86_macro_fusion_pair_p): Use it. * config/i386/x86-tune.def (X86_TUNE_FUSE_ALU_AND_BRANCH): Add ZNVER5. (X86_TUNE_FUSE_MOV_AND_ALU): New tune; (cherry picked from commit e2125a600552bc6e0329e3f1224eea14804db8d3) diff --git a/gcc/config/i386/i386.h b/gcc/config/i386/i386.h index 2bf294eb172..ed988ca280e 100644 --- a/gcc/config/i386/i386.h +++ b/gcc/config/i386/i386.h @@ -413,6 +413,8 @@ extern unsigned char ix86_tune_features[X86_TUNE_LAST]; ix86_tune_features[X86_TUNE_FUSE_CMP_AND_BRANCH_SOFLAGS] #define TARGET_FUSE_ALU_AND_BRANCH \ ix86_tune_features[X86_TUNE_FUSE_ALU_AND_BRANCH] +#define TARGET_FUSE_MOV_AND_ALU \ + ix86_tune_features[X86_TUNE_FUSE_MOV_AND_ALU] #define TARGET_OPT_AGU ix86_tune_features[X86_TUNE_OPT_AGU] #define TARGET_AVOID_LEA_FOR_ADDR \ ix86_tune_features[X86_TUNE_AVOID_LEA_FOR_ADDR] diff --git a/gcc/config/i386/x86-tune-sched.cc b/gcc/config/i386/x86-tune-sched.cc index ebfde596249..f170f6476ce 100644 --- a/gcc/config/i386/x86-tune-sched.cc +++ b/gcc/config/i386/x86-tune-sched.cc @@ -419,6 +419,8 @@ ix86_adjust_cost (rtx_insn *insn, int dep_type, rtx_insn *dep_insn, int cost, enum attr_unit unit = get_attr_unit (insn); int loadcost; + /* TODO: On znver5 complex addressing modes have + greater latency. */ if (unit == UNIT_INTEGER || unit == UNIT_UNKNOWN) loadcost = 4; else @@ -524,6 +526,60 @@ ix86_macro_fusion_p () return TARGET_FUSE_CMP_AND_BRANCH; } +static bool +ix86_fuse_mov_alu_p (rtx_insn *mov, rtx_insn *alu) +{ + /* Validate mov: + - It should be reg-reg move with opcode 0x89 or 0x8B. */ + rtx set1 = PATTERN (mov); + if (GET_CODE (set1) != SET + || !GENERAL_REG_P (SET_SRC (set1)) + || !GENERAL_REG_P (SET_DEST (set1))) + return false; + rtx reg = SET_DEST (set1); + /* - it should have 0x89 or 0x8B opcode. */ + if (!INTEGRAL_MODE_P (GET_MODE (reg)) + || GET_MODE_SIZE (GET_MODE (reg)) < 2 + || GET_MODE_SIZE (GET_MODE (reg)) > 8) + return false; + /* Validate ALU. */ + if (GET_CODE (PATTERN (alu)) != PARALLEL) + return false; + rtx set2 = XVECEXP (PATTERN (alu), 0, 0); + if (GET_CODE (set2) != SET) + return false; + /* Match one of: + ADD ADC AND XOR OR SUB SBB INC DEC NOT SAL SHL SHR SAR + We also may add insn attribute to handle some of sporadic + case we output those with different RTX expressions. */ + + if (GET_CODE (SET_SRC (set2)) != PLUS + && GET_CODE (SET_SRC (set2)) != MINUS + && GET_CODE (SET_SRC (set2)) != XOR + && GET_CODE (SET_SRC (set2)) != AND + && GET_CODE (SET_SRC (set2)) != IOR + && GET_CODE (SET_SRC (set2)) != NOT + && GET_CODE (SET_SRC (set2)) != ASHIFT + && GET_CODE (SET_SRC (set2)) != ASHIFTRT + && GET_CODE (SET_SRC (set2)) != LSHIFTRT) + return false; + rtx op0 = XEXP (SET_SRC (set2), 0); + rtx op1 = GET_CODE (SET_SRC (set2)) != NOT ? XEXP (SET_SRC (set2), 1) : NULL; + /* One of operands should be register. */ + if (op1 && (!REG_P (op0) || REGNO (op0) != REGNO (reg))) + std::swap (op0, op1); + if (!REG_P (op0) || REGNO (op1) != REGNO (reg)) + return false; + if (op1 + && !REG_P (op1) + && !x86_64_immediate_operand (op1, VOIDmode)) + return false; + /* Only one of two paramters must be move destination. */ + if (op1 && REG_P (op1) && REGNO (op1) == REGNO (reg)) + return false; + return true; +} + /* Check whether current microarchitecture support macro fusion for insn pair "CONDGEN + CONDJMP". Refer to "Intel Architectures Optimization Reference Manual". */ @@ -531,6 +587,9 @@ ix86_macro_fusion_p () bool ix86_macro_fusion_pair_p (rtx_insn *condgen, rtx_insn *condjmp) { + if (TARGET_FUSE_MOV_AND_ALU + && ix86_fuse_mov_alu_p (condgen, condjmp)) + return true; rtx src, dest; enum rtx_code ccode; rtx compare_set = NULL_RTX, test_if, cond; diff --git a/gcc/config/i386/x86-tune.def b/gcc/config/i386/x86-tune.def index 249a239de77..561bd17b6e5 100644 --- a/gcc/config/i386/x86-tune.def +++ b/gcc/config/i386/x86-tune.def @@ -140,8 +140,12 @@ DEF_TUNE (X86_TUNE_FUSE_CMP_AND_BRANCH_SOFLAGS, "fuse_cmp_and_branch_soflags", jump instruction when the alu instruction produces the CCFLAG consumed by the conditional jump instruction. */ DEF_TUNE (X86_TUNE_FUSE_ALU_AND_BRANCH, "fuse_alu_and_branch", - m_SANDYBRIDGE | m_CORE_AVX2 | m_GENERIC) + m_SANDYBRIDGE | m_CORE_AVX2 | m_GENERIC | m_ZNVER5) +/* X86_TUNE_FUSE_MOV_AND_ALU: mov and alu in case mov is reg-reg mov + and the destination is used by alu. alu must be one of + ADD, ADC, AND, XOR, OR, SUB, SBB, INC, DEC, NOT, SAL, SHL, SHR, SAR. */ +DEF_TUNE (X86_TUNE_FUSE_MOV_AND_ALU, "fuse_mov_and_alu", m_ZNVER5) /*****************************************************************************/ /* Function prologue, epilogue and function calling sequences. */ commit 4d7efc031fbd925565b049670bf755aca21bd2e3 Author: Jan Hubicka Date: Tue Sep 3 18:20:34 2024 +0200 Zen5 tuning part 4: update reassocation width Zen5 has 6 instead of 4 ALUs and the integer multiplication can now execute in 3 of them. FP units can do 2 additions and 2 multiplications with latency 2 and 3. This patch updates reassociation width accordingly. This has potential of increasing register pressure but unlike while benchmarking znver1 tuning I did not noticed this actually causing problem on spec, so this patch bumps up reassociation width to 6 for everything except for integer vectors, where there are 4 units with typical latency of 1. Bootstrapped/regtested x86_64-linux, comitted. gcc/ChangeLog: * config/i386/i386.cc (ix86_reassociation_width): Update for Znver5. * config/i386/x86-tune-costs.h (znver5_costs): Update reassociation widths. (cherry picked from commit f0ab3de6ec0e3540f2e57f3f5628005f0a4e3fa5) diff --git a/gcc/config/i386/i386.cc b/gcc/config/i386/i386.cc index 2087f8633eb..ea25e56ad64 100644 --- a/gcc/config/i386/i386.cc +++ b/gcc/config/i386/i386.cc @@ -22923,13 +22923,17 @@ ix86_reassociation_width (unsigned int op, machine_mode mode) if (width == 1) return 1; - /* Integer vector instructions execute in FP unit + /* Znver1-4 Integer vector instructions execute in FP unit and can execute 3 additions and one multiplication per cycle. */ if ((ix86_tune == PROCESSOR_ZNVER1 || ix86_tune == PROCESSOR_ZNVER2 - || ix86_tune == PROCESSOR_ZNVER3 || ix86_tune == PROCESSOR_ZNVER4 - || ix86_tune == PROCESSOR_ZNVER5) + || ix86_tune == PROCESSOR_ZNVER3 || ix86_tune == PROCESSOR_ZNVER4) && INTEGRAL_MODE_P (mode) && op != PLUS && op != MINUS) return 1; + /* Znver5 can do 2 integer multiplications per cycle with latency + of 3. */ + if (ix86_tune == PROCESSOR_ZNVER5 + && INTEGRAL_MODE_P (mode) && op != PLUS && op != MINUS) + width = 6; /* Account for targets that splits wide vectors into multiple parts. */ if (TARGET_AVX512_SPLIT_REGS && GET_MODE_BITSIZE (mode) > 256) diff --git a/gcc/config/i386/x86-tune-costs.h b/gcc/config/i386/x86-tune-costs.h index b8e7ab9372e..0f2308bb079 100644 --- a/gcc/config/i386/x86-tune-costs.h +++ b/gcc/config/i386/x86-tune-costs.h @@ -2068,16 +2068,19 @@ struct processor_costs znver5_cost = { COSTS_N_INSNS (13), /* cost of DIVSD instruction. */ COSTS_N_INSNS (14), /* cost of SQRTSS instruction. */ COSTS_N_INSNS (20), /* cost of SQRTSD instruction. */ - /* Zen can execute 4 integer operations per cycle. FP operations - take 3 cycles and it can execute 2 integer additions and 2 - multiplications thus reassociation may make sense up to with of 6. - SPEC2k6 bencharks suggests - that 4 works better than 6 probably due to register pressure. - - Integer vector operations are taken by FP unit and execute 3 vector - plus/minus operations per cycle but only one multiply. This is adjusted - in ix86_reassociation_width. */ - 4, 4, 3, 6, /* reassoc int, fp, vec_int, vec_fp. */ + /* Zen5 can execute: + - integer ops: 6 per cycle, at most 3 multiplications. + latency 1 for additions, 3 for multiplications (pipelined) + + Setting width of 9 for multiplication is probably excessive + for register pressure. + - fp ops: 2 additions per cycle, latency 2-3 + 2 multiplicaitons per cycle, latency 3 + - vector intger ops: 4 additions, latency 1 + 2 multiplications, latency 4 + We increase width to 6 for multiplications + in ix86_reassociation_width. */ + 6, 6, 4, 6, /* reassoc int, fp, vec_int, vec_fp. */ znver2_memcpy, znver2_memset, COSTS_N_INSNS (4), /* cond_taken_branch_cost. */ commit 3cbf7173e6aaed3d1e67c5f6829ffd3cb701e5d4 Author: GCC Administrator Date: Wed Jan 8 00:20:50 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 0892d146c17..61acf3c7dd8 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,25 @@ +2025-01-07 Jan Hubicka + + Backported from master: + 2024-09-03 Jan Hubicka + + * config/i386/i386.cc (ix86_reassociation_width): Update for Znver5. + * config/i386/x86-tune-costs.h (znver5_costs): Update reassociation + widths. + +2025-01-07 Jan Hubicka + + Backported from master: + 2024-09-03 Jan Hubicka + + * config/i386/i386.h (TARGET_FUSE_MOV_AND_ALU): New tune. + * config/i386/x86-tune-sched.cc (ix86_issue_rate): Updat for znver5. + (ix86_adjust_cost): Add TODO about znver5 memory latency. + (ix86_fuse_mov_alu_p): New. + (ix86_macro_fusion_pair_p): Use it. + * config/i386/x86-tune.def (X86_TUNE_FUSE_ALU_AND_BRANCH): Add ZNVER5. + (X86_TUNE_FUSE_MOV_AND_ALU): New tune; + 2024-12-05 Georg-Johann Lay Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 0bf88b7cb92..de7be3a30f5 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250107 +20250108 commit f3f351ad6e9eca32a542b78fdaff032b9905654e Author: GCC Administrator Date: Thu Jan 9 00:20:03 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index de7be3a30f5..f8b56893859 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250108 +20250109 commit f075683d812e6987e27523f6640f121cdd9f0195 Author: GCC Administrator Date: Fri Jan 10 00:20:08 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index f8b56893859..a6b5049790e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250109 +20250110 commit 36db43d30282521732736ba5da23c397fb51f1e0 Author: Sam James Date: Wed Jan 1 17:16:17 2025 +0000 doc: cpp: fix version test example syntax gcc/ChangeLog: * doc/cpp.texi (Common Predefined Macros): Fix syntax. diff --git a/gcc/doc/cpp.texi b/gcc/doc/cpp.texi index 90b2767e39a..fe92522a7b7 100644 --- a/gcc/doc/cpp.texi +++ b/gcc/doc/cpp.texi @@ -1970,7 +1970,7 @@ like this: #if __GNUC__ > 3 || \ (__GNUC__ == 3 && (__GNUC_MINOR__ > 2 || \ (__GNUC_MINOR__ == 2 && \ - __GNUC_PATCHLEVEL__ > 0)) + __GNUC_PATCHLEVEL__ > 0))) @end smallexample @noindent commit 5d6afc601443fa5c03680fb7f39b7dc1f36766a8 Author: Richard Biener Date: Tue Jun 25 16:13:02 2024 +0200 tree-optimization/115646 - ICE with pow shrink-wrapping from bitfield The following makes analysis and transform agree on constraints. PR tree-optimization/115646 * tree-call-cdce.cc (check_pow): Check for bit_sz values as allowed by transform. * gcc.dg/pr115646.c: New testcase. (cherry picked from commit 453b1d291d1a0f89087ad91cf6b1bed1ec68eff3) diff --git a/gcc/testsuite/gcc.dg/pr115646.c b/gcc/testsuite/gcc.dg/pr115646.c new file mode 100644 index 00000000000..7938a309513 --- /dev/null +++ b/gcc/testsuite/gcc.dg/pr115646.c @@ -0,0 +1,14 @@ +/* { dg-do compile } */ +/* { dg-options "-O2" } */ +/* { dg-require-effective-target int32plus } */ + +extern double pow(double x, double y); + +struct S { + unsigned int a : 3, b : 8, c : 21; +}; + +void foo (struct S *p) +{ + pow (p->c, 42); +} diff --git a/gcc/tree-call-cdce.cc b/gcc/tree-call-cdce.cc index 83991fe373e..91878129835 100644 --- a/gcc/tree-call-cdce.cc +++ b/gcc/tree-call-cdce.cc @@ -260,7 +260,7 @@ check_pow (gcall *pow_call) /* If the type of the base is too wide, the resulting shrink wrapping condition will be too conservative. */ - if (bit_sz > MAX_BASE_INT_BIT_SIZE) + if (bit_sz != 8 && bit_sz != 16 && bit_sz != MAX_BASE_INT_BIT_SIZE) return false; return true; commit 882f7e53a7664f2c76b10dc020e720ba9f55f022 Author: Richard Biener Date: Thu Jun 27 11:26:08 2024 +0200 tree-optimization/115669 - fix SLP reduction association The following avoids associating a reduction path as that might get STMT_VINFO_REDUC_IDX out-of-sync with the SLP operand order. This is a latent issue with SLP reductions but now easily exposed as we're doing single-lane SLP reductions. When we achieved SLP only we can move and update this meta-data. PR tree-optimization/115669 * tree-vect-slp.cc (vect_build_slp_tree_2): Do not reassociate chains that participate in a reduction. * gcc.dg/vect/pr115669.c: New testcase. (cherry picked from commit 7886830bb45c4f5dca0496d4deae9a45204d78f5) diff --git a/gcc/testsuite/gcc.dg/vect/pr115669.c b/gcc/testsuite/gcc.dg/vect/pr115669.c new file mode 100644 index 00000000000..361a17a64e6 --- /dev/null +++ b/gcc/testsuite/gcc.dg/vect/pr115669.c @@ -0,0 +1,22 @@ +/* { dg-additional-options "-fwrapv" } */ + +#include "tree-vect.h" + +int a = 10; +unsigned b; +long long c[100]; +int foo() +{ + long long *d = c; + for (short e = 0; e < a; e++) + b += ~(d ? d[e] : 0); + return b; +} + +int main() +{ + check_vect (); + if (foo () != -10) + abort (); + return 0; +} diff --git a/gcc/tree-vect-slp.cc b/gcc/tree-vect-slp.cc index 19cab93761c..0462fa01020 100644 --- a/gcc/tree-vect-slp.cc +++ b/gcc/tree-vect-slp.cc @@ -1825,6 +1825,9 @@ vect_build_slp_tree_2 (vec_info *vinfo, slp_tree node, else if (is_a (vinfo) /* ??? We don't handle !vect_internal_def defs below. */ && STMT_VINFO_DEF_TYPE (stmt_info) == vect_internal_def + /* ??? Do not associate a reduction, this will wreck REDUC_IDX + mapping as long as that exists on the stmt_info level. */ + && STMT_VINFO_REDUC_IDX (stmt_info) == -1 && is_gimple_assign (stmt_info->stmt) && (associative_tree_code (gimple_assign_rhs_code (stmt_info->stmt)) || gimple_assign_rhs_code (stmt_info->stmt) == MINUS_EXPR) commit c8b549857d968d634a74709112e5acc9f9caf35c Author: Richard Biener Date: Wed Jul 24 13:16:35 2024 +0200 tree-optimization/116057 - wrong code with CCP and vector CTORs The following fixes an issue with CCPs likely_value when faced with a vector CTOR containing undef SSA names and constants. This should be classified as CONSTANT and not UNDEFINED. PR tree-optimization/116057 * tree-ssa-ccp.cc (likely_value): Also walk CTORs in stmt operands to look for constants. * gcc.dg/torture/pr116057.c: New testcase. (cherry picked from commit 1ea551514b9c285d801ac5ab8d78b22483ff65af) diff --git a/gcc/testsuite/gcc.dg/torture/pr116057.c b/gcc/testsuite/gcc.dg/torture/pr116057.c new file mode 100644 index 00000000000..a7021c8e746 --- /dev/null +++ b/gcc/testsuite/gcc.dg/torture/pr116057.c @@ -0,0 +1,20 @@ +/* { dg-do run } */ +/* { dg-additional-options "-Wno-psabi" } */ + +#define vect8 __attribute__((vector_size(8))) + +vect8 int __attribute__((noipa)) +f(int a) +{ + int b; + vect8 int t={1,1}; + if(a) return t; + return (vect8 int){0, b}; +} + +int main () +{ + if (f(0)[0] != 0) + __builtin_abort (); + return 0; +} diff --git a/gcc/tree-ssa-ccp.cc b/gcc/tree-ssa-ccp.cc index 3c63f2dd8a3..629cb3c2d82 100644 --- a/gcc/tree-ssa-ccp.cc +++ b/gcc/tree-ssa-ccp.cc @@ -750,6 +750,17 @@ likely_value (gimple *stmt) continue; if (is_gimple_min_invariant (op)) has_constant_operand = true; + else if (TREE_CODE (op) == CONSTRUCTOR) + { + unsigned j; + tree val; + FOR_EACH_CONSTRUCTOR_VALUE (CONSTRUCTOR_ELTS (op), j, val) + if (CONSTANT_CLASS_P (val)) + { + has_constant_operand = true; + break; + } + } } if (has_constant_operand) commit ce8a50c872488a715434a015ca1dd99703caeef8 Author: GCC Administrator Date: Sat Jan 11 00:21:21 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 61acf3c7dd8..1bb03cf7883 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,34 @@ +2025-01-10 Richard Biener + + Backported from master: + 2024-07-24 Richard Biener + + PR tree-optimization/116057 + * tree-ssa-ccp.cc (likely_value): Also walk CTORs in stmt + operands to look for constants. + +2025-01-10 Richard Biener + + Backported from master: + 2024-06-27 Richard Biener + + PR tree-optimization/115669 + * tree-vect-slp.cc (vect_build_slp_tree_2): Do not reassociate + chains that participate in a reduction. + +2025-01-10 Richard Biener + + Backported from master: + 2024-06-26 Richard Biener + + PR tree-optimization/115646 + * tree-call-cdce.cc (check_pow): Check for bit_sz values + as allowed by transform. + +2025-01-10 Sam James + + * doc/cpp.texi (Common Predefined Macros): Fix syntax. + 2025-01-07 Jan Hubicka Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index a6b5049790e..cbd205340e3 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250110 +20250111 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 5b58977a7a1..bd0203f16e9 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,27 @@ +2025-01-10 Richard Biener + + Backported from master: + 2024-07-24 Richard Biener + + PR tree-optimization/116057 + * gcc.dg/torture/pr116057.c: New testcase. + +2025-01-10 Richard Biener + + Backported from master: + 2024-06-27 Richard Biener + + PR tree-optimization/115669 + * gcc.dg/vect/pr115669.c: New testcase. + +2025-01-10 Richard Biener + + Backported from master: + 2024-06-26 Richard Biener + + PR tree-optimization/115646 + * gcc.dg/pr115646.c: New testcase. + 2024-12-12 Eric Botcazou * gnat.dg/matrix1.adb: New test. commit 1a2cfde875463e2f068f57b735f0512c6bde8260 Author: GCC Administrator Date: Sun Jan 12 00:19:38 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index cbd205340e3..71203ea20bc 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250111 +20250112 commit 166cea6886435c18b557666e40b5fd3805fc1036 Author: GCC Administrator Date: Mon Jan 13 00:19:09 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 71203ea20bc..aafaca3fe8e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250112 +20250113 commit b95b3406c1f7ea603495d71f62b21a9d2c8e15b9 Author: Harald Anlauf Date: Thu Jan 2 20:22:23 2025 +0100 Fortran: Cray pointer comparison wrongly optimized away [PR106692] PR fortran/106692 gcc/fortran/ChangeLog: * trans-expr.cc (gfc_conv_expr_op): Inhibit excessive optimization of Cray pointers by treating them as volatile in comparisons. gcc/testsuite/ChangeLog: * gfortran.dg/cray_pointers_13.f90: New test. (cherry picked from commit c7754a2fb2e60987524947fe189f3ffac035ea1d) diff --git a/gcc/fortran/trans-expr.cc b/gcc/fortran/trans-expr.cc index 54cf246fd0d..7a7a7be1793 100644 --- a/gcc/fortran/trans-expr.cc +++ b/gcc/fortran/trans-expr.cc @@ -3923,6 +3923,19 @@ gfc_conv_expr_op (gfc_se * se, gfc_expr * expr) if (lop) { + // Inhibit overeager optimization of Cray pointer comparisons (PR106692). + if (expr->value.op.op1->expr_type == EXPR_VARIABLE + && expr->value.op.op1->ts.type == BT_INTEGER + && expr->value.op.op1->symtree + && expr->value.op.op1->symtree->n.sym->attr.cray_pointer) + TREE_THIS_VOLATILE (lse.expr) = 1; + + if (expr->value.op.op2->expr_type == EXPR_VARIABLE + && expr->value.op.op2->ts.type == BT_INTEGER + && expr->value.op.op2->symtree + && expr->value.op.op2->symtree->n.sym->attr.cray_pointer) + TREE_THIS_VOLATILE (rse.expr) = 1; + /* The result of logical ops is always logical_type_node. */ tmp = fold_build2_loc (input_location, code, logical_type_node, lse.expr, rse.expr); diff --git a/gcc/testsuite/gfortran.dg/cray_pointers_13.f90 b/gcc/testsuite/gfortran.dg/cray_pointers_13.f90 new file mode 100644 index 00000000000..766d24546ab --- /dev/null +++ b/gcc/testsuite/gfortran.dg/cray_pointers_13.f90 @@ -0,0 +1,51 @@ +! { dg-do run } +! { dg-additional-options "-fcray-pointer" } +! +! PR fortran/106692 - Cray pointer comparison wrongly optimized away +! +! Contributed by Marek Polacek + +program test + call test_cray() + call test_cray2() +end + +subroutine test_cray() + pointer(ptrzz1 , zz1) + ptrzz1=0 + if (ptrzz1 .ne. 0) then + print *, "test_cray: ptrzz1=", ptrzz1 + stop 1 + else + call shape_cray(zz1) + end if +end + +subroutine shape_cray(zz1) + pointer(ptrzz , zz) + ptrzz=loc(zz1) + if (ptrzz .ne. 0) then + print *, "shape_cray: ptrzz=", ptrzz + stop 3 + end if +end + +subroutine test_cray2() + pointer(ptrzz1 , zz1) + ptrzz1=0 + if (0 == ptrzz1) then + call shape_cray2(zz1) + else + print *, "test_cray2: ptrzz1=", ptrzz1 + stop 2 + end if +end + +subroutine shape_cray2(zz1) + pointer(ptrzz , zz) + ptrzz=loc(zz1) + if (.not. (0 == ptrzz)) then + print *, "shape_cray2: ptrzz=", ptrzz + stop 4 + end if +end commit 836ecab09fbe91e2aa16ee87eb9aeef53512612b Author: GCC Administrator Date: Tue Jan 14 00:20:26 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index aafaca3fe8e..0907fc0d8ad 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250113 +20250114 diff --git a/gcc/fortran/ChangeLog b/gcc/fortran/ChangeLog index 170af6090ac..7cc60f76995 100644 --- a/gcc/fortran/ChangeLog +++ b/gcc/fortran/ChangeLog @@ -1,3 +1,12 @@ +2025-01-13 Harald Anlauf + + Backported from master: + 2025-01-03 Harald Anlauf + + PR fortran/106692 + * trans-expr.cc (gfc_conv_expr_op): Inhibit excessive optimization + of Cray pointers by treating them as volatile in comparisons. + 2024-11-30 Paul Thomas Backported from master: diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index bd0203f16e9..8660ef078df 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2025-01-13 Harald Anlauf + + Backported from master: + 2025-01-03 Harald Anlauf + + PR fortran/106692 + * gfortran.dg/cray_pointers_13.f90: New test. + 2025-01-10 Richard Biener Backported from master: commit 3911b6366ee49dffe2f16578093b49664b3a2d72 Author: Jan Hubicka Date: Wed Sep 4 09:19:08 2024 +0200 Zen5 tuning part 5: update instruction latencies in x86-tune-costs there is nothing exciting in this patch. I measured latencies and also compared them with newly released optimization guide. There are no dramatic changes compared to zen4. One interesting new bit is that addss is faster and can be 2 cycles when fed by another addss. I also increased the large insn bound since decoders seems no longer require instructions to be 8 bytes or less. gcc/ChangeLog: * config/i386/x86-tune-costs.h (znver5_cost): Update instruction costs. (cherry picked from commit 4292297a0f938ffc953422fa246ff00fe345fe3d) diff --git a/gcc/config/i386/x86-tune-costs.h b/gcc/config/i386/x86-tune-costs.h index 0f2308bb079..6bf09342feb 100644 --- a/gcc/config/i386/x86-tune-costs.h +++ b/gcc/config/i386/x86-tune-costs.h @@ -2002,6 +2002,7 @@ struct processor_costs znver5_cost = { COSTS_N_INSNS (1), /* cost of a lea instruction. */ COSTS_N_INSNS (1), /* variable shift costs. */ COSTS_N_INSNS (1), /* constant shift costs. */ + /* mul has latency 3, executes in 3 integer units. */ {COSTS_N_INSNS (3), /* cost of starting multiply for QI. */ COSTS_N_INSNS (3), /* HI. */ COSTS_N_INSNS (3), /* SI. */ @@ -2009,6 +2010,8 @@ struct processor_costs znver5_cost = { COSTS_N_INSNS (3)}, /* other. */ 0, /* cost of multiply per each bit set. */ + /* integer divide has latency of 8 cycles + plus 1 for every 9 bits of quotient. */ {COSTS_N_INSNS (10), /* cost of a divide/mod for QI. */ COSTS_N_INSNS (11), /* HI. */ COSTS_N_INSNS (13), /* SI. */ @@ -2016,7 +2019,7 @@ struct processor_costs znver5_cost = { COSTS_N_INSNS (16)}, /* other. */ COSTS_N_INSNS (1), /* cost of movsx. */ COSTS_N_INSNS (1), /* cost of movzx. */ - 8, /* "large" insn. */ + 15, /* "large" insn. */ 9, /* MOVE_RATIO. */ 6, /* CLEAR_RATIO */ {6, 6, 6}, /* cost of loading integer registers @@ -2033,12 +2036,13 @@ struct processor_costs znver5_cost = { 2, 2, 2, /* cost of moving XMM,YMM,ZMM register. */ 6, /* cost of moving SSE register to integer. */ - /* VGATHERDPD is 17 uops and throughput is 4, VGATHERDPS is 24 uops, - throughput 5. Approx 7 uops do not depend on vector size and every load - is 5 uops. */ + + /* TODO: gather and scatter instructions are currently disabled in + x86-tune.def. In some cases they are however a win, see PR116582 + We however need good cost model for them. */ 14, 10, /* Gather load static, per_elt. */ 14, 20, /* Gather store static, per_elt. */ - 32, /* size of l1 cache. */ + 48, /* size of l1 cache. */ 1024, /* size of l2 cache. */ 64, /* size of prefetch block. */ /* New AMD processors never drop prefetches; if they cannot be performed @@ -2048,6 +2052,8 @@ struct processor_costs znver5_cost = { time). */ 100, /* number of parallel prefetches. */ 3, /* Branch cost. */ + /* TODO x87 latencies are still based on znver4. + Probably not very important these days. */ COSTS_N_INSNS (7), /* cost of FADD and FSUB insns. */ COSTS_N_INSNS (7), /* cost of FMUL instruction. */ /* Latency of fdiv is 8-15. */ @@ -2057,16 +2063,24 @@ struct processor_costs znver5_cost = { /* Latency of fsqrt is 4-10. */ COSTS_N_INSNS (25), /* cost of FSQRT instruction. */ + /* SSE instructions have typical throughput 4 and latency 1. */ COSTS_N_INSNS (1), /* cost of cheap SSE instruction. */ - COSTS_N_INSNS (3), /* cost of ADDSS/SD SUBSS/SD insns. */ + /* ADDSS has throughput 2 and latency 2 + (in some cases when source is another addition). */ + COSTS_N_INSNS (2), /* cost of ADDSS/SD SUBSS/SD insns. */ + /* MULSS has throughput 2 and latency 3. */ COSTS_N_INSNS (3), /* cost of MULSS instruction. */ COSTS_N_INSNS (3), /* cost of MULSD instruction. */ + /* FMA had throughput 2 and latency 4. */ COSTS_N_INSNS (4), /* cost of FMA SS instruction. */ COSTS_N_INSNS (4), /* cost of FMA SD instruction. */ + /* DIVSS has throughtput 0.4 and latency 10. */ COSTS_N_INSNS (10), /* cost of DIVSS instruction. */ - /* 9-13. */ + /* DIVSD has throughtput 0.25 and latency 13. */ COSTS_N_INSNS (13), /* cost of DIVSD instruction. */ + /* DIVSD has throughtput 0.22 and latency 14. */ COSTS_N_INSNS (14), /* cost of SQRTSS instruction. */ + /* DIVSD has throughtput 0.13 and latency 20. */ COSTS_N_INSNS (20), /* cost of SQRTSD instruction. */ /* Zen5 can execute: - integer ops: 6 per cycle, at most 3 multiplications. commit 2e31832f610bb78d3dfce44e8eb1fefa4aade81d Author: GCC Administrator Date: Wed Jan 15 00:19:25 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 1bb03cf7883..9b455cc3871 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,11 @@ +2025-01-14 Jan Hubicka + + Backported from master: + 2024-09-04 Jan Hubicka + + * config/i386/x86-tune-costs.h (znver5_cost): Update instruction + costs. + 2025-01-10 Richard Biener Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 0907fc0d8ad..956862fd33e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250114 +20250115 commit f7e3214ccc0ba51c6a9b0df9fecbf13652c4f31c Author: GCC Administrator Date: Thu Jan 16 00:22:28 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 956862fd33e..629b78beae4 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250115 +20250116 commit d5acaa088e9da95d3f99c12e2d8c80bc8403f9e5 Author: GCC Administrator Date: Fri Jan 17 00:22:41 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 629b78beae4..51d36d43d1b 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250116 +20250117 commit 3fd1c1a8640bf84c32695531a3ed81d4338409aa Author: Richard Biener Date: Mon Jan 9 12:46:28 2023 +0100 middle-end/69482 - not preserving volatile accesses The following addresses a long standing issue with not preserving accesses to non-volatile objects through volatile qualified pointers in the case that object gets expanded to a register. The fix is to treat accesses to an object with a volatile qualified access as forcing that object to memory. This issue got more exposed recently so it regressed more since GCC 11. PR middle-end/69482 * cfgexpand.cc (discover_nonconstant_array_refs_r): Volatile qualified accesses also force objects to memory. * gcc.target/i386/pr69482-1.c: New testcase. * gcc.target/i386/pr69482-2.c: Likewise. (cherry picked from commit a5a8242153d078f1ebe60f00409415da260a29ee) diff --git a/gcc/cfgexpand.cc b/gcc/cfgexpand.cc index eadec9a2bfd..006f46fa9db 100644 --- a/gcc/cfgexpand.cc +++ b/gcc/cfgexpand.cc @@ -6311,6 +6311,15 @@ discover_nonconstant_array_refs_r (tree * tp, int *walk_subtrees, if (IS_TYPE_OR_DECL_P (t)) *walk_subtrees = 0; + else if (REFERENCE_CLASS_P (t) && TREE_THIS_VOLATILE (t)) + { + t = get_base_address (t); + if (t && DECL_P (t) + && DECL_MODE (t) != BLKmode + && !TREE_ADDRESSABLE (t)) + bitmap_set_bit (forced_stack_vars, DECL_UID (t)); + *walk_subtrees = 0; + } else if (TREE_CODE (t) == ARRAY_REF || TREE_CODE (t) == ARRAY_RANGE_REF) { while (((TREE_CODE (t) == ARRAY_REF || TREE_CODE (t) == ARRAY_RANGE_REF) diff --git a/gcc/testsuite/gcc.target/i386/pr69482-1.c b/gcc/testsuite/gcc.target/i386/pr69482-1.c new file mode 100644 index 00000000000..7ef0e71b17c --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr69482-1.c @@ -0,0 +1,16 @@ +/* { dg-do compile } */ +/* { dg-options "-O3 -fno-stack-protector -fomit-frame-pointer" } */ + +static inline void memset_s(void* s, int n) { + volatile unsigned char * p = s; + for(int i = 0; i < n; ++i) { + p[i] = 0; + } +} + +void test() { + unsigned char x[4]; + memset_s(x, sizeof x); +} + +/* { dg-final { scan-assembler-times "mov" 4 } } */ diff --git a/gcc/testsuite/gcc.target/i386/pr69482-2.c b/gcc/testsuite/gcc.target/i386/pr69482-2.c new file mode 100644 index 00000000000..6aabe4fb393 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr69482-2.c @@ -0,0 +1,10 @@ +/* { dg-do compile } */ +/* { dg-options "-O2 -fomit-frame-pointer" } */ + +void bar () +{ + int j; + *(volatile int *)&j = 0; +} + +/* { dg-final { scan-assembler-times "mov" 1 } } */ commit 567547a89763b583e8b829b81f1db9053c246b96 Author: Richard Biener Date: Sun Oct 13 15:12:44 2024 +0200 tree-optimization/116290 - fix compare-debug issue in ldist Loop distribution does different analysis with -g0/-g due to counting a debug stmt starting a BB against a limit which will everntually lead to different IVOPTs choices. I've fixed a possible IVOPTs issue on the way even though it doesn't make a difference here. PR tree-optimization/116290 * tree-loop-distribution.cc (determine_reduction_stmt_1): PHIs have no debug variants. Start with first non-debug real stmt. * tree-ssa-loop-ivopts.cc (find_givs_in_bb): Do not analyze debug stmts. * gcc.dg/pr116290.c: New testcase. (cherry picked from commit 566740013b3445162b8c4bc2205e4e568d014968) diff --git a/gcc/testsuite/gcc.dg/pr116290.c b/gcc/testsuite/gcc.dg/pr116290.c new file mode 100644 index 00000000000..97b946bda89 --- /dev/null +++ b/gcc/testsuite/gcc.dg/pr116290.c @@ -0,0 +1,18 @@ +/* { dg-do compile } */ +/* { dg-options "-g -O2 -fcompare-debug" } */ + +char *camel_message_info_class_intern_init_part; +void g_once_init_enter(); +void camel_message_info_class_intern_init() { + int ii; + char *label; + for (; camel_message_info_class_intern_init_part[ii]; ii++) + if (camel_message_info_class_intern_init_part) { + if (label && *label) + g_once_init_enter(); + label = &camel_message_info_class_intern_init_part[ii + 1]; + camel_message_info_class_intern_init_part[ii] = ' '; + } + if (label) + g_once_init_enter(); +} diff --git a/gcc/tree-loop-distribution.cc b/gcc/tree-loop-distribution.cc index 1b7d2a1ea7d..449b9ffd438 100644 --- a/gcc/tree-loop-distribution.cc +++ b/gcc/tree-loop-distribution.cc @@ -3492,7 +3492,7 @@ determine_reduction_stmt_1 (const loop_p loop, const basic_block *bbs) basic_block bb = bbs[i]; for (gphi_iterator bsi = gsi_start_phis (bb); !gsi_end_p (bsi); - gsi_next_nondebug (&bsi)) + gsi_next (&bsi)) { gphi *phi = bsi.phi (); if (virtual_operand_p (gimple_phi_result (phi))) @@ -3505,8 +3505,8 @@ determine_reduction_stmt_1 (const loop_p loop, const basic_block *bbs) } } - for (gimple_stmt_iterator bsi = gsi_start_bb (bb); !gsi_end_p (bsi); - gsi_next_nondebug (&bsi), ++ninsns) + for (gimple_stmt_iterator bsi = gsi_start_nondebug_bb (bb); + !gsi_end_p (bsi); gsi_next_nondebug (&bsi), ++ninsns) { /* Bail out early for loops which are unlikely to match. */ if (ninsns > 16) diff --git a/gcc/tree-ssa-loop-ivopts.cc b/gcc/tree-ssa-loop-ivopts.cc index eb4d8388906..597c4b0e8a6 100644 --- a/gcc/tree-ssa-loop-ivopts.cc +++ b/gcc/tree-ssa-loop-ivopts.cc @@ -1458,7 +1458,8 @@ find_givs_in_bb (struct ivopts_data *data, basic_block bb) gimple_stmt_iterator bsi; for (bsi = gsi_start_bb (bb); !gsi_end_p (bsi); gsi_next (&bsi)) - find_givs_in_stmt (data, gsi_stmt (bsi)); + if (!is_gimple_debug (gsi_stmt (bsi))) + find_givs_in_stmt (data, gsi_stmt (bsi)); } /* Finds general ivs. */ commit 4e5e70e645f02bc956e1ac529ec6498eb281ec26 Author: Richard Biener Date: Thu Sep 19 14:58:18 2024 +0200 tree-optimization/116768 - wrong dependence analysis The following reverts a bogus fix done for PR101009 and instead makes sure we get into the same_access_functions () case when computing the distance vector for g[1] and g[1] where the constants ended up having different types. The generic code doesn't seem to handle loop invariant dependences. The special case gets us both ( 0 ) and ( 1 ) as distance vectors while formerly we got ( 1 ), which the PR101009 fix changed to ( 0 ) with bad effects on other cases as shown in this PR. PR tree-optimization/116768 * tree-data-ref.cc (build_classic_dist_vector_1): Revert PR101009 change. * tree-chrec.cc (eq_evolutions_p): Make sure (sizetype)1 and (int)1 compare equal. * gcc.dg/torture/pr116768.c: New testcase. (cherry picked from commit 5b5a36b122e1205449f1512bf39521b669e713ef) diff --git a/gcc/testsuite/gcc.dg/torture/pr116768.c b/gcc/testsuite/gcc.dg/torture/pr116768.c new file mode 100644 index 00000000000..57b5d00e7b7 --- /dev/null +++ b/gcc/testsuite/gcc.dg/torture/pr116768.c @@ -0,0 +1,32 @@ +/* { dg-do run } */ + +#define numwords 2 + +typedef struct { + unsigned words[numwords]; +} Child; + +typedef struct { + Child child; +} Parent; + +Parent my_or(Parent x, const Parent *y) { + const Child *y_child = &y->child; + for (int i = 0; i < numwords; i++) { + x.child.words[i] |= y_child->words[i]; + } + return x; +} + +int main() { + Parent bs[4]; + __builtin_memset(bs, 0, sizeof(bs)); + + bs[0].child.words[0] = 1; + for (int i = 1; i <= 3; i++) { + bs[i] = my_or(bs[i], &bs[i - 1]); + } + if (bs[2].child.words[0] != 1) + __builtin_abort (); + return 0; +} diff --git a/gcc/tree-chrec.cc b/gcc/tree-chrec.cc index c44cea75434..685e975cf2b 100644 --- a/gcc/tree-chrec.cc +++ b/gcc/tree-chrec.cc @@ -1594,7 +1594,7 @@ eq_evolutions_p (const_tree chrec0, const_tree chrec1) || TREE_CODE (chrec0) != TREE_CODE (chrec1)) return false; - if (chrec0 == chrec1) + if (operand_equal_p (chrec0, chrec1, 0)) return true; if (! types_compatible_p (TREE_TYPE (chrec0), TREE_TYPE (chrec1))) @@ -1621,7 +1621,7 @@ eq_evolutions_p (const_tree chrec0, const_tree chrec1) TREE_OPERAND (chrec1, 0)); default: - return operand_equal_p (chrec0, chrec1, 0); + return false; } } diff --git a/gcc/tree-data-ref.cc b/gcc/tree-data-ref.cc index b7bca6a9d06..ea55f81a92f 100644 --- a/gcc/tree-data-ref.cc +++ b/gcc/tree-data-ref.cc @@ -5168,8 +5168,6 @@ build_classic_dist_vector_1 (struct data_dependence_relation *ddr, non_affine_dependence_relation (ddr); return false; } - else - *init_b = true; } return true; commit c0c5db68331ea8814db45ecd90ba842818e623fa Author: Richard Biener Date: Mon Oct 14 08:11:22 2024 +0200 middle-end/116891 - fix (negate (IFN_FNMS@3 @0 @1 @2)) -> (IFN_FMA @0 @1 @2) Transforming -fma (-a, b, -c) to fma (a, b, c) is only valid when not rounding towards -inf or +inf as the sign of the multiplication changes. PR middle-end/116891 * match.pd ((negate (IFN_FNMS@3 @0 @1 @2)) -> (IFN_FMA @0 @1 @2)): Only enable for !HONOR_SIGN_DEPENDENT_ROUNDING. (cherry picked from commit c53bd48c6920bc1f4039b6682aafbf414a600e47) diff --git a/gcc/match.pd b/gcc/match.pd index 45ed3420510..d32338c2780 100644 --- a/gcc/match.pd +++ b/gcc/match.pd @@ -7083,7 +7083,7 @@ DEFINE_INT_AND_FLOAT_ROUND_FN (RINT) (IFN_FMA @0 @1 @2)) (simplify (negate (IFN_FNMS@3 @0 @1 @2)) - (if (single_use (@3)) + (if (!HONOR_SIGN_DEPENDENT_ROUNDING (type) && single_use (@3)) (IFN_FMA @0 @1 @2)))) /* CLZ simplifications. */ commit 3af4410b7aed19e4fdcfef1bc669d067e51342b3 Author: Jakub Jelinek Date: Tue Oct 15 19:38:46 2024 +0200 match.pd: Further fma negation fixes [PR116891] On Mon, Oct 14, 2024 at 08:53:29AM +0200, Jakub Jelinek wrote: > > PR middle-end/116891 > > * match.pd ((negate (IFN_FNMS@3 @0 @1 @2)) -> (IFN_FMA @0 @1 @2)): > > Only enable for !HONOR_SIGN_DEPENDENT_ROUNDING. > > Guess it would be nice to have a testcase which FAILs without the patch and > PASSes with it, but it can be added later. I've added such a testcase now, and additionally found the fix only fixed one of the 4 problematic similar cases. Here is a patch which fixes the others too and adds the testcases. fma-pr116891.c FAILed without your patch, FAILs with your patch too (but only due to the bar/baz/qux checks) and PASSes with the patch. 2024-10-15 Jakub Jelinek PR middle-end/116891 * match.pd ((negate (fmas@3 @0 @1 @2)) -> (IFN_FNMS @0 @1 @2)): Only enable for !HONOR_SIGN_DEPENDENT_ROUNDING. ((negate (IFN_FMS@3 @0 @1 @2)) -> (IFN_FNMA @0 @1 @2)): Likewise. ((negate (IFN_FNMA@3 @0 @1 @2)) -> (IFN_FMS @0 @1 @2)): Likewise. * gcc.dg/pr116891.c: New test. * gcc.target/i386/fma-pr116891.c: New test. (cherry picked from commit 4366f0c7e296ea0d7279343c9b0a1d597588a1da) diff --git a/gcc/match.pd b/gcc/match.pd index d32338c2780..06149a74086 100644 --- a/gcc/match.pd +++ b/gcc/match.pd @@ -7041,7 +7041,7 @@ DEFINE_INT_AND_FLOAT_ROUND_FN (RINT) (IFN_FNMS @0 @1 @2)) (simplify (negate (fmas@3 @0 @1 @2)) - (if (single_use (@3)) + (if (!HONOR_SIGN_DEPENDENT_ROUNDING (type) && single_use (@3)) (IFN_FNMS @0 @1 @2)))) (simplify @@ -7055,7 +7055,7 @@ DEFINE_INT_AND_FLOAT_ROUND_FN (RINT) (IFN_FNMA @0 @1 @2)) (simplify (negate (IFN_FMS@3 @0 @1 @2)) - (if (single_use (@3)) + (if (!HONOR_SIGN_DEPENDENT_ROUNDING (type) && single_use (@3)) (IFN_FNMA @0 @1 @2))) (simplify @@ -7069,7 +7069,7 @@ DEFINE_INT_AND_FLOAT_ROUND_FN (RINT) (IFN_FMS @0 @1 @2)) (simplify (negate (IFN_FNMA@3 @0 @1 @2)) - (if (single_use (@3)) + (if (!HONOR_SIGN_DEPENDENT_ROUNDING (type) && single_use (@3)) (IFN_FMS @0 @1 @2))) (simplify diff --git a/gcc/testsuite/gcc.dg/pr116891.c b/gcc/testsuite/gcc.dg/pr116891.c new file mode 100644 index 00000000000..446e5ec5a4a --- /dev/null +++ b/gcc/testsuite/gcc.dg/pr116891.c @@ -0,0 +1,47 @@ +/* PR middle-end/116891 */ +/* { dg-do run } */ +/* { dg-require-effective-target fenv } */ +/* { dg-require-effective-target hard_float } */ +/* { dg-require-effective-target c99_runtime } */ +/* { dg-options "-O2 -frounding-math" } */ + +#include + +__attribute__((noipa)) double +foo (double x, double y, double z) +{ + return -__builtin_fma (-x, y, -z); +} + +__attribute__((noipa)) double +bar (double x, double y, double z) +{ + return -__builtin_fma (-x, y, z); +} + +__attribute__((noipa)) double +baz (double x, double y, double z) +{ + return -__builtin_fma (x, y, -z); +} + +__attribute__((noipa)) double +qux (double x, double y, double z) +{ + return -__builtin_fma (x, y, z); +} + +int +main () +{ +#if defined (FE_DOWNWARD) && __DBL_MANT_DIG__ == 53 && __DBL_MAX_EXP__ == 1024 + fesetround (FE_DOWNWARD); + double a = foo (-0x1.p256, 0x1.p256, 0x1.p-256); + if (a != -__builtin_nextafter (0x1p256 * 0x1p256, 0.)) + __builtin_abort (); + if (a != bar (-0x1.p256, 0x1.p256, -0x1.p-256) + || a != baz (0x1.p256, 0x1.p256, 0x1.p-256) + || a != qux (0x1.p256, 0x1.p256, -0x1.p-256)) + __builtin_abort (); +#endif +} diff --git a/gcc/testsuite/gcc.target/i386/fma-pr116891.c b/gcc/testsuite/gcc.target/i386/fma-pr116891.c new file mode 100644 index 00000000000..34689f44c41 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/fma-pr116891.c @@ -0,0 +1,19 @@ +/* PR middle-end/116891 */ +/* { dg-do run } */ +/* { dg-require-effective-target fenv } */ +/* { dg-require-effective-target hard_float } */ +/* { dg-require-effective-target c99_runtime } */ +/* { dg-require-effective-target fma } */ +/* { dg-options "-O2 -mfma -frounding-math" } */ + +#include +#include "fma-check.h" + +#define main() do_main () +#include "../../gcc.dg/pr116891.c" + +static void +fma_test (void) +{ + do_main (); +} commit 72d7bdbdd9fe1e575b7ad6da0fd7587627912ddb Author: Richard Biener Date: Sat Oct 12 14:51:37 2024 +0200 tree-optimization/117104 - add missed guards to max(a,b) != a simplification For vector types we have to make sure the comparison result is a vector type and the resulting compare operation is supported. As the resulting compare is never an equality compare I didn't bother to check for the cbranch case. PR tree-optimization/117104 * match.pd ((cmp:c (minmax:c @0 @1) @0) -> (out @0 @1)): Properly guard the vector case. * gcc.dg/pr117104.c: New testcase. (cherry picked from commit f54d42e00007e7a558b273d87f95b3e5b1938f5a) diff --git a/gcc/match.pd b/gcc/match.pd index 06149a74086..9b7c9f3f0d5 100644 --- a/gcc/match.pd +++ b/gcc/match.pd @@ -3255,7 +3255,11 @@ DEFINE_INT_AND_FLOAT_ROUND_FN (RINT) out (le gt ge lt ) (simplify (cmp:c (minmax:c @0 @1) @0) - (if (ANY_INTEGRAL_TYPE_P (TREE_TYPE (@0))) + (if (ANY_INTEGRAL_TYPE_P (TREE_TYPE (@0)) + && (!VECTOR_TYPE_P (TREE_TYPE (@0)) + || (VECTOR_TYPE_P (type) + && (!expand_vec_cmp_expr_p (TREE_TYPE (@0), type, cmp) + || expand_vec_cmp_expr_p (TREE_TYPE (@0), type, out))))) (out @0 @1)))) /* MIN (X, 5) == 0 -> X == 0 MIN (X, 5) == 7 -> false */ diff --git a/gcc/testsuite/gcc.dg/pr117104.c b/gcc/testsuite/gcc.dg/pr117104.c new file mode 100644 index 00000000000..9aa5734f792 --- /dev/null +++ b/gcc/testsuite/gcc.dg/pr117104.c @@ -0,0 +1,12 @@ +/* { dg-do compile } */ +/* { dg-options "-O2 -fno-vect-cost-model" } */ +/* { dg-additional-options "-mavx" { target { x86_64-*-* i?86-*-* } } } */ + +void g(); +void f(long *a) +{ + long b0 = a[0] > 0 ? a[0] : 0; + long b1 = a[1] > 0 ? a[1] : 0; + if ((b0|b1) == 0) + g(); +} commit 2b12998ff9370f4a0aa8c0562f933c2e7efdb41d Author: Richard Biener Date: Tue Oct 22 11:46:47 2024 +0200 tree-optimization/117254 - ICE with access diangostics The diagnostics code fails to handle non-constant domain max. PR tree-optimization/117254 * gimple-ssa-warn-access.cc (maybe_warn_nonstring_arg): Check the array domain max is constant before using it. * gcc.dg/pr117254.c: New testcase. (cherry picked from commit d464a52d0678dfea523a60efe8b792ba1b8d40db) diff --git a/gcc/gimple-ssa-warn-access.cc b/gcc/gimple-ssa-warn-access.cc index e70a6f1fb87..8e12e49816f 100644 --- a/gcc/gimple-ssa-warn-access.cc +++ b/gcc/gimple-ssa-warn-access.cc @@ -604,7 +604,8 @@ maybe_warn_nonstring_arg (tree fndecl, GimpleOrTree exp) { if (tree arrbnd = TYPE_DOMAIN (type)) { - if ((arrbnd = TYPE_MAX_VALUE (arrbnd))) + if ((arrbnd = TYPE_MAX_VALUE (arrbnd)) + && TREE_CODE (arrbnd) == INTEGER_CST) { asize = wi::to_offset (arrbnd) + 1; known_size = true; diff --git a/gcc/testsuite/gcc.dg/pr117254.c b/gcc/testsuite/gcc.dg/pr117254.c new file mode 100644 index 00000000000..c7a510677f1 --- /dev/null +++ b/gcc/testsuite/gcc.dg/pr117254.c @@ -0,0 +1,10 @@ +/* { dg-do compile } */ +/* { dg-options "" } */ + +int g; +void e(int s) { + struct { + __attribute__((nonstring)) char bn[g]; + } f; + __builtin_strncpy (f.bn, f.bn, s); +} commit 39d3c3f06f969b6dbe5c50596c060d1210f8c9f0 Author: Richard Biener Date: Mon Oct 28 09:52:08 2024 +0100 tree-optimization/117307 - STMT_VINFO_SLP_VECT_ONLY mis-computation STMT_VINFO_SLP_VECT_ONLY isn't properly computed as union of all group members and when the group is later split due to duplicates not all sub-groups inherit the flag. PR tree-optimization/117307 * tree-vect-data-refs.cc (vect_analyze_data_ref_accesses): Properly compute STMT_VINFO_SLP_VECT_ONLY. Set it on all parts of a split group. * gcc.dg/vect/pr117307.c: New testcase. (cherry picked from commit 19722308a286d9a00eead8ac82b948da8c4ca38b) diff --git a/gcc/testsuite/gcc.dg/vect/pr117307.c b/gcc/testsuite/gcc.dg/vect/pr117307.c new file mode 100644 index 00000000000..dc853d61706 --- /dev/null +++ b/gcc/testsuite/gcc.dg/vect/pr117307.c @@ -0,0 +1,17 @@ +/* { dg-do compile } */ +/* { dg-additional-options "-march=x86-64-v4" { target { x86_64-*-* i?86-*-* } } } */ + +int a; +float *b, *c; +float d; +void e() { + for (; a; a++) { + if (d) { + c[0] = b[0]; + c[1] = b[1]; + } else if (b[1]) + c[0] = b[0] * 0; + b += 2; + c += 2; + } +} diff --git a/gcc/tree-vect-data-refs.cc b/gcc/tree-vect-data-refs.cc index b47d0122aec..7af01c6806e 100644 --- a/gcc/tree-vect-data-refs.cc +++ b/gcc/tree-vect-data-refs.cc @@ -3255,12 +3255,15 @@ vect_analyze_data_ref_accesses (vec_info *vinfo, DR_GROUP_NEXT_ELEMENT (lastinfo) = stmtinfo_b; lastinfo = stmtinfo_b; - STMT_VINFO_SLP_VECT_ONLY (stmtinfo_a) - = !can_group_stmts_p (stmtinfo_a, stmtinfo_b, false); + if (! STMT_VINFO_SLP_VECT_ONLY (stmtinfo_a)) + { + STMT_VINFO_SLP_VECT_ONLY (stmtinfo_a) + = !can_group_stmts_p (stmtinfo_a, stmtinfo_b, false); - if (dump_enabled_p () && STMT_VINFO_SLP_VECT_ONLY (stmtinfo_a)) - dump_printf_loc (MSG_NOTE, vect_location, - "Load suitable for SLP vectorization only.\n"); + if (dump_enabled_p () && STMT_VINFO_SLP_VECT_ONLY (stmtinfo_a)) + dump_printf_loc (MSG_NOTE, vect_location, + "Load suitable for SLP vectorization only.\n"); + } if (init_b == init_prev && !to_fixup.add (DR_GROUP_FIRST_ELEMENT (stmtinfo_a)) @@ -3304,7 +3307,11 @@ vect_analyze_data_ref_accesses (vec_info *vinfo, { DR_GROUP_NEXT_ELEMENT (g) = DR_GROUP_NEXT_ELEMENT (next); if (!newgroup) - newgroup = next; + { + newgroup = next; + STMT_VINFO_SLP_VECT_ONLY (newgroup) + = STMT_VINFO_SLP_VECT_ONLY (grp); + } else DR_GROUP_NEXT_ELEMENT (ng) = next; ng = next; commit a5d05111bde1e07824fd85d3a97785022ae30b8a Author: Richard Biener Date: Tue Nov 12 11:15:15 2024 +0100 tree-optimization/117417 - ICE with complex load optimization When we decompose a complex load only used as real and imaginary parts we fail to honor IL constraints which are that a BIT_FIELD_REF of register type should be outermost in a ref. The following simply avoids the transform when the complex load has such a BIT_FIELD_REF. PR tree-optimization/117417 * tree-ssa-forwprop.cc (pass_forwprop::execute): Avoid decomposing BIT_FIELD_REF complex load. * gcc.dg/torture/pr117417.c: New testcase. (cherry picked from commit d976daa931642d940b7b27032ca6139210c07eed) diff --git a/gcc/testsuite/gcc.dg/torture/pr117417.c b/gcc/testsuite/gcc.dg/torture/pr117417.c new file mode 100644 index 00000000000..2c80dd5d77c --- /dev/null +++ b/gcc/testsuite/gcc.dg/torture/pr117417.c @@ -0,0 +1,16 @@ +/* { dg-do compile } */ + +typedef __attribute__((__vector_size__ (8))) double V; +int bar (int a, V *p) +{ + V v; + v = *p; + a += *(_Complex short *) &v; + return a; +} +V x; +int +foo () +{ + return bar (0, &x); +} diff --git a/gcc/tree-ssa-forwprop.cc b/gcc/tree-ssa-forwprop.cc index 062f5667f14..1b36ffb0bbe 100644 --- a/gcc/tree-ssa-forwprop.cc +++ b/gcc/tree-ssa-forwprop.cc @@ -3542,9 +3542,9 @@ pass_forwprop::execute (function *fun) else if (TREE_CODE (TREE_TYPE (lhs)) == COMPLEX_TYPE && gimple_assign_load_p (stmt) && !gimple_has_volatile_ops (stmt) - && (TREE_CODE (gimple_assign_rhs1 (stmt)) - != TARGET_MEM_REF) - && !stmt_can_throw_internal (cfun, stmt)) + && TREE_CODE (rhs) != TARGET_MEM_REF + && TREE_CODE (rhs) != BIT_FIELD_REF + && !stmt_can_throw_internal (fun, stmt)) { /* Rewrite loads used only in real/imagpart extractions to component-wise loads. */ commit 9d2f4c8b0f849839ff1b9b7d5f9e734271c03634 Author: Nathaniel Shead Date: Fri Dec 20 22:09:39 2024 +1100 c++: Allow pragmas in NSDMIs [PR118147] This patch removes the (unnecessary) CPP_PRAGMA_EOL case from cp_parser_cache_defarg, which currently has the result that any pragmas in the NSDMI cause an error. PR c++/118147 gcc/cp/ChangeLog: * parser.cc (cp_parser_cache_defarg): Don't error when CPP_PRAGMA_EOL. gcc/testsuite/ChangeLog: * g++.dg/cpp0x/nsdmi-defer7.C: New test. Signed-off-by: Nathaniel Shead (cherry picked from commit f3ccc57e5f044031a1b07e79330de9220e93afe7) diff --git a/gcc/cp/parser.cc b/gcc/cp/parser.cc index 9c1bd32cff1..64cefb6b4e4 100644 --- a/gcc/cp/parser.cc +++ b/gcc/cp/parser.cc @@ -34110,7 +34110,6 @@ cp_parser_cache_defarg (cp_parser *parser, bool nsdmi) /* If we run out of tokens, issue an error message. */ case CPP_EOF: - case CPP_PRAGMA_EOL: error_at (token->location, "file ends in default argument"); return error_mark_node; diff --git a/gcc/testsuite/g++.dg/cpp0x/nsdmi-defer7.C b/gcc/testsuite/g++.dg/cpp0x/nsdmi-defer7.C new file mode 100644 index 00000000000..3bef636ccbd --- /dev/null +++ b/gcc/testsuite/g++.dg/cpp0x/nsdmi-defer7.C @@ -0,0 +1,13 @@ +// PR c++/118147 +// { dg-do compile { target c++11 } } + +struct F { + int i = []{ + #pragma message "test" // { dg-message "test" } + return 1; + }(); +}; + +struct G { + int i = + #pragma GCC diagnostic push // { dg-error "file ends in default argument|expected" } commit d180e392d7a8ba1346bfe7580de953234f0c2f9d Author: Eugene Rozenfeld Date: Fri Jan 10 19:48:52 2025 -0800 Fix setting of call graph node AutoFDO count We are initializing both the call graph node count and the entry block count of the function with the head_count value from the profile. Count propagation algorithm may refine the entry block count and we may end up with a case where the call graph node count is set to zero but the entry block count is non-zero. That becomes a problem because we have this code in execute_fixup_cfg: profile_count num = node->count; profile_count den = ENTRY_BLOCK_PTR_FOR_FN (cfun)->count; bool scale = num.initialized_p () && !(num == den); Here if num is 0 but den is not 0, scale becomes true and we lose the counts in if (scale) bb->count = bb->count.apply_scale (num, den); This is what happened in the issue reported in PR116743 (a 10% regression in MySQL HAMMERDB tests). 3d9e6767939e9658260e2506e81ec32b37cba041 made an improvement in AutoFDO count propagation, which caused a mismatch between the call graph node count (zero) and the entry block count (non-zero) and subsequent loss of counts as described above. The fix is to update the call graph node count once we've done count propagation. Tested on x86_64-pc-linux-gnu. gcc/ChangeLog: PR gcov-profile/116743 * auto-profile.cc (afdo_annotate_cfg): Fix mismatch between the call graph node count and the entry block count. (cherry picked from commit e683c6b029f809c7a1981b4341c95d9652c22e18) diff --git a/gcc/auto-profile.cc b/gcc/auto-profile.cc index 2b34b80b82d..89dd05252c2 100644 --- a/gcc/auto-profile.cc +++ b/gcc/auto-profile.cc @@ -1537,8 +1537,6 @@ afdo_annotate_cfg (const stmt_set &promoted_stmts) if (s == NULL) return; - cgraph_node::get (current_function_decl)->count - = profile_count::from_gcov_type (s->head_count ()).afdo (); ENTRY_BLOCK_PTR_FOR_FN (cfun)->count = profile_count::from_gcov_type (s->head_count ()).afdo (); EXIT_BLOCK_PTR_FOR_FN (cfun)->count = profile_count::zero ().afdo (); @@ -1577,6 +1575,8 @@ afdo_annotate_cfg (const stmt_set &promoted_stmts) /* Calculate, propagate count and probability information on CFG. */ afdo_calculate_branch_prob (&annotated_bb); } + cgraph_node::get(current_function_decl)->count + = ENTRY_BLOCK_PTR_FOR_FN(cfun)->count; update_max_bb_count (); profile_status_for_fn (cfun) = PROFILE_READ; if (flag_value_profile_transformations) commit a5ea6f44a9ae2a12c3781bae0dd53f30f83dd7b0 Author: GCC Administrator Date: Sat Jan 18 00:20:52 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 9b455cc3871..6c1b3704571 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,100 @@ +2025-01-17 Eugene Rozenfeld + + Backported from master: + 2025-01-16 Eugene Rozenfeld + + PR gcov-profile/116743 + * auto-profile.cc (afdo_annotate_cfg): Fix mismatch between the call graph node count + and the entry block count. + +2025-01-17 Richard Biener + + Backported from master: + 2024-11-12 Richard Biener + + PR tree-optimization/117417 + * tree-ssa-forwprop.cc (pass_forwprop::execute): Avoid + decomposing BIT_FIELD_REF complex load. + +2025-01-17 Richard Biener + + Backported from master: + 2024-10-28 Richard Biener + + PR tree-optimization/117307 + * tree-vect-data-refs.cc (vect_analyze_data_ref_accesses): + Properly compute STMT_VINFO_SLP_VECT_ONLY. Set it on all + parts of a split group. + +2025-01-17 Richard Biener + + Backported from master: + 2024-10-22 Richard Biener + + PR tree-optimization/117254 + * gimple-ssa-warn-access.cc (maybe_warn_nonstring_arg): + Check the array domain max is constant before using it. + +2025-01-17 Richard Biener + + Backported from master: + 2024-10-12 Richard Biener + + PR tree-optimization/117104 + * match.pd ((cmp:c (minmax:c @0 @1) @0) -> (out @0 @1)): Properly + guard the vector case. + +2025-01-17 Jakub Jelinek + + Backported from master: + 2024-10-15 Jakub Jelinek + + PR middle-end/116891 + * match.pd ((negate (fmas@3 @0 @1 @2)) -> (IFN_FNMS @0 @1 @2)): + Only enable for !HONOR_SIGN_DEPENDENT_ROUNDING. + ((negate (IFN_FMS@3 @0 @1 @2)) -> (IFN_FNMA @0 @1 @2)): Likewise. + ((negate (IFN_FNMA@3 @0 @1 @2)) -> (IFN_FMS @0 @1 @2)): Likewise. + +2025-01-17 Richard Biener + + Backported from master: + 2024-10-14 Richard Biener + + PR middle-end/116891 + * match.pd ((negate (IFN_FNMS@3 @0 @1 @2)) -> (IFN_FMA @0 @1 @2)): + Only enable for !HONOR_SIGN_DEPENDENT_ROUNDING. + +2025-01-17 Richard Biener + + Backported from master: + 2024-09-19 Richard Biener + + PR tree-optimization/116768 + * tree-data-ref.cc (build_classic_dist_vector_1): Revert + PR101009 change. + * tree-chrec.cc (eq_evolutions_p): Make sure (sizetype)1 + and (int)1 compare equal. + +2025-01-17 Richard Biener + + Backported from master: + 2024-10-14 Richard Biener + + PR tree-optimization/116290 + * tree-loop-distribution.cc (determine_reduction_stmt_1): PHIs + have no debug variants. Start with first non-debug real stmt. + * tree-ssa-loop-ivopts.cc (find_givs_in_bb): Do not analyze + debug stmts. + +2025-01-17 Richard Biener + + Backported from master: + 2023-01-09 Richard Biener + + PR middle-end/69482 + * cfgexpand.cc (discover_nonconstant_array_refs_r): Volatile + qualified accesses also force objects to memory. + 2025-01-14 Jan Hubicka Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 51d36d43d1b..d05d499d8a4 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250117 +20250118 diff --git a/gcc/cp/ChangeLog b/gcc/cp/ChangeLog index f7830dd119a..970bb9c21f5 100644 --- a/gcc/cp/ChangeLog +++ b/gcc/cp/ChangeLog @@ -1,3 +1,12 @@ +2025-01-17 Nathaniel Shead + + Backported from master: + 2025-01-17 Nathaniel Shead + + PR c++/118147 + * parser.cc (cp_parser_cache_defarg): Don't error when + CPP_PRAGMA_EOL. + 2024-12-05 Simon Martin Backported from master: diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 8660ef078df..61c38629f5b 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,77 @@ +2025-01-17 Nathaniel Shead + + Backported from master: + 2025-01-17 Nathaniel Shead + + PR c++/118147 + * g++.dg/cpp0x/nsdmi-defer7.C: New test. + +2025-01-17 Richard Biener + + Backported from master: + 2024-11-12 Richard Biener + + PR tree-optimization/117417 + * gcc.dg/torture/pr117417.c: New testcase. + +2025-01-17 Richard Biener + + Backported from master: + 2024-10-28 Richard Biener + + PR tree-optimization/117307 + * gcc.dg/vect/pr117307.c: New testcase. + +2025-01-17 Richard Biener + + Backported from master: + 2024-10-22 Richard Biener + + PR tree-optimization/117254 + * gcc.dg/pr117254.c: New testcase. + +2025-01-17 Richard Biener + + Backported from master: + 2024-10-12 Richard Biener + + PR tree-optimization/117104 + * gcc.dg/pr117104.c: New testcase. + +2025-01-17 Jakub Jelinek + + Backported from master: + 2024-10-15 Jakub Jelinek + + PR middle-end/116891 + * gcc.dg/pr116891.c: New test. + * gcc.target/i386/fma-pr116891.c: New test. + +2025-01-17 Richard Biener + + Backported from master: + 2024-09-19 Richard Biener + + PR tree-optimization/116768 + * gcc.dg/torture/pr116768.c: New testcase. + +2025-01-17 Richard Biener + + Backported from master: + 2024-10-14 Richard Biener + + PR tree-optimization/116290 + * gcc.dg/pr116290.c: New testcase. + +2025-01-17 Richard Biener + + Backported from master: + 2023-01-09 Richard Biener + + PR middle-end/69482 + * gcc.target/i386/pr69482-1.c: New testcase. + * gcc.target/i386/pr69482-2.c: Likewise. + 2025-01-13 Harald Anlauf Backported from master: commit 51c1abd6cd63963e1149d0864dda5d0ab3bd4c95 Author: GCC Administrator Date: Sun Jan 19 00:20:22 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d05d499d8a4..df2e7f62043 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250118 +20250119 commit 94fdc545fb03daf261694902bc50e39b44d86191 Author: GCC Administrator Date: Mon Jan 20 00:20:46 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index df2e7f62043..1216e7215b8 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250119 +20250120 commit 7bb462dd2a6a6551d142e7ad983fa2afd1df9253 Author: Simon Martin Date: Sun Jan 5 10:36:47 2025 +0100 c++: Friend classes don't shadow enclosing template class paramater [PR118255] We currently reject the following code === code here === template struct S { friend class non_template; }; class non_template {}; S<0> s; === code here === While EDG agrees with the current behaviour, clang and MSVC don't (see https://godbolt.org/z/69TGaabhd), and I believe that this code is valid, since the friend clause does not actually declare a type, so it cannot shadow anything. The fact that we didn't error out if the non_template class was declared before S backs this up as well. This patch fixes this by skipping the call to check_template_shadow for hidden bindings. PR c++/118255 gcc/cp/ChangeLog: * name-lookup.cc (pushdecl): Don't call check_template_shadow for hidden bindings. gcc/testsuite/ChangeLog: * g++.dg/lookup/pr99116-1.C: Adjust test expectation. * g++.dg/template/friend84.C: New test. (cherry picked from commit b5a069203fc074ab75d994c4a7e0f2db6a0a00fd) diff --git a/gcc/cp/name-lookup.cc b/gcc/cp/name-lookup.cc index 48c7badc865..41bd4dd8dca 100644 --- a/gcc/cp/name-lookup.cc +++ b/gcc/cp/name-lookup.cc @@ -3741,7 +3741,10 @@ pushdecl (tree decl, bool hiding) if (old && anticipated_builtin_p (old)) old = OVL_CHAIN (old); - check_template_shadow (decl); + if (hiding) + ; /* Hidden bindings don't shadow anything. */ + else + check_template_shadow (decl); if (DECL_DECLARES_FUNCTION_P (decl)) { diff --git a/gcc/testsuite/g++.dg/lookup/pr99116-1.C b/gcc/testsuite/g++.dg/lookup/pr99116-1.C index 01b483ea915..efee3e4aca3 100644 --- a/gcc/testsuite/g++.dg/lookup/pr99116-1.C +++ b/gcc/testsuite/g++.dg/lookup/pr99116-1.C @@ -2,7 +2,7 @@ template struct Z { - friend struct T; // { dg-error "shadows template parameter" } + friend struct T; // { dg-bogus "shadows template parameter" } }; struct Y { diff --git a/gcc/testsuite/g++.dg/template/friend84.C b/gcc/testsuite/g++.dg/template/friend84.C new file mode 100644 index 00000000000..64ea41a552b --- /dev/null +++ b/gcc/testsuite/g++.dg/template/friend84.C @@ -0,0 +1,26 @@ +// PR c++/118255 +// { dg-do "compile" } + +// The PR's case, that used to error out. +template +struct S { + friend class non_template; // { dg-bogus "shadows template parameter" } +}; + +class non_template {}; +S<0> s; + +// We already accepted cases where the friend is already declared. +template +struct T { + friend class non_template; +}; +T<0> t; + +// We should reject (re)declarations. +template +struct U { + class non_template {}; // { dg-error "shadows template parameter" } + void non_template () {} // { dg-error "shadows template parameter" } +}; +U<0> u; commit 9a1efd1ee2509abb93878bd911d8c07143b10e33 Author: Uros Bizjak Date: Mon Jan 20 16:19:43 2025 +0100 i386: Disable SImode/DImode moves from/to mask regs without avx512bw [PR118067] SImode and DImode moves from/to mask registers are valid only with AVX512BW, so mark relevant alternatives in *movsi_internal and *movdi_internal as such. PR target/118067 gcc/ChangeLog: * config/i386/i386.md (*movdi_internal): Disable alternatives from/to mask registers without AVX512BW. (*movsi_internal): Ditto. diff --git a/gcc/config/i386/i386.md b/gcc/config/i386/i386.md index 4142d7fec4d..911edddaec9 100644 --- a/gcc/config/i386/i386.md +++ b/gcc/config/i386/i386.md @@ -2251,12 +2251,16 @@ [(set (attr "isa") (cond [(eq_attr "alternative" "0,1,17,18") (const_string "nox64") - (eq_attr "alternative" "2,3,4,5,10,11,23,25") + (eq_attr "alternative" "2,3,4,5,10,11") (const_string "x64") (eq_attr "alternative" "19,20") (const_string "x64_sse2") (eq_attr "alternative" "21,22") (const_string "sse2") + (eq_attr "alternative" "23,25") + (const_string "x64_avx512bw") + (eq_attr "alternative" "24,26,27") + (const_string "avx512bw") ] (const_string "*"))) (set (attr "type") @@ -2467,6 +2471,8 @@ [(set (attr "isa") (cond [(eq_attr "alternative" "12,13") (const_string "sse2") + (eq_attr "alternative" "14,15,16,17") + (const_string "avx512bw") ] (const_string "*"))) (set (attr "type") commit f4b8c08c86237e7eb29defb2f8ae3bff22581e5f Author: Iain Buclaw Date: Mon Jan 20 20:01:03 2025 +0100 d: Fix failing test with 32-bit compiler [PR114434] Since the introduction of gdc.test/runnable/test23514.d, it's exposed an incorrect compilation when adding a 64-bit constant to a link-time address. The current cast to size_t causes a loss of precision, which can result in incorrect compilation. PR d/114434 gcc/d/ChangeLog: * expr.cc (ExprVisitor::visit (PtrExp *)): Get the offset as a dinteger_t rather than a size_t. (ExprVisitor::visit (SymOffExp *)): Likewise. gcc/testsuite/ChangeLog: * gdc.test/runnable/test23514.d: New test. (cherry picked from commit 9ab38952a2033d6d4a8e31c3c4d2ab1a25a406c6) diff --git a/gcc/d/expr.cc b/gcc/d/expr.cc index 7afd98975b1..c308a068fc1 100644 --- a/gcc/d/expr.cc +++ b/gcc/d/expr.cc @@ -1537,7 +1537,7 @@ public: void visit (PtrExp *e) { Type *tnext = NULL; - size_t offset; + dinteger_t offset; tree result; if (e->e1->op == EXP::add) @@ -2115,7 +2115,7 @@ public: void visit (SymOffExp *e) { /* Build the address and offset of the symbol. */ - size_t soffset = e->isSymOffExp ()->offset; + dinteger_t soffset = e->isSymOffExp ()->offset; tree result = get_decl_tree (e->var); TREE_USED (result) = 1; diff --git a/gcc/testsuite/gdc.test/runnable/test23514.d b/gcc/testsuite/gdc.test/runnable/test23514.d new file mode 100644 index 00000000000..1ba7e218d52 --- /dev/null +++ b/gcc/testsuite/gdc.test/runnable/test23514.d @@ -0,0 +1,13 @@ +// DISABLED: win64 +// https://issues.dlang.org/show_bug.cgi?id=23514 + +// Note: this test is disabled on Win64 because of an issue with the Windows +// MS-COFF backend causing it to fail. + +enum ulong offset = 0xFFFF_FFFF_0000_0000UL; + +void main() +{ + ulong voffset = offset; + assert((cast(ulong)&main + voffset) == (cast(ulong)&main + offset)); +} commit 8aaddf0573f1e9ff204d4cc95f41d789277ba346 Author: GCC Administrator Date: Tue Jan 21 00:20:51 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 6c1b3704571..87d56a93812 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,10 @@ +2025-01-20 Uros Bizjak + + PR target/118067 + * config/i386/i386.md (*movdi_internal): + Disable alternatives from/to mask registers without AVX512BW. + (*movsi_internal): Ditto. + 2025-01-17 Eugene Rozenfeld Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 1216e7215b8..309e1782f83 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250120 +20250121 diff --git a/gcc/cp/ChangeLog b/gcc/cp/ChangeLog index 970bb9c21f5..2ba52d0408d 100644 --- a/gcc/cp/ChangeLog +++ b/gcc/cp/ChangeLog @@ -1,3 +1,12 @@ +2025-01-20 Simon Martin + + Backported from master: + 2025-01-17 Simon Martin + + PR c++/118255 + * name-lookup.cc (pushdecl): Don't call check_template_shadow + for hidden bindings. + 2025-01-17 Nathaniel Shead Backported from master: diff --git a/gcc/d/ChangeLog b/gcc/d/ChangeLog index d696ba51a6d..35d8394da3e 100644 --- a/gcc/d/ChangeLog +++ b/gcc/d/ChangeLog @@ -1,3 +1,13 @@ +2025-01-20 Iain Buclaw + + Backported from master: + 2025-01-20 Iain Buclaw + + PR d/114434 + * expr.cc (ExprVisitor::visit (PtrExp *)): Get the offset as a + dinteger_t rather than a size_t. + (ExprVisitor::visit (SymOffExp *)): Likewise. + 2024-06-20 Release Manager * GCC 12.4.0 released. diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 61c38629f5b..22ef7c0786e 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,20 @@ +2025-01-20 Iain Buclaw + + Backported from master: + 2025-01-20 Iain Buclaw + + PR d/114434 + * gdc.test/runnable/test23514.d: New test. + +2025-01-20 Simon Martin + + Backported from master: + 2025-01-17 Simon Martin + + PR c++/118255 + * g++.dg/lookup/pr99116-1.C: Adjust test expectation. + * g++.dg/template/friend84.C: New test. + 2025-01-17 Nathaniel Shead Backported from master: commit e909afe8a8a2924dd6ced6bdf7d8e397f14310b5 Author: Jan Hubicka Date: Tue Sep 3 15:07:41 2024 +0200 Zen5 tuning part 2: disable gather and scatter We disable gathers for zen4. It seems that gather has improved a bit compared to zen4 and Zen5 optimization manual suggests "Avoid GATHER instructions when the indices are known ahead of time. Vector loads followed by shuffles result in a higher load bandwidth." however the situation seems to be more complicated. gather is 5-10% loss on parest benchmark as well as 30% loss on sparse dot products in TSVC. Curiously enough breaking these out into microbenchmark reversed the situation and it turns out that the performance depends on how indices are distributed. gather is loss if indices are sequential, neutral if they are random and win for some strides (4, 8). This seems to be similar to earlier zens, so I think (especially for backporting znver5 support) that it makes sense to be conistent and disable gather unless we work out a good heuristics on when to use it. Since we typically do not know the indices in advance, I don't see how that can be done. I opened PR116582 with some examples of wins and loses gcc/ChangeLog: * config/i386/x86-tune.def (X86_TUNE_USE_GATHER_2PARTS): Disable for ZNVER5. (X86_TUNE_USE_SCATTER_2PARTS): Disable for ZNVER5. (X86_TUNE_USE_GATHER_4PARTS): Disable for ZNVER5. (X86_TUNE_USE_SCATTER_4PARTS): Disable for ZNVER5. (X86_TUNE_USE_GATHER_8PARTS): Disable for ZNVER5. (X86_TUNE_USE_SCATTER_8PARTS): Disable for ZNVER5. (cherry picked from commit d82edbe92eed53a479736fcbbe6d54d0fb42daa4) diff --git a/gcc/config/i386/x86-tune.def b/gcc/config/i386/x86-tune.def index 561bd17b6e5..3fa7501fc72 100644 --- a/gcc/config/i386/x86-tune.def +++ b/gcc/config/i386/x86-tune.def @@ -471,35 +471,32 @@ DEF_TUNE (X86_TUNE_AVOID_4BYTE_PREFIXES, "avoid_4byte_prefixes", /* X86_TUNE_USE_GATHER_2PARTS: Use gather instructions for vectors with 2 elements. */ DEF_TUNE (X86_TUNE_USE_GATHER_2PARTS, "use_gather_2parts", - ~(m_ZNVER1 | m_ZNVER2 | m_ZNVER3 | m_ZNVER4 | m_ALDERLAKE - | m_GENERIC | m_GDS)) + ~(m_ZNVER | m_ALDERLAKE | m_GENERIC | m_GDS)) /* X86_TUNE_USE_SCATTER_2PARTS: Use scater instructions for vectors with 2 elements. */ DEF_TUNE (X86_TUNE_USE_SCATTER_2PARTS, "use_scatter_2parts", - ~(m_ZNVER4)) + ~(m_ZNVER4 | m_ZNVER5)) /* X86_TUNE_USE_GATHER_4PARTS: Use gather instructions for vectors with 4 elements. */ DEF_TUNE (X86_TUNE_USE_GATHER_4PARTS, "use_gather_4parts", - ~(m_ZNVER1 | m_ZNVER2 | m_ZNVER3 | m_ZNVER4 | m_ALDERLAKE - | m_GENERIC | m_GDS)) + ~(m_ZNVER | m_ALDERLAKE | m_GENERIC | m_GDS)) /* X86_TUNE_USE_SCATTER_4PARTS: Use scater instructions for vectors with 4 elements. */ DEF_TUNE (X86_TUNE_USE_SCATTER_4PARTS, "use_scatter_4parts", - ~(m_ZNVER4)) + ~(m_ZNVER4 | m_ZNVER5)) /* X86_TUNE_USE_GATHER: Use gather instructions for vectors with 8 or more elements. */ DEF_TUNE (X86_TUNE_USE_GATHER_8PARTS, "use_gather_8parts", - ~(m_ZNVER1 | m_ZNVER2 | m_ZNVER4 | m_ALDERLAKE - | m_GENERIC | m_GDS)) + ~(m_ZNVER | m_ALDERLAKE | m_GENERIC | m_GDS)) /* X86_TUNE_USE_SCATTER: Use scater instructions for vectors with 8 or more elements. */ DEF_TUNE (X86_TUNE_USE_SCATTER_8PARTS, "use_scatter_8parts", - ~(m_ZNVER4)) + ~(m_ZNVER4 | m_ZNVER5)) /* X86_TUNE_AVOID_128FMA_CHAINS: Avoid creating loops with tight 128bit or smaller FMA chain. */ commit 4d320a7df4b25c2eb060a2a16fee8b993301be55 Author: Iain Buclaw Date: Fri Apr 19 10:51:12 2024 +0200 d: Fix ICE in build_deref, at d/d-codegen.cc:1650 [PR111650] PR d/111650 gcc/d/ChangeLog: * decl.cc (get_fndecl_arguments): Move generation of frame type to ... (DeclVisitor::visit (FuncDeclaration *)): ... here, after the call to build_closure. gcc/testsuite/ChangeLog: * gdc.dg/pr111650.d: New test. (cherry picked from commit 4d4929fe0654d51b52a2bf6e6188d7aad0bf17ac) diff --git a/gcc/d/decl.cc b/gcc/d/decl.cc index a2dd8b84c59..6c2705d9864 100644 --- a/gcc/d/decl.cc +++ b/gcc/d/decl.cc @@ -162,16 +162,6 @@ get_fndecl_arguments (FuncDeclaration *decl) tree parm_decl = get_symbol_decl (decl->vthis); DECL_ARTIFICIAL (parm_decl) = 1; TREE_READONLY (parm_decl) = 1; - - if (decl->vthis->type == Type::tvoidptr) - { - /* Replace generic pointer with back-end closure type - (this wins for gdb). */ - tree frame_type = FRAMEINFO_TYPE (get_frameinfo (decl)); - gcc_assert (frame_type != NULL_TREE); - TREE_TYPE (parm_decl) = build_pointer_type (frame_type); - } - param_list = chainon (param_list, parm_decl); } @@ -1047,6 +1037,16 @@ public: /* May change cfun->static_chain. */ build_closure (d); + /* Replace generic pointer with back-end closure type + (this wins for gdb). */ + if (d->vthis && d->vthis->type == Type::tvoidptr) + { + tree frame_type = FRAMEINFO_TYPE (get_frameinfo (d)); + gcc_assert (frame_type != NULL_TREE); + tree parm_decl = get_symbol_decl (d->vthis); + TREE_TYPE (parm_decl) = build_pointer_type (frame_type); + } + if (d->vresult) declare_local_var (d->vresult); diff --git a/gcc/testsuite/gdc.dg/pr111650.d b/gcc/testsuite/gdc.dg/pr111650.d new file mode 100644 index 00000000000..4298a76d38f --- /dev/null +++ b/gcc/testsuite/gdc.dg/pr111650.d @@ -0,0 +1,21 @@ +// { dg-do compile } +ref V require(K, V)(ref V[K] aa, K key, lazy V value); + +struct Root +{ + ulong[3] f; +} + +Root[ulong] roots; + +Root getRoot(int fd, ulong rootID) +{ + return roots.require(rootID, + { + Root result; + inoLookup(fd, () => result); + return result; + }()); +} + +void inoLookup(int, scope Root delegate()) { } commit e24b17e87899a4b7645db89c9af98632d4946850 Author: GCC Administrator Date: Wed Jan 22 00:22:59 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 87d56a93812..78655a49749 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,16 @@ +2025-01-21 Jan Hubicka + + Backported from master: + 2024-09-03 Jan Hubicka + + * config/i386/x86-tune.def (X86_TUNE_USE_GATHER_2PARTS): Disable for + ZNVER5. + (X86_TUNE_USE_SCATTER_2PARTS): Disable for ZNVER5. + (X86_TUNE_USE_GATHER_4PARTS): Disable for ZNVER5. + (X86_TUNE_USE_SCATTER_4PARTS): Disable for ZNVER5. + (X86_TUNE_USE_GATHER_8PARTS): Disable for ZNVER5. + (X86_TUNE_USE_SCATTER_8PARTS): Disable for ZNVER5. + 2025-01-20 Uros Bizjak PR target/118067 diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 309e1782f83..b0aec664e8d 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250121 +20250122 diff --git a/gcc/d/ChangeLog b/gcc/d/ChangeLog index 35d8394da3e..29b1af50a6d 100644 --- a/gcc/d/ChangeLog +++ b/gcc/d/ChangeLog @@ -1,3 +1,13 @@ +2025-01-21 Iain Buclaw + + Backported from master: + 2024-04-19 Iain Buclaw + + PR d/111650 + * decl.cc (get_fndecl_arguments): Move generation of frame type to ... + (DeclVisitor::visit (FuncDeclaration *)): ... here, after the call to + build_closure. + 2025-01-20 Iain Buclaw Backported from master: diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 22ef7c0786e..7e42b32482a 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2025-01-21 Iain Buclaw + + Backported from master: + 2024-04-19 Iain Buclaw + + PR d/111650 + * gdc.dg/pr111650.d: New test. + 2025-01-20 Iain Buclaw Backported from master: commit 5f010252bb24607b8915358bca634746736e08ba Author: GCC Administrator Date: Thu Jan 23 00:21:48 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index b0aec664e8d..3aae2a0a2ef 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250122 +20250123 commit 40eafb7b58f95a71444d7f4506d9abee54df7de4 Author: John David Anglin Date: Thu Jan 23 14:35:22 2025 -0500 hppa: Fix typo in ADDITIONAL_REGISTER_NAMES in pa32-regs.h 2025-01-23 John David Anglin gcc/ChangeLog: * config/pa/pa32-regs.h (ADDITIONAL_REGISTER_NAMES): Change register 86 name to "%fr31L". diff --git a/gcc/config/pa/pa32-regs.h b/gcc/config/pa/pa32-regs.h index 62084a50abc..8199c476d35 100644 --- a/gcc/config/pa/pa32-regs.h +++ b/gcc/config/pa/pa32-regs.h @@ -344,7 +344,7 @@ enum reg_class { NO_REGS, R1_REGS, GENERAL_REGS, FPUPPER_REGS, FP_REGS, {"%fr16L",56}, {"%fr17L",58}, {"%fr18L",60}, {"%fr19L",62}, \ {"%fr20L",64}, {"%fr21L",66}, {"%fr22L",68}, {"%fr23L",70}, \ {"%fr24L",72}, {"%fr25L",74}, {"%fr26L",76}, {"%fr27L",78}, \ - {"%fr28L",80}, {"%fr29L",82}, {"%fr30L",84}, {"%fr31R",86}, \ + {"%fr28L",80}, {"%fr29L",82}, {"%fr30L",84}, {"%fr31L",86}, \ {"%cr11",88}} #define FP_SAVED_REG_LAST 66 commit 11e11d496ffc3e2e52759bd52c6bd1aa3c4e83c0 Author: GCC Administrator Date: Fri Jan 24 00:21:29 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 78655a49749..2c6c29d1ffd 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,8 @@ +2025-01-23 John David Anglin + + * config/pa/pa32-regs.h (ADDITIONAL_REGISTER_NAMES): Change + register 86 name to "%fr31L". + 2025-01-21 Jan Hubicka Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 3aae2a0a2ef..56a27e9356c 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250123 +20250124 commit 4e508cbb938a8936bc6aefd7823b55107aa4a7f9 Author: Peter Bergner Date: Thu Jan 16 10:49:45 2025 -0600 rs6000: Fix loop limit for built-in constant checking The loop checking for built-in constant operand restrictions was missing some operands due to the loop limit being too small. Fixing that exposed a testsuite failure which is caused by a typo in the pmxvi4ger8pp definition where we had made the PMASK field too small. 2025-01-16 Peter Bergner gcc/ * config/rs6000/rs6000-builtin.cc (rs6000_expand_builtin): Use correct array size for the loop limit. * config/rs6000/rs6000-builtins.def: Fix field size for PMASK operand. (cherry picked from commit 1a2d63a78f99b7fdc2eff5bf9065682d5bbbaaca) diff --git a/gcc/config/rs6000/rs6000-builtin.cc b/gcc/config/rs6000/rs6000-builtin.cc index f01f3aded36..d467db29e8e 100644 --- a/gcc/config/rs6000/rs6000-builtin.cc +++ b/gcc/config/rs6000/rs6000-builtin.cc @@ -3584,7 +3584,7 @@ rs6000_expand_builtin (tree exp, rtx target, rtx /* subtarget */, } /* Check for restricted constant arguments. */ - for (int i = 0; i < 2; i++) + for (size_t i = 0; i < ARRAY_SIZE (bifaddr->restr); i++) { switch (bifaddr->restr[i]) { diff --git a/gcc/config/rs6000/rs6000-builtins.def b/gcc/config/rs6000/rs6000-builtins.def index d2c0565dc62..eca7ad2f5fa 100644 --- a/gcc/config/rs6000/rs6000-builtins.def +++ b/gcc/config/rs6000/rs6000-builtins.def @@ -3861,11 +3861,11 @@ PMXVI4GER8_INTERNAL mma_pmxvi4ger8 {mma} void __builtin_mma_pmxvi4ger8pp (v512 *, vuc, vuc, const int<4>, \ - const int<4>, const int<4>); + const int<4>, const int<8>); PMXVI4GER8PP nothing {mma,quad,mmaint} v512 __builtin_mma_pmxvi4ger8pp_internal (v512, vuc, vuc, const int<4>, \ - const int<4>, const int<4>); + const int<4>, const int<8>); PMXVI4GER8PP_INTERNAL mma_pmxvi4ger8pp {mma,quad} void __builtin_mma_pmxvi8ger4 (v512 *, vuc, vuc, const int<4>, \ commit 4dbace39f1149984e6b85574d4665ce18240db8e Author: Peter Bergner Date: Thu Jan 16 10:53:27 2025 -0600 rs6000: Fix ICE for invalid constants in built-in functions For invalid constant operand values used in built-in functions, return const0_rtx to signify an error occurred during expansion. 2025-01-16 Peter Bergner gcc/ * config/rs6000/rs6000-builtin.cc (rs6000_expand_builtin): Return const0_rtx when there is an error. gcc/testsuite/ * gcc.target/powerpc/mma-builtin-error.c: New test. (cherry picked from commit 0696af74b3392e2178215607337b116d1bb53e34) diff --git a/gcc/config/rs6000/rs6000-builtin.cc b/gcc/config/rs6000/rs6000-builtin.cc index d467db29e8e..a1549fe2a15 100644 --- a/gcc/config/rs6000/rs6000-builtin.cc +++ b/gcc/config/rs6000/rs6000-builtin.cc @@ -3602,7 +3602,7 @@ rs6000_expand_builtin (tree exp, rtx target, rtx /* subtarget */, error ("argument %d must be a literal between 0 and %d," " inclusive", bifaddr->restr_opnd[i], p); - return CONST0_RTX (mode[0]); + return const0_rtx; } break; } @@ -3619,7 +3619,7 @@ rs6000_expand_builtin (tree exp, rtx target, rtx /* subtarget */, " inclusive", bifaddr->restr_opnd[i], bifaddr->restr_val1[i], bifaddr->restr_val2[i]); - return CONST0_RTX (mode[0]); + return const0_rtx; } break; } @@ -3636,7 +3636,7 @@ rs6000_expand_builtin (tree exp, rtx target, rtx /* subtarget */, "between %d and %d, inclusive", bifaddr->restr_opnd[i], bifaddr->restr_val1[i], bifaddr->restr_val2[i]); - return CONST0_RTX (mode[0]); + return const0_rtx; } break; } @@ -3652,7 +3652,7 @@ rs6000_expand_builtin (tree exp, rtx target, rtx /* subtarget */, "literal %d", bifaddr->restr_opnd[i], bifaddr->restr_val1[i], bifaddr->restr_val2[i]); - return CONST0_RTX (mode[0]); + return const0_rtx; } break; } diff --git a/gcc/testsuite/gcc.target/powerpc/mma-builtin-error.c b/gcc/testsuite/gcc.target/powerpc/mma-builtin-error.c new file mode 100644 index 00000000000..a87a1570925 --- /dev/null +++ b/gcc/testsuite/gcc.target/powerpc/mma-builtin-error.c @@ -0,0 +1,11 @@ +/* { dg-do compile } */ +/* { dg-require-effective-target power10_ok } */ +/* { dg-options "-O2 -mdejagnu-cpu=power10" } */ + +typedef unsigned char vec_t __attribute__((vector_size(16))); + +void +foo (__vector_quad *dst, vec_t vec0, vec_t vec1) /* { dg-error "argument 5 must be a literal between 0 and 15, inclusive" } */ +{ + __builtin_mma_pmxvi8ger4 (dst, vec0, vec1, 15, 15, -1); +} commit 809bf0cb21aa10e0ad4648d13201a0ab405972e4 Author: GCC Administrator Date: Sat Jan 25 00:22:01 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 2c6c29d1ffd..c9f267fe0fa 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,20 @@ +2025-01-24 Peter Bergner + + Backported from master: + 2025-01-16 Peter Bergner + + * config/rs6000/rs6000-builtin.cc (rs6000_expand_builtin): Return + const0_rtx when there is an error. + +2025-01-24 Peter Bergner + + Backported from master: + 2025-01-16 Peter Bergner + + * config/rs6000/rs6000-builtin.cc (rs6000_expand_builtin): Use correct + array size for the loop limit. + * config/rs6000/rs6000-builtins.def: Fix field size for PMASK operand. + 2025-01-23 John David Anglin * config/pa/pa32-regs.h (ADDITIONAL_REGISTER_NAMES): Change diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 56a27e9356c..da8bd402f29 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250124 +20250125 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 7e42b32482a..33e17302507 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,10 @@ +2025-01-24 Peter Bergner + + Backported from master: + 2025-01-16 Peter Bergner + + * gcc.target/powerpc/mma-builtin-error.c: New test. + 2025-01-21 Iain Buclaw Backported from master: commit 5032e623d2b767eb4683b67f290b0406fa0ab031 Author: GCC Administrator Date: Sun Jan 26 00:20:34 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index da8bd402f29..1e4e11a933a 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250125 +20250126 commit 71c036f7add6cb46f377321354519a6b8e085328 Author: GCC Administrator Date: Mon Jan 27 00:21:19 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 1e4e11a933a..06131700b01 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250126 +20250127 commit 88e26baf12ba7ff457b0ab56d6790e4727973b23 Author: GCC Administrator Date: Tue Jan 28 00:24:05 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 06131700b01..6f429ff09a0 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250127 +20250128 commit d2183f89d8e78082e6d869f7454b372ba52671a5 Author: GCC Administrator Date: Wed Jan 29 00:21:37 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 6f429ff09a0..ee73a5fce07 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250128 +20250129 commit 5d0dcc6acd3294cd3adc4fab8ee9b4257b22180c Author: GCC Administrator Date: Thu Jan 30 00:21:03 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index ee73a5fce07..850652b6bce 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250129 +20250130 commit 19aaf8c81912b8decc8767e87ed791f23c759210 Author: GCC Administrator Date: Fri Jan 31 00:21:40 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 850652b6bce..d42c41347a3 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250130 +20250131 commit 823a3b7a6dbb30dfd69fa9f2f56a3f35b71791da Author: GCC Administrator Date: Sat Feb 1 00:20:04 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d42c41347a3..50ae9039749 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250131 +20250201 commit 71c41f4fa2d0945d650f90e3f43c24b0e9d05751 Author: GCC Administrator Date: Sun Feb 2 00:19:46 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 50ae9039749..ad51826fd80 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250201 +20250202 commit c15745b40205a9625bf79470eb33ff04cd3312f1 Author: GCC Administrator Date: Mon Feb 3 00:19:35 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index ad51826fd80..7f9b55c65c3 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250202 +20250203 commit 11d14c3affedff8b0dd963a902c0d01f50ed1380 Author: GCC Administrator Date: Tue Feb 4 00:19:57 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7f9b55c65c3..9be659a9cfd 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250203 +20250204 commit 251f6ba9131ebc8deab463c052e099a065796c2a Author: Lewis Hyatt Date: Sun Jan 26 18:57:00 2025 -0500 options: Adjust cl_optimization_compare to avoid checking ICE [PR115913] At the end of a sequence like: #pragma GCC push_options ... #pragma GCC pop_options the handler for pop_options calls cl_optimization_compare() (as generated by optc-save-gen.awk) to make sure that all global state has been restored to the value it had prior to the push_options call. The verification is performed for almost all entries in the global_options struct. This leads to unexpected checking asserts, as discussed in the PR, in case the state of warnings-related options has been intentionally modified in between push_options and pop_options via a call to #pragma GCC diagnostic. Address that by skipping the verification for CL_WARNING-flagged options. gcc/ChangeLog: PR middle-end/115913 * optc-save-gen.awk (cl_optimization_compare): Skip options with CL_WARNING flag. gcc/testsuite/ChangeLog: PR middle-end/115913 * c-c++-common/cpp/pr115913.c: New test. diff --git a/gcc/optc-save-gen.awk b/gcc/optc-save-gen.awk index 76e9b3cb940..dd09fd38c87 100644 --- a/gcc/optc-save-gen.awk +++ b/gcc/optc-save-gen.awk @@ -1451,6 +1451,11 @@ for (i = 0; i < n_opts; i++) { if (name == "") continue; + # We do not want to compare warning-related options, since they + # might have been modified by a #pragma GCC diagnostic. + if (flag_set_p("Warning", flags[i])) + continue; + if (name in checked_options) continue; checked_options[name]++ diff --git a/gcc/testsuite/c-c++-common/cpp/pr115913.c b/gcc/testsuite/c-c++-common/cpp/pr115913.c new file mode 100644 index 00000000000..b9d10cda8d2 --- /dev/null +++ b/gcc/testsuite/c-c++-common/cpp/pr115913.c @@ -0,0 +1,7 @@ +/* { dg-do preprocess } */ +/* PR middle-end/115913 */ +#pragma GCC push_options +#pragma GCC diagnostic warning "-Wundef" +/* The call to cl_optimization_compare performed by pop_options should not + lead to a checking failure. */ +#pragma GCC pop_options commit 351801905e31eeccc89ef529652b0070ae413dff Author: GCC Administrator Date: Wed Feb 5 00:20:24 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index c9f267fe0fa..0ad9d1f656d 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,9 @@ +2025-02-04 Lewis Hyatt + + PR middle-end/115913 + * optc-save-gen.awk (cl_optimization_compare): Skip options with + CL_WARNING flag. + 2025-01-24 Peter Bergner Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 9be659a9cfd..e9825778341 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250204 +20250205 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 33e17302507..a4b2830121d 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,8 @@ +2025-02-04 Lewis Hyatt + + PR middle-end/115913 + * c-c++-common/cpp/pr115913.c: New test. + 2025-01-24 Peter Bergner Backported from master: commit 14de2b63c9fc3997d8a430566dfb4d970542b03c Author: GCC Administrator Date: Thu Feb 6 00:19:58 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e9825778341..8817645f579 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250205 +20250206 commit a2d7edba5dc51c2c7c095b1f461ee12e145fab9d Author: GCC Administrator Date: Fri Feb 7 00:20:42 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 8817645f579..293639e3137 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250206 +20250207 commit 312576be254929dc184b6f25425e7a8664edb3c0 Author: GCC Administrator Date: Sat Feb 8 00:19:15 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 293639e3137..f9d9d8c45b9 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250207 +20250208 commit 5bcf1819c16f36c058c551b9fcef2b6ef16e45c2 Author: GCC Administrator Date: Sun Feb 9 00:18:50 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index f9d9d8c45b9..11e801e0ed9 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250208 +20250209 commit e15b19691bb1bcf2a06d57cfa2c4cb3b21c1bb2d Author: GCC Administrator Date: Mon Feb 10 00:19:20 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 11e801e0ed9..039de05fe6a 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250209 +20250210 commit 53b84cff8c602ddc59014a002e1b25db7c7cdb60 Author: GCC Administrator Date: Tue Feb 11 00:19:35 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 039de05fe6a..68b0ff9265d 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250210 +20250211 commit bd52571d713749f1a4cf0f58ca4922dbc42b5752 Author: H.J. Lu Date: Tue Feb 11 13:47:54 2025 +0800 x86: Correct ASM_OUTPUT_SYMBOL_REF x is not a macro argument. It just happens to work as final.cc passes x for 2nd argument: final.cc: ASM_OUTPUT_SYMBOL_REF (file, x); PR target/118825 * config/i386/i386.h (ASM_OUTPUT_SYMBOL_REF): Replace x with SYM. Signed-off-by: H.J. Lu (cherry picked from commit 7317fc0b03380a83ad03a5fc4fabef5f38c44c9d) diff --git a/gcc/config/i386/i386.h b/gcc/config/i386/i386.h index ed988ca280e..8416e5b02b6 100644 --- a/gcc/config/i386/i386.h +++ b/gcc/config/i386/i386.h @@ -2174,7 +2174,7 @@ extern int const svr4_dbx_register_map[FIRST_PSEUDO_REGISTER]; #define ASM_OUTPUT_SYMBOL_REF(FILE, SYM) \ do { \ const char *name \ - = assemble_name_resolve (XSTR (x, 0)); \ + = assemble_name_resolve (XSTR (SYM, 0)); \ /* In -masm=att wrap identifiers that start with $ \ into parens. */ \ if (ASSEMBLER_DIALECT == ASM_ATT \ commit 775b03b6ad63be0ac68ebcfab9ad3ed30d582da2 Author: GCC Administrator Date: Wed Feb 12 00:19:33 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 0ad9d1f656d..15c83c310b4 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2025-02-11 H.J. Lu + + Backported from master: + 2025-02-11 H.J. Lu + + PR target/118825 + * config/i386/i386.h (ASM_OUTPUT_SYMBOL_REF): Replace x with + SYM. + 2025-02-04 Lewis Hyatt PR middle-end/115913 diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 68b0ff9265d..4bcf0fa7545 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250211 +20250212 commit c72f9c0a3ad8eefd0706957ba054c2c2f388d3d5 Author: GCC Administrator Date: Thu Feb 13 00:19:41 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 4bcf0fa7545..d38af5a6357 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250212 +20250213 commit be0d9f7d26519ff9d966daeae9e7f34e06758453 Author: GCC Administrator Date: Fri Feb 14 00:19:31 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d38af5a6357..c2b6f8fd1b4 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250213 +20250214 commit 0e6f7bc43988286800ff6ca34cd48ba387397ee7 Author: GCC Administrator Date: Sat Feb 15 00:18:50 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c2b6f8fd1b4..92ec7cafd89 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250214 +20250215 commit f4eb061bfd0cdf4012827d36f509041c887c7dd3 Author: GCC Administrator Date: Sun Feb 16 00:19:25 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 92ec7cafd89..b0189f2c256 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250215 +20250216 commit 20bea930cc0970b90778468ff082f55697c05c02 Author: GCC Administrator Date: Mon Feb 17 00:18:55 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index b0189f2c256..328350d1a96 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250216 +20250217 commit d468550621e0116cf2889e619cd2a2e4572b0b9e Author: GCC Administrator Date: Tue Feb 18 00:19:17 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 328350d1a96..bec432ba466 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250217 +20250218 commit ed78c0937b7d6b4307dae38e38063a0a8fb34a43 Author: GCC Administrator Date: Wed Feb 19 00:19:10 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index bec432ba466..7462e91a3a7 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250218 +20250219 commit 9ffaf0b5094d5e011b42efa70debdac1b9fea959 Author: GCC Administrator Date: Thu Feb 20 00:19:13 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7462e91a3a7..efb19b9646b 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250219 +20250220 commit 23541b23deb5504c6d3c0a3e96a0858e10c3c627 Author: Dimitry Andric Date: Tue Jan 28 18:36:16 2025 +0100 libgcc: On FreeBSD use GCC's crt objects for static linking Add crtbeginT.o to extra_parts on FreeBSD. This ensures we use GCC's crt objects for static linking. Otherwise it could mix crtbeginT.o from the base system with libgcc's crtend.o, possibly leading to segfaults. libgcc: PR target/118685 * config.host (*-*-freebsd*): Add crtbeginT.o to extra_parts. Signed-off-by: Dimitry Andric diff --git a/libgcc/config.host b/libgcc/config.host index 89e3dbc7c8a..8aa22123add 100644 --- a/libgcc/config.host +++ b/libgcc/config.host @@ -271,7 +271,7 @@ case ${host} in # machine-specific sections may refine and add to this # configuration. tmake_file="$tmake_file t-freebsd t-crtstuff-pic t-libgcc-pic t-eh-dw2-dip t-slibgcc t-slibgcc-gld t-slibgcc-elf-ver" - extra_parts="crtbegin.o crtend.o crtbeginS.o crtendS.o" + extra_parts="crtbegin.o crtend.o crtbeginS.o crtbeginT.o crtendS.o" case ${target_thread_file} in posix) tmake_file="${tmake_file} t-freebsd-thread" commit 6fd1f3eb6712aae538526e649954238e62694f8f Author: GCC Administrator Date: Fri Feb 21 00:19:33 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index efb19b9646b..1591c466619 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250220 +20250221 diff --git a/libgcc/ChangeLog b/libgcc/ChangeLog index 28fb20bc5cc..e25ba6a86c0 100644 --- a/libgcc/ChangeLog +++ b/libgcc/ChangeLog @@ -1,3 +1,8 @@ +2025-02-20 Dimitry Andric + + PR target/118685 + * config.host (*-*-freebsd*): Add crtbeginT.o to extra_parts. + 2024-06-20 Release Manager * GCC 12.4.0 released. commit e2a1588a1c3a79e8c9249646f4e98d8c4e766c0a Author: GCC Administrator Date: Sat Feb 22 00:19:13 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 1591c466619..e1fa9784f9e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250221 +20250222 commit 7466f745b92d8187e160cbd1a2081a005d5cd156 Author: GCC Administrator Date: Sun Feb 23 00:19:56 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e1fa9784f9e..e4d939c1430 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250222 +20250223 commit 647dbc909bb43f46e8725d666d5516e62aeb7cc5 Author: GCC Administrator Date: Mon Feb 24 00:20:39 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e4d939c1430..927bc531bc8 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250223 +20250224 commit 07d24a1feb654035212b13c165e2f5cf790c8d6c Author: GCC Administrator Date: Tue Feb 25 00:20:11 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 927bc531bc8..8781321aefe 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250224 +20250225 commit 6517238fadf6f24a7184290759f412d960f5f4b9 Author: GCC Administrator Date: Wed Feb 26 00:20:22 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 8781321aefe..5faab8b9a3f 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250225 +20250226 commit b27fa6a7ca86a9b885cb4dbe8a55991e7fb666f0 Author: Stefan Schulze Frielinghaus Date: Thu Feb 13 09:13:06 2025 +0100 s390: Fix s390_valid_shift_count() for TI mode [PR118835] During combine we may end up with (set (reg:DI 66 [ _6 ]) (ashift:DI (reg:DI 72 [ x ]) (subreg:QI (and:TI (reg:TI 67 [ _1 ]) (const_wide_int 0x0aaaaaaaaaaaaaabf)) 15))) where the shift count operand does not trivially fit the scheme of address operands. Reject those operands, especially since strip_address_mutations() expects expressions of the form (and ... (const_int ...)) and fails for (and ... (const_wide_int ...)). Thus, be more strict here and accept only CONST_INT operands. Done by replacing immediate_operand() with const_int_operand() which is enough since the former only additionally checks for LEGITIMATE_PIC_OPERAND_P and targetm.legitimate_constant_p which are always true for CONST_INT operands. While on it, fix indentation of the if block. gcc/ChangeLog: PR target/118835 * config/s390/s390.cc (s390_valid_shift_count): Reject shift count operands which do not trivially fit the scheme of address operands. gcc/testsuite/ChangeLog: * gcc.target/s390/pr118835.c: New test. (cherry picked from commit ac9806dae30d07ab082ac341fe5646987753adcb) diff --git a/gcc/config/s390/s390.cc b/gcc/config/s390/s390.cc index a8f804ffe4f..4e7a01ae6c9 100644 --- a/gcc/config/s390/s390.cc +++ b/gcc/config/s390/s390.cc @@ -3216,26 +3216,31 @@ s390_valid_shift_count (rtx op, HOST_WIDE_INT implicit_mask) /* Check for an and with proper constant. */ if (GET_CODE (op) == AND) - { - rtx op1 = XEXP (op, 0); - rtx imm = XEXP (op, 1); + { + rtx op1 = XEXP (op, 0); + rtx imm = XEXP (op, 1); - if (GET_CODE (op1) == SUBREG && subreg_lowpart_p (op1)) - op1 = XEXP (op1, 0); + if (GET_CODE (op1) == SUBREG && subreg_lowpart_p (op1)) + op1 = XEXP (op1, 0); - if (!(register_operand (op1, GET_MODE (op1)) || GET_CODE (op1) == PLUS)) - return false; + if (!(register_operand (op1, GET_MODE (op1)) || GET_CODE (op1) == PLUS)) + return false; - if (!immediate_operand (imm, GET_MODE (imm))) - return false; + /* Accept only CONST_INT as immediates, i.e., reject shift count operands + which do not trivially fit the scheme of address operands. Especially + since strip_address_mutations() expects expressions of the form + (and ... (const_int ...)) and fails for + (and ... (const_wide_int ...)). */ + if (!const_int_operand (imm, GET_MODE (imm))) + return false; - HOST_WIDE_INT val = INTVAL (imm); - if (implicit_mask > 0 - && (val & implicit_mask) != implicit_mask) - return false; + HOST_WIDE_INT val = INTVAL (imm); + if (implicit_mask > 0 + && (val & implicit_mask) != implicit_mask) + return false; - op = op1; - } + op = op1; + } /* Check the rest. */ return s390_decompose_addrstyle_without_index (op, NULL, NULL); diff --git a/gcc/testsuite/gcc.target/s390/pr118835.c b/gcc/testsuite/gcc.target/s390/pr118835.c new file mode 100644 index 00000000000..1ca6cd95543 --- /dev/null +++ b/gcc/testsuite/gcc.target/s390/pr118835.c @@ -0,0 +1,21 @@ +/* { dg-do compile { target int128 } } */ +/* { dg-options "-O2" } */ + +/* During combine we may end up with patterns of the form + + (set (reg:DI 66 [ _6 ]) + (ashift:DI (reg:DI 72 [ x ]) + (subreg:QI (and:TI (reg:TI 67 [ _1 ]) + (const_wide_int 0x0aaaaaaaaaaaaaabf)) + 15))) + + which should be rejected since the shift count does not trivially fit the + scheme of address operands. */ + +long +test (long x, int y) +{ + __int128 z = 0xAAAAAAAAAAAAAABF; + z &= y; + return x << z; +} commit 9211a99d95fda2c0a5a11754449726ee9db77e6a Author: GCC Administrator Date: Thu Feb 27 00:20:42 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 15c83c310b4..f0e8abd526b 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,13 @@ +2025-02-26 Stefan Schulze Frielinghaus + + Backported from master: + 2025-02-13 Stefan Schulze Frielinghaus + + PR target/118835 + * config/s390/s390.cc (s390_valid_shift_count): Reject shift + count operands which do not trivially fit the scheme of + address operands. + 2025-02-11 H.J. Lu Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 5faab8b9a3f..78e945cf0f0 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250226 +20250227 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index a4b2830121d..8b2767c0f3a 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,10 @@ +2025-02-26 Stefan Schulze Frielinghaus + + Backported from master: + 2025-02-13 Stefan Schulze Frielinghaus + + * gcc.target/s390/pr118835.c: New test. + 2025-02-04 Lewis Hyatt PR middle-end/115913 commit 90a791bb651d38d35f639d34880792a1d4aea3cb Author: GCC Administrator Date: Fri Feb 28 00:20:33 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 78e945cf0f0..e392026598c 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250227 +20250228 commit b7fd09c66fafab066901f746c835dcc1ff78e5ee Author: GCC Administrator Date: Sat Mar 1 13:06:08 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e392026598c..a607fb00164 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250228 +20250301 commit 6c95603033812f413d1f404baae594fa0fdf4f5d Author: GCC Administrator Date: Sun Mar 2 00:20:29 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index a607fb00164..08a14c75034 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250301 +20250302 commit f6be26657c8b974a5caf39350e4965cfb2c4b936 Author: Iain Buclaw Date: Fri Feb 28 19:22:36 2025 +0100 d: Fix comparing uninitialized memory in dstruct.d [PR116961] Floating-point emulation in the D front-end is done via a type named `struct longdouble`, which in GDC is a small interface around the real_value type. Because the D code cannot include gcc/real.h directly, a big enough buffer is used for the data instead. On x86_64, this buffer is actually bigger than real_value itself, so when a new longdouble object is created with longdouble r; real_from_string3 (&r.rv (), buffer, mode); return r; there is uninitialized padding at the end of `r`. This was never a problem when D was implemented in C++ (until GCC 12) as comparing two longdouble objects with `==' would be forwarded to the relevant operator== overload that extracted the underlying real_value. However when the front-end was translated to D, such conditions were instead rewritten into identity comparisons return exp.toReal() is CTFloat.zero The `is` operator gets lowered as a call to `memcmp() == 0', which is where the read of uninitialized memory occurs, as seen by valgrind. ==26778== Conditional jump or move depends on uninitialised value(s) ==26778== at 0x911F41: dmd.dstruct._isZeroInit(dmd.expression.Expression) (dstruct.d:635) ==26778== by 0x9123BE: StructDeclaration::finalizeSize() (dstruct.d:373) ==26778== by 0x86747C: dmd.aggregate.AggregateDeclaration.determineSize(ref const(dmd.location.Loc)) (aggregate.d:226) [...] To avoid accidentally reading uninitialized data, explicitly initialize all `longdouble` variables with an empty constructor on C++ side of the implementation before initializing underlying real_value type it holds. PR d/116961 gcc/d/ChangeLog: * d-codegen.cc (build_float_cst): Change new_value type from real_t to real_value. * d-ctfloat.cc (CTFloat::fabs): Default initialize the return value. (CTFloat::ldexp): Likewise. (CTFloat::parse): Likewise. * d-longdouble.cc (longdouble::add): Likewise. (longdouble::sub): Likewise. (longdouble::mul): Likewise. (longdouble::div): Likewise. (longdouble::mod): Likewise. (longdouble::neg): Likewise. * d-port.cc (Port::isFloat32LiteralOutOfRange): Likewise. (Port::isFloat64LiteralOutOfRange): Likewise. gcc/testsuite/ChangeLog: * gdc.dg/pr116961.d: New test. (cherry picked from commit f7bc17ebc9ef89700672ed7125da719f3558f3b7) diff --git a/gcc/d/d-codegen.cc b/gcc/d/d-codegen.cc index 31d593ca7fd..4f20d063c90 100644 --- a/gcc/d/d-codegen.cc +++ b/gcc/d/d-codegen.cc @@ -244,15 +244,15 @@ build_integer_cst (dinteger_t value, tree type) tree build_float_cst (const real_t &value, Type *totype) { - real_t new_value; + real_value new_value; TypeBasic *tb = totype->isTypeBasic (); gcc_assert (tb != NULL); tree type_node = build_ctype (tb); - real_convert (&new_value.rv (), TYPE_MODE (type_node), &value.rv ()); + real_convert (&new_value, TYPE_MODE (type_node), &value.rv ()); - return build_real (type_node, new_value.rv ()); + return build_real (type_node, new_value); } /* Returns the .length component from the D dynamic array EXP. */ diff --git a/gcc/d/d-ctfloat.cc b/gcc/d/d-ctfloat.cc index c4d9a44c59b..3afddc8fe1a 100644 --- a/gcc/d/d-ctfloat.cc +++ b/gcc/d/d-ctfloat.cc @@ -33,7 +33,7 @@ along with GCC; see the file COPYING3. If not see real_t CTFloat::fabs (real_t r) { - real_t x; + real_t x = {}; real_arithmetic (&x.rv (), ABS_EXPR, &r.rv (), NULL); return x.normalize (); } @@ -43,7 +43,7 @@ CTFloat::fabs (real_t r) real_t CTFloat::ldexp (real_t r, int exp) { - real_t x; + real_t x = {}; real_ldexp (&x.rv (), &r.rv (), exp); return x.normalize (); } @@ -87,7 +87,7 @@ CTFloat::isInfinity (real_t r) real_t CTFloat::parse (const char *buffer, bool *overflow) { - real_t r; + real_t r = {}; real_from_string3 (&r.rv (), buffer, TYPE_MODE (long_double_type_node)); /* Front-end checks overflow to see if the value is representable. */ diff --git a/gcc/d/d-longdouble.cc b/gcc/d/d-longdouble.cc index cf0b68b2f2d..2fb48e2722f 100644 --- a/gcc/d/d-longdouble.cc +++ b/gcc/d/d-longdouble.cc @@ -113,7 +113,7 @@ longdouble::to_bool (void) const longdouble longdouble::add (const longdouble &r) const { - longdouble x; + longdouble x = {}; real_arithmetic (&x.rv (), PLUS_EXPR, &this->rv (), &r.rv ()); return x.normalize (); } @@ -121,7 +121,7 @@ longdouble::add (const longdouble &r) const longdouble longdouble::sub (const longdouble &r) const { - longdouble x; + longdouble x = {}; real_arithmetic (&x.rv (), MINUS_EXPR, &this->rv (), &r.rv ()); return x.normalize (); } @@ -129,7 +129,7 @@ longdouble::sub (const longdouble &r) const longdouble longdouble::mul (const longdouble &r) const { - longdouble x; + longdouble x = {}; real_arithmetic (&x.rv (), MULT_EXPR, &this->rv (), &r.rv ()); return x.normalize (); } @@ -137,7 +137,7 @@ longdouble::mul (const longdouble &r) const longdouble longdouble::div (const longdouble &r) const { - longdouble x; + longdouble x = {}; real_arithmetic (&x.rv (), RDIV_EXPR, &this->rv (), &r.rv ()); return x.normalize (); } @@ -145,7 +145,7 @@ longdouble::div (const longdouble &r) const longdouble longdouble::mod (const longdouble &r) const { - longdouble x; + longdouble x = {}; real_value q; if (r.rv ().cl == rvc_zero || REAL_VALUE_ISINF (this->rv ())) @@ -172,7 +172,7 @@ longdouble::mod (const longdouble &r) const longdouble longdouble::neg (void) const { - longdouble x; + longdouble x = {}; real_arithmetic (&x.rv (), NEGATE_EXPR, &this->rv (), NULL); return x.normalize (); } diff --git a/gcc/d/d-port.cc b/gcc/d/d-port.cc index a908cc8dbb0..9eff992090f 100644 --- a/gcc/d/d-port.cc +++ b/gcc/d/d-port.cc @@ -74,7 +74,7 @@ Port::strupr (char *s) bool Port::isFloat32LiteralOutOfRange (const char *buffer) { - real_t r; + real_t r = {}; real_from_string3 (&r.rv (), buffer, TYPE_MODE (float_type_node)); @@ -87,7 +87,7 @@ Port::isFloat32LiteralOutOfRange (const char *buffer) bool Port::isFloat64LiteralOutOfRange (const char *buffer) { - real_t r; + real_t r = {}; real_from_string3 (&r.rv (), buffer, TYPE_MODE (double_type_node)); diff --git a/gcc/testsuite/gdc.dg/pr116961.d b/gcc/testsuite/gdc.dg/pr116961.d new file mode 100644 index 00000000000..fd51308636c --- /dev/null +++ b/gcc/testsuite/gdc.dg/pr116961.d @@ -0,0 +1,7 @@ +// { dg-do compile } +struct S116961 +{ + float thing = 0.0; +} + +static assert(__traits(isZeroInit, S116961) == true); commit cc2653123bc2e251013c30386d382eb5744a7d83 Author: GCC Administrator Date: Mon Mar 3 00:20:33 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 08a14c75034..b12838a3943 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250302 +20250303 diff --git a/gcc/d/ChangeLog b/gcc/d/ChangeLog index 29b1af50a6d..3ac009b8491 100644 --- a/gcc/d/ChangeLog +++ b/gcc/d/ChangeLog @@ -1,3 +1,23 @@ +2025-03-02 Iain Buclaw + + Backported from master: + 2025-02-28 Iain Buclaw + + PR d/116961 + * d-codegen.cc (build_float_cst): Change new_value type from real_t to + real_value. + * d-ctfloat.cc (CTFloat::fabs): Default initialize the return value. + (CTFloat::ldexp): Likewise. + (CTFloat::parse): Likewise. + * d-longdouble.cc (longdouble::add): Likewise. + (longdouble::sub): Likewise. + (longdouble::mul): Likewise. + (longdouble::div): Likewise. + (longdouble::mod): Likewise. + (longdouble::neg): Likewise. + * d-port.cc (Port::isFloat32LiteralOutOfRange): Likewise. + (Port::isFloat64LiteralOutOfRange): Likewise. + 2025-01-21 Iain Buclaw Backported from master: diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 8b2767c0f3a..7a9271cad3e 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2025-03-02 Iain Buclaw + + Backported from master: + 2025-02-28 Iain Buclaw + + PR d/116961 + * gdc.dg/pr116961.d: New test. + 2025-02-26 Stefan Schulze Frielinghaus Backported from master: commit 7e92949ad3c874675cc3eb1e1fdb006e17f54266 Author: GCC Administrator Date: Tue Mar 4 00:20:32 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index b12838a3943..7b95daf8170 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250303 +20250304 commit 283e554e5cb9e3d3cf3853d1d180c201cd77f494 Author: GCC Administrator Date: Wed Mar 5 00:22:21 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7b95daf8170..0cf6d8a9f9a 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250304 +20250305 commit 77172598dcb07b6ea71f4549917f2647eb34f38a Author: Uros Bizjak Date: Wed Feb 12 11:19:57 2025 +0100 combine: Discard REG_UNUSED note in i2 when register is also referenced in i3 [PR118739] The combine pass is trying to combine: Trying 16, 22, 21 -> 23: 16: r104:QI=flags:CCNO>0 22: {r120:QI=r104:QI^0x1;clobber flags:CC;} REG_UNUSED flags:CC 21: r119:QI=flags:CCNO<=0 REG_DEAD flags:CCNO 23: {r110:QI=r119:QI|r120:QI;clobber flags:CC;} REG_DEAD r120:QI REG_DEAD r119:QI REG_UNUSED flags:CC and creates the following two insn sequence: modifying insn i2 22: r104:QI=flags:CCNO>0 REG_DEAD flags:CC deferring rescan insn with uid = 22. modifying insn i3 23: r110:QI=flags:CCNO<=0 REG_DEAD flags:CC deferring rescan insn with uid = 23. where the REG_DEAD note in i2 is not correct, because the flags register is still referenced in i3. In try_combine() megafunction, we have this part: --cut here-- /* Distribute all the LOG_LINKS and REG_NOTES from I1, I2, and I3. */ if (i3notes) distribute_notes (i3notes, i3, i3, newi2pat ? i2 : NULL, elim_i2, elim_i1, elim_i0); if (i2notes) distribute_notes (i2notes, i2, i3, newi2pat ? i2 : NULL, elim_i2, elim_i1, elim_i0); if (i1notes) distribute_notes (i1notes, i1, i3, newi2pat ? i2 : NULL, elim_i2, local_elim_i1, local_elim_i0); if (i0notes) distribute_notes (i0notes, i0, i3, newi2pat ? i2 : NULL, elim_i2, elim_i1, local_elim_i0); if (midnotes) distribute_notes (midnotes, NULL, i3, newi2pat ? i2 : NULL, elim_i2, elim_i1, elim_i0); --cut here-- where the compiler distributes REG_UNUSED note from i2: 22: {r120:QI=r104:QI^0x1;clobber flags:CC;} REG_UNUSED flags:CC via distribute_notes() using the following: --cut here-- /* Otherwise, if this register is used by I3, then this register now dies here, so we must put a REG_DEAD note here unless there is one already. */ else if (reg_referenced_p (XEXP (note, 0), PATTERN (i3)) && ! (REG_P (XEXP (note, 0)) ? find_regno_note (i3, REG_DEAD, REGNO (XEXP (note, 0))) : find_reg_note (i3, REG_DEAD, XEXP (note, 0)))) { PUT_REG_NOTE_KIND (note, REG_DEAD); place = i3; } --cut here-- Flags register is used in I3, but there already is a REG_DEAD note in I3. The above condition doesn't trigger and continues in the "else" part where REG_DEAD note is put to I2. The proposed solution corrects the above logic to trigger every time the register is referenced in I3, avoiding the "else" part. PR rtl-optimization/118739 gcc/ChangeLog: * combine.cc (distribute_notes) : Correct the logic when the register is used by I3. gcc/testsuite/ChangeLog: * gcc.target/i386/pr118739.c: New test. (cherry picked from commit a92dc3fe31c95d56019b2fb95a58414bca06241f) diff --git a/gcc/combine.cc b/gcc/combine.cc index e79500d40c9..718da753b87 100644 --- a/gcc/combine.cc +++ b/gcc/combine.cc @@ -14324,14 +14324,15 @@ distribute_notes (rtx notes, rtx_insn *from_insn, rtx_insn *i3, rtx_insn *i2, /* Otherwise, if this register is used by I3, then this register now dies here, so we must put a REG_DEAD note here unless there is one already. */ - else if (reg_referenced_p (XEXP (note, 0), PATTERN (i3)) - && ! (REG_P (XEXP (note, 0)) - ? find_regno_note (i3, REG_DEAD, - REGNO (XEXP (note, 0))) - : find_reg_note (i3, REG_DEAD, XEXP (note, 0)))) + else if (reg_referenced_p (XEXP (note, 0), PATTERN (i3))) { - PUT_REG_NOTE_KIND (note, REG_DEAD); - place = i3; + if (! (REG_P (XEXP (note, 0)) + ? find_regno_note (i3, REG_DEAD, REGNO (XEXP (note, 0))) + : find_reg_note (i3, REG_DEAD, XEXP (note, 0)))) + { + PUT_REG_NOTE_KIND (note, REG_DEAD); + place = i3; + } } /* A SET or CLOBBER of the REG_UNUSED reg has been removed, diff --git a/gcc/testsuite/gcc.target/i386/pr118739.c b/gcc/testsuite/gcc.target/i386/pr118739.c new file mode 100644 index 00000000000..89bed546363 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr118739.c @@ -0,0 +1,50 @@ +/* PR rtl-optimization/118739 */ +/* { dg-do run } */ +/* { dg-options "-O3 -fno-tree-forwprop -fno-tree-vrp" } */ + +volatile int a; +int b, c, d = 1, e, f, g; + +int h (void) +{ + int i = 1; + + j: + for (b = 1; b; b--) + { + asm ("#"); + + g = 0; + + for (; g <= 1; g++) + { + int k = f = 0; + + for (; f <= 1; f++) + k = (1 == i) >= k || ((d = 0) >= a) + k; + } + } + + for (; i < 3; i++) + { + if (!c) + return g; + + if (e) + goto j; + + asm ("#"); + } + + return 0; +} + +int main() +{ + h(); + + if (d != 1) + __builtin_abort(); + + return 0; +} commit 9496f5088111d9330bba2659b024f7e7a2175b4b Author: Jakub Jelinek Date: Tue Mar 4 09:52:22 2025 +0100 testsuite: Add tests for already fixed PR [PR119071] Uros' r15-7793 fixed this PR as well, I'm just committing tests from the PR so that it can be closed. 2025-03-04 Jakub Jelinek PR rtl-optimization/119071 * gcc.dg/pr119071.c: New test. * gcc.c-torture/execute/pr119071.c: New test. (cherry picked from commit ccf9db9a6fa4b5bc7aad5e9603e2ac71984142a0) diff --git a/gcc/testsuite/gcc.c-torture/execute/pr119071.c b/gcc/testsuite/gcc.c-torture/execute/pr119071.c new file mode 100644 index 00000000000..91f29cce3d5 --- /dev/null +++ b/gcc/testsuite/gcc.c-torture/execute/pr119071.c @@ -0,0 +1,15 @@ +/* PR rtl-optimization/119071 */ + +int a, b; + +int +main () +{ + int c = 0; + if (a + 2) + c = 1; + int d = (1 + c - 2 + c == 1) - 1; + b = ((d + 1) << d) + d; + if (b != 1) + __builtin_abort (); +} diff --git a/gcc/testsuite/gcc.dg/pr119071.c b/gcc/testsuite/gcc.dg/pr119071.c new file mode 100644 index 00000000000..ade1d288d2a --- /dev/null +++ b/gcc/testsuite/gcc.dg/pr119071.c @@ -0,0 +1,45 @@ +/* PR rtl-optimization/119071 */ +/* { dg-do run } */ +/* { dg-options "-O2 -fgimple" } */ + +int a, b; + +int __GIMPLE (ssa,startwith("expand")) +foo (void) +{ + int _1; + int _2; + int _3; + int _5; + _Bool _7; + int _8; + int _9; + _Bool _14; + int _15; + int _16; + _Bool _17; + int _18; + + __BB(2): + _1 = a; + _17 = _1 != _Literal (int) -2; + _18 = (int) _17; + _2 = _18 + _Literal (int) -1; + _3 = _2 + _18; + _14 = _3 != 1; + _15 = (int) _14; + _16 = -_15; + _7 = _3 == 1; + _9 = (int) _7; + _5 = _9 << _16; + _8 = _5 - _15; + b = _8; + return _8; +} + +int +main () +{ + if (foo () != 1) + __builtin_abort (); +} commit 6eec3f754319eeda377f1773e684ebc8792dc27a Author: GCC Administrator Date: Thu Mar 6 00:20:56 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index f0e8abd526b..814e23d514b 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2025-03-05 Uros Bizjak + + Backported from master: + 2025-03-03 Uros Bizjak + + PR rtl-optimization/118739 + * combine.cc (distribute_notes) : Correct the + logic when the register is used by I3. + 2025-02-26 Stefan Schulze Frielinghaus Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 0cf6d8a9f9a..31df6d356f3 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250305 +20250306 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 7a9271cad3e..5bc8a0143cd 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,20 @@ +2025-03-05 Jakub Jelinek + + Backported from master: + 2025-03-04 Jakub Jelinek + + PR rtl-optimization/119071 + * gcc.dg/pr119071.c: New test. + * gcc.c-torture/execute/pr119071.c: New test. + +2025-03-05 Uros Bizjak + + Backported from master: + 2025-03-03 Uros Bizjak + + PR rtl-optimization/118739 + * gcc.target/i386/pr118739.c: New test. + 2025-03-02 Iain Buclaw Backported from master: commit c96a7aa466bfda1f48a5b572bfb03708209bdf30 Author: GCC Administrator Date: Fri Mar 7 00:19:12 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 31df6d356f3..5f4d7bc93d2 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250306 +20250307 commit 149b38a44d041d3b4142f50a9b2d6e1190df3b11 Author: Christophe Lyon Date: Mon Mar 3 11:12:18 2025 +0000 arm: Handle fixed PIC register in require_pic_register (PR target/115485) Commit r9-4307-g89d7557202d25a forgot to accept a fixed PIC register when extending the assert in require_pic_register. arm_pic_register can be set explicitly by the user (e.g. -mpic-register=r9) or implicitly as the default value with -fpic/-fPIC/-fPIE and -mno-pic-data-is-text-relative -mlong-calls, and we want to use/accept it when recording cfun->machine->pic_reg as used to be the case. PR target/115485 gcc/ * config/arm/arm.cc (require_pic_register): Fix typos in comment. Handle fixed arm_pic_register. gcc/testsuite/ * g++.target/arm/pr115485.C: New test. (cherry picked from commit b1d0ac28de643e7c810e407a0668737131cdcc00) diff --git a/gcc/config/arm/arm.cc b/gcc/config/arm/arm.cc index afaa599b6f0..f6f0e88ed15 100644 --- a/gcc/config/arm/arm.cc +++ b/gcc/config/arm/arm.cc @@ -7954,8 +7954,8 @@ legitimate_pic_operand_p (rtx x) /* Record that the current function needs a PIC register. If PIC_REG is null, a new pseudo is allocated as PIC register, otherwise PIC_REG is used. In - both case cfun->machine->pic_reg is initialized if we have not already done - so. COMPUTE_NOW decide whether and where to set the PIC register. If true, + both cases cfun->machine->pic_reg is initialized if we have not already done + so. COMPUTE_NOW decides whether and where to set the PIC register. If true, PIC register is reloaded in the current position of the instruction stream irregardless of whether it was loaded before. Otherwise, it is only loaded if not already done so (crtl->uses_pic_offset_table is null). Note that @@ -7975,6 +7975,7 @@ require_pic_register (rtx pic_reg, bool compute_now) if (!crtl->uses_pic_offset_table || compute_now) { gcc_assert (can_create_pseudo_p () + || (arm_pic_register != INVALID_REGNUM) || (pic_reg != NULL_RTX && REG_P (pic_reg) && GET_MODE (pic_reg) == Pmode)); diff --git a/gcc/testsuite/g++.target/arm/pr115485.C b/gcc/testsuite/g++.target/arm/pr115485.C new file mode 100644 index 00000000000..491b48c726a --- /dev/null +++ b/gcc/testsuite/g++.target/arm/pr115485.C @@ -0,0 +1,16 @@ +/* { dg-do compile } */ +/* { dg-options "-fPIE -mno-pic-data-is-text-relative -mlong-calls -ffunction-sections" } */ + +struct c1 { + virtual void func1() = 0; +}; +struct c2 { + virtual ~c2() {} +}; +struct c3 : c2, c1 { + void func1() override; + void func3(); +}; +void c3::func1() { + func3(); +} commit c5588cbfa340d2cf79fd6f9370a17a86c54cb688 Author: GCC Administrator Date: Sat Mar 8 00:19:50 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 814e23d514b..0d9991af398 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,12 @@ +2025-03-07 Christophe Lyon + + Backported from master: + 2025-03-07 Christophe Lyon + + PR target/115485 + * config/arm/arm.cc (require_pic_register): Fix typos in + comment. Handle fixed arm_pic_register. + 2025-03-05 Uros Bizjak Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 5f4d7bc93d2..e0b09102792 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250307 +20250308 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 5bc8a0143cd..b08e26603d7 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,11 @@ +2025-03-07 Christophe Lyon + + Backported from master: + 2025-03-07 Christophe Lyon + + PR target/115485 + * g++.target/arm/pr115485.C: New test. + 2025-03-05 Jakub Jelinek Backported from master: commit 11c933c3b6528251765f807b2a82d9e43dfa4673 Author: GCC Administrator Date: Sun Mar 9 00:19:08 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index e0b09102792..747d240bdb0 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250308 +20250309 commit ecdf9441e6aea65720a38bc35aa6857f83a2297d Author: GCC Administrator Date: Mon Mar 10 00:18:58 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 747d240bdb0..95dbf27d32a 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250309 +20250310 commit dfc6b286fd7248ad671fb6b78bf5ba22f2083cc3 Author: GCC Administrator Date: Tue Mar 11 00:20:15 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 95dbf27d32a..4bd9e51ab1b 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250310 +20250311 commit 8643edbb888b4325d81393cebc5bacf9cb22d202 Author: Richard Sandiford Date: Tue Mar 11 15:51:54 2025 +0000 aarch64: Fix caller saves of VNx2QI [PR116238] The testcase contains a VNx2QImode pseudo that is live across a call and that cannot be allocated a call-preserved register. LRA quite reasonably tried to save it before the call and restore it afterwards. Unfortunately, the target told it to do that in SImode, even though punning between SImode and VNx2QImode is disallowed by both TARGET_CAN_CHANGE_MODE_CLASS and TARGET_MODES_TIEABLE_P. The natural class to use for SImode is GENERAL_REGS, so this led to an unsalvageable situation in which we had: (set (subreg:VNx2QI (reg:SI A) 0) (reg:VNx2QI B)) where A needed GENERAL_REGS and B needed FP_REGS. We therefore ended up in a reload loop. The hooks above should ensure that this situation can never occur for incoming subregs. It only happened here because the target explicitly forced it. The decision to use SImode for modes smaller than 4 bytes dates back to the beginning of the port, before 16-bit floating-point modes existed. I'm not sure whether promoting to SImode really makes sense for any FPR, but that's a separate performance/QoI discussion. For now, this patch just disallows using SImode when it is wrong for correctness reasons, since that should be safer to backport. gcc/ PR testsuite/116238 * config/aarch64/aarch64.cc (aarch64_hard_regno_caller_save_mode): Only return SImode if we can convert to and from it. gcc/testsuite/ PR testsuite/116238 * gcc.target/aarch64/sve/pr116238.c: New test. (cherry picked from commit ec9d6d45191f639482344362d048294e74587ca3) diff --git a/gcc/config/aarch64/aarch64.cc b/gcc/config/aarch64/aarch64.cc index cd2f4053a1a..be0d958dcf6 100644 --- a/gcc/config/aarch64/aarch64.cc +++ b/gcc/config/aarch64/aarch64.cc @@ -4229,10 +4229,11 @@ aarch64_hard_regno_caller_save_mode (unsigned regno, unsigned, unnecessarily significant. */ if (PR_REGNUM_P (regno)) return mode; - if (known_ge (GET_MODE_SIZE (mode), 4)) - return mode; - else + if (known_lt (GET_MODE_SIZE (mode), 4) + && REG_CAN_CHANGE_MODE_P (regno, mode, SImode) + && REG_CAN_CHANGE_MODE_P (regno, SImode, mode)) return SImode; + return mode; } /* Return true if I's bits are consecutive ones from the MSB. */ diff --git a/gcc/testsuite/gcc.target/aarch64/sve/pr116238.c b/gcc/testsuite/gcc.target/aarch64/sve/pr116238.c new file mode 100644 index 00000000000..fe66b198107 --- /dev/null +++ b/gcc/testsuite/gcc.target/aarch64/sve/pr116238.c @@ -0,0 +1,13 @@ +/* { dg-additional-options "-O2 -msve-vector-bits=128" } */ + +void foo(); +typedef unsigned char v2qi __attribute__((vector_size(2))); +void f(v2qi *ptr) +{ + v2qi x = *ptr; + asm volatile ("" :: "w" (x)); + asm volatile ("" ::: "d8", "d9", "d10", "d11", "d12", "d13", "d14", "d15"); + foo(); + asm volatile ("" :: "w" (x)); + *ptr = x; +} commit 4f61ae081d694d3a1e99b45680f3af25ab17e3a8 Author: Richard Sandiford Date: Tue Mar 11 15:51:54 2025 +0000 aarch64: Fix folding of degenerate svwhilele case [PR117045] The svwhilele folder mishandled the degenerate case in which the second argument is the maximum integer. In that case, the result is all-true regardless of the first parameter: If the second scalar operand is equal to the maximum signed integer value then a condition which includes an equality test can never fail and the result will be an all-true predicate. This is because the conceptual "increment the first operand by 1 after each element" is done modulo the range of the operand. The GCC code was instead treating it as infinite precision. whilele_5.c even had a test for the incorrect behaviour. The easiest fix seemed to be to handle that case specially before doing constant folding. This also copes with variable first operands. gcc/ PR target/116999 PR target/117045 * config/aarch64/aarch64-sve-builtins-base.cc (svwhilelx_impl::fold): Check for WHILELTs of the minimum value and WHILELEs of the maximum value. Fold them to all-false and all-true respectively. gcc/testsuite/ PR target/116999 PR target/117045 * gcc.target/aarch64/sve/acle/general/whilele_5.c: Fix bogus expected result. * gcc.target/aarch64/sve/acle/general/whilele_11.c: New test. * gcc.target/aarch64/sve/acle/general/whilele_12.c: Likewise. (cherry picked from commit 50e7c51b0a0e9dc1d93f829016ae743b4f2e5070) diff --git a/gcc/config/aarch64/aarch64-sve-builtins-base.cc b/gcc/config/aarch64/aarch64-sve-builtins-base.cc index f96cb3ccc7b..a3fc474f92e 100644 --- a/gcc/config/aarch64/aarch64-sve-builtins-base.cc +++ b/gcc/config/aarch64/aarch64-sve-builtins-base.cc @@ -2412,7 +2412,9 @@ public: : while_comparison (unspec_for_sint, unspec_for_uint), m_eq_p (eq_p) {} - /* Try to fold a call by treating its arguments as constants of type T. */ + /* Try to fold a call by treating its arguments as constants of type T. + We have already filtered out the degenerate cases of X .LT. MIN + and X .LE. MAX. */ template gimple * fold_type (gimple_folder &f) const @@ -2465,6 +2467,13 @@ public: gimple * fold (gimple_folder &f) const OVERRIDE { + /* Filter out cases where the condition is always true or always false. */ + tree arg1 = gimple_call_arg (f.call, 1); + if (!m_eq_p && operand_equal_p (arg1, TYPE_MIN_VALUE (TREE_TYPE (arg1)))) + return f.fold_to_pfalse (); + if (m_eq_p && operand_equal_p (arg1, TYPE_MAX_VALUE (TREE_TYPE (arg1)))) + return f.fold_to_ptrue (); + if (f.type_suffix (1).unsigned_p) return fold_type (f); else diff --git a/gcc/testsuite/gcc.target/aarch64/sve/acle/general/whilele_11.c b/gcc/testsuite/gcc.target/aarch64/sve/acle/general/whilele_11.c new file mode 100644 index 00000000000..2be9dc5c534 --- /dev/null +++ b/gcc/testsuite/gcc.target/aarch64/sve/acle/general/whilele_11.c @@ -0,0 +1,31 @@ +/* { dg-do compile } */ +/* { dg-options "-O2" } */ + +#include +#include + +svbool_t +f1 (volatile int32_t *ptr) +{ + return svwhilelt_b8_s32 (*ptr, INT32_MIN); +} + +svbool_t +f2 (volatile uint32_t *ptr) +{ + return svwhilelt_b16_u32 (*ptr, 0); +} + +svbool_t +f3 (volatile int64_t *ptr) +{ + return svwhilelt_b32_s64 (*ptr, INT64_MIN); +} + +svbool_t +f4 (volatile uint64_t *ptr) +{ + return svwhilelt_b64_u64 (*ptr, 0); +} + +/* { dg-final { scan-assembler-times {\tpfalse\tp[0-9]+\.b\n} 4 } } */ diff --git a/gcc/testsuite/gcc.target/aarch64/sve/acle/general/whilele_12.c b/gcc/testsuite/gcc.target/aarch64/sve/acle/general/whilele_12.c new file mode 100644 index 00000000000..713065c3145 --- /dev/null +++ b/gcc/testsuite/gcc.target/aarch64/sve/acle/general/whilele_12.c @@ -0,0 +1,34 @@ +/* { dg-do compile } */ +/* { dg-options "-O2" } */ + +#include +#include + +svbool_t +f1 (volatile int32_t *ptr) +{ + return svwhilele_b8_s32 (*ptr, INT32_MAX); +} + +svbool_t +f2 (volatile uint32_t *ptr) +{ + return svwhilele_b16_u32 (*ptr, UINT32_MAX); +} + +svbool_t +f3 (volatile int64_t *ptr) +{ + return svwhilele_b32_s64 (*ptr, INT64_MAX); +} + +svbool_t +f4 (volatile uint64_t *ptr) +{ + return svwhilele_b64_u64 (*ptr, UINT64_MAX); +} + +/* { dg-final { scan-assembler {\tptrue\tp[0-9]+\.b(?:, all)\n} } } */ +/* { dg-final { scan-assembler {\tptrue\tp[0-9]+\.h(?:, all)\n} } } */ +/* { dg-final { scan-assembler {\tptrue\tp[0-9]+\.s(?:, all)\n} } } */ +/* { dg-final { scan-assembler {\tptrue\tp[0-9]+\.d(?:, all)\n} } } */ diff --git a/gcc/testsuite/gcc.target/aarch64/sve/acle/general/whilele_5.c b/gcc/testsuite/gcc.target/aarch64/sve/acle/general/whilele_5.c index ada958b29c1..dac4515cf40 100644 --- a/gcc/testsuite/gcc.target/aarch64/sve/acle/general/whilele_5.c +++ b/gcc/testsuite/gcc.target/aarch64/sve/acle/general/whilele_5.c @@ -28,7 +28,7 @@ test3 (svbool_t *ptr) *ptr = svwhilele_b16_s32 (0x7ffffffb, 0x7fffffff); } -/* { dg-final { scan-assembler {\tptrue\tp[0-7]\.h, vl5\n} } } */ +/* { dg-final { scan-assembler {\tptrue\tp[0-9]+\.h(?:, all)\n} } } */ void test4 (svbool_t *ptr) commit 587b370c8492aadaab14c57e242c66778cc78891 Author: Richard Sandiford Date: Tue Mar 11 15:51:55 2025 +0000 Fix folding of BIT_NOT_EXPR for POLY_INT_CST [PR118976] There was an embarrassing typo in the folding of BIT_NOT_EXPR for POLY_INT_CSTs: it used - rather than ~ on the poly_int. Not sure how that happened, but it might have been due to the way that ~x is implemented as -1 - x internally. gcc/ PR tree-optimization/118976 * fold-const.cc (const_unop): Use ~ rather than - for BIT_NOT_EXPR. * config/aarch64/aarch64.cc (aarch64_test_sve_folding): New function. (aarch64_run_selftests): Run it. (cherry picked from commit 78380fd7f743e23dfdf013d68a2f0347e1511550) diff --git a/gcc/config/aarch64/aarch64.cc b/gcc/config/aarch64/aarch64.cc index be0d958dcf6..72d737d6222 100644 --- a/gcc/config/aarch64/aarch64.cc +++ b/gcc/config/aarch64/aarch64.cc @@ -27541,6 +27541,16 @@ aarch64_test_fractional_cost () ASSERT_EQ (cf (1, 2).as_double (), 0.5); } +/* Test SVE arithmetic folding. */ + +static void +aarch64_test_sve_folding () +{ + tree res = fold_unary (BIT_NOT_EXPR, ssizetype, + ssize_int (poly_int64 (1, 1))); + ASSERT_TRUE (operand_equal_p (res, ssize_int (poly_int64 (-2, -1)))); +} + /* Run all target-specific selftests. */ static void @@ -27548,6 +27558,7 @@ aarch64_run_selftests (void) { aarch64_test_loading_full_dump (); aarch64_test_fractional_cost (); + aarch64_test_sve_folding (); } } // namespace selftest diff --git a/gcc/fold-const.cc b/gcc/fold-const.cc index d81a71c41a1..391f1109540 100644 --- a/gcc/fold-const.cc +++ b/gcc/fold-const.cc @@ -1802,7 +1802,7 @@ const_unop (enum tree_code code, tree type, tree arg0) if (TREE_CODE (arg0) == INTEGER_CST) return fold_not_const (arg0, type); else if (POLY_INT_CST_P (arg0)) - return wide_int_to_tree (type, -poly_int_cst_value (arg0)); + return wide_int_to_tree (type, ~poly_int_cst_value (arg0)); /* Perform BIT_NOT_EXPR on each element individually. */ else if (TREE_CODE (arg0) == VECTOR_CST) { commit 8d8eb5327f44472e8769807a6e632e2e0d445bb9 Author: Iain Buclaw Date: Tue Mar 11 17:56:18 2025 +0100 d: Fix regression returning from function with invariants [PR119139] An optimization was added in GDC-12 which sets the TREE_READONLY flag on all local variables with the storage class `const' assigned. For some reason, const is also being added by the front-end to `__result' variables in non-virtual functions, which ends up getting wrong code by the gimplify pass promoting the local to static storage. A bug has been raised upstream, as this looks like an error in the AST. For now, turn off setting TREE_READONLY on all result variables. PR d/119139 gcc/d/ChangeLog: * decl.cc (get_symbol_decl): Don't set TREE_READONLY for __result declarations. gcc/testsuite/ChangeLog: * gdc.dg/pr119139.d: New test. (cherry picked from commit 81582ca6cb692098c1bda7995aec46c6cbfbfcb3) diff --git a/gcc/d/decl.cc b/gcc/d/decl.cc index 6c2705d9864..00eac62d7a3 100644 --- a/gcc/d/decl.cc +++ b/gcc/d/decl.cc @@ -1300,7 +1300,7 @@ get_symbol_decl (Declaration *decl) /* `const` applies to data that cannot be changed by the const reference to that data. It may, however, be changed by another reference to that same data. */ - if (vd->isConst () && !vd->isDataseg ()) + if (vd->isConst () && !vd->isResult () && !vd->isDataseg ()) TREE_READONLY (decl->csym) = 1; } diff --git a/gcc/testsuite/gdc.dg/pr119139.d b/gcc/testsuite/gdc.dg/pr119139.d new file mode 100644 index 00000000000..dc42c411e39 --- /dev/null +++ b/gcc/testsuite/gdc.dg/pr119139.d @@ -0,0 +1,24 @@ +// https://gcc.gnu.org/bugzilla/show_bug.cgi?id=119139 +// { dg-do compile } +// { dg-options "-fdump-tree-gimple" } +string toString() +{ + return "1"; +} + +struct B +{ + ulong n; + + invariant{} + + string str() + { + if (n == 0) + { + return "0"; + } + return toString(); + } +} +// { dg-final { scan-tree-dump-not "static const struct __result =" "gimple" } } commit be4a99fa3eb3438a77eca19be45b2a39b24c57a6 Author: GCC Administrator Date: Wed Mar 12 00:20:32 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 0d9991af398..bdda9d5b047 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,34 @@ +2025-03-11 Richard Sandiford + + Backported from master: + 2025-03-04 Richard Sandiford + + PR tree-optimization/118976 + * fold-const.cc (const_unop): Use ~ rather than - for BIT_NOT_EXPR. + * config/aarch64/aarch64.cc (aarch64_test_sve_folding): New function. + (aarch64_run_selftests): Run it. + +2025-03-11 Richard Sandiford + + Backported from master: + 2024-10-14 Richard Sandiford + + PR target/116999 + PR target/117045 + * config/aarch64/aarch64-sve-builtins-base.cc + (svwhilelx_impl::fold): Check for WHILELTs of the minimum value + and WHILELEs of the maximum value. Fold them to all-false and + all-true respectively. + +2025-03-11 Richard Sandiford + + Backported from master: + 2024-08-21 Richard Sandiford + + PR testsuite/116238 + * config/aarch64/aarch64.cc (aarch64_hard_regno_caller_save_mode): + Only return SImode if we can convert to and from it. + 2025-03-07 Christophe Lyon Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 4bd9e51ab1b..a2876d18ceb 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250311 +20250312 diff --git a/gcc/d/ChangeLog b/gcc/d/ChangeLog index 3ac009b8491..5bd0ba9f79f 100644 --- a/gcc/d/ChangeLog +++ b/gcc/d/ChangeLog @@ -1,3 +1,12 @@ +2025-03-11 Iain Buclaw + + Backported from master: + 2025-03-11 Iain Buclaw + + PR d/119139 + * decl.cc (get_symbol_decl): Don't set TREE_READONLY for __result + declarations. + 2025-03-02 Iain Buclaw Backported from master: diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index b08e26603d7..2cc52c8d6d8 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,31 @@ +2025-03-11 Iain Buclaw + + Backported from master: + 2025-03-11 Iain Buclaw + + PR d/119139 + * gdc.dg/pr119139.d: New test. + +2025-03-11 Richard Sandiford + + Backported from master: + 2024-10-14 Richard Sandiford + + PR target/116999 + PR target/117045 + * gcc.target/aarch64/sve/acle/general/whilele_5.c: Fix bogus + expected result. + * gcc.target/aarch64/sve/acle/general/whilele_11.c: New test. + * gcc.target/aarch64/sve/acle/general/whilele_12.c: Likewise. + +2025-03-11 Richard Sandiford + + Backported from master: + 2024-08-21 Richard Sandiford + + PR testsuite/116238 + * gcc.target/aarch64/sve/pr116238.c: New test. + 2025-03-07 Christophe Lyon Backported from master: commit 901ed4c8191b9f849b8a4381393c5c55d3882e2f Author: GCC Administrator Date: Thu Mar 13 00:19:57 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index a2876d18ceb..450c315bb77 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250312 +20250313 commit 194237722bb2e9290d17c0b4e34ec541d8b99bf3 Author: GCC Administrator Date: Fri Mar 14 00:19:23 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 450c315bb77..f662df52798 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250313 +20250314 commit fcf59b21db0b9f2daf38c4e3e68a9a1be17d7358 Author: GCC Administrator Date: Sat Mar 15 00:20:20 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index f662df52798..75b3f29cb4b 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250314 +20250315 commit 446c1e3f4fd96f4fe89f0cacf0f483e868c329f8 Author: GCC Administrator Date: Sun Mar 16 00:20:00 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 75b3f29cb4b..a99ba55663e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250315 +20250316 commit a4259a045c03ac2463fba8d57a501e08f9f373cc Author: GCC Administrator Date: Mon Mar 17 00:19:25 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index a99ba55663e..b0b6a230627 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250316 +20250317 commit 6dba189937f30e69ea4c33b09559d29dbadab25e Author: GCC Administrator Date: Tue Mar 18 00:20:36 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index b0b6a230627..c0dd0682172 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250317 +20250318 commit 8b36160fa0eb27aa6006ded824c7b22da2ef99ec Author: GCC Administrator Date: Wed Mar 19 00:19:26 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c0dd0682172..22523f5c82f 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250318 +20250319 commit 2f7620748f0f8da4cd4017d4507063f7065f2f25 Author: GCC Administrator Date: Thu Mar 20 00:20:51 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 22523f5c82f..b76e7dd6410 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250319 +20250320 commit 33c79764340314a5dabd2c68fa9ea0f33bac9a56 Author: GCC Administrator Date: Fri Mar 21 00:20:10 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index b76e7dd6410..2340aa0f862 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250320 +20250321 commit bea69439c596d37f7d33918ac7e1f864b4ab274d Author: GCC Administrator Date: Sat Mar 22 09:28:13 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 2340aa0f862..a03cb0179d3 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250321 +20250322 commit 35406a0930e4855864df8ce01544d112e1a37c52 Author: GCC Administrator Date: Sun Mar 23 00:18:55 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index a03cb0179d3..2e55fc0d012 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250322 +20250323 commit fadc25ed2e82fdaf7f44a9722f5555504474e210 Author: GCC Administrator Date: Mon Mar 24 00:19:41 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 2e55fc0d012..2b52eecc8dd 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250323 +20250324 commit 4b3c7e34a806421995cb5db566aa4d2a499dd886 Author: GCC Administrator Date: Tue Mar 25 00:20:53 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 2b52eecc8dd..5daf0b7b9c6 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250324 +20250325 commit 7f18798956fe184c69b23ddadb7acc573e3ff583 Author: GCC Administrator Date: Wed Mar 26 00:18:57 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 5daf0b7b9c6..f90d5fa6219 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250325 +20250326 commit 8204a7295ed9531787796d19188eedb5a4bda427 Author: GCC Administrator Date: Thu Mar 27 00:20:08 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index f90d5fa6219..0e0aca037f5 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250326 +20250327 commit 4b83e7c83427eb540951589ca1936c7d7399e080 Author: GCC Administrator Date: Fri Mar 28 00:21:02 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 0e0aca037f5..2bb0fd3370e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250327 +20250328 commit 5e688dbf80446d6832cb82b04463389e4eccb6ac Author: GCC Administrator Date: Sat Mar 29 00:21:19 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 2bb0fd3370e..d03f9f16051 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250328 +20250329 commit ff21bb42b87c265ff9919fdcf7bedd265a262d64 Author: GCC Administrator Date: Sun Mar 30 00:19:50 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d03f9f16051..a929d4e763e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250329 +20250330 commit 76806519625754787c1c707856143ae364151579 Author: GCC Administrator Date: Mon Mar 31 00:20:31 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index a929d4e763e..7430ba012c0 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250330 +20250331 commit 2b200f96b59ccb0f99e182630058545eb60c5206 Author: GCC Administrator Date: Tue Apr 1 00:20:52 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 7430ba012c0..9908187aed0 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250331 +20250401 commit 0c66b2f14b5d4cd46388d79b4a73a77433e1dd22 Author: Andrew Carlotti Date: Tue Jul 30 16:26:04 2024 +0100 aarch64: Use PAUTH instead of V8_3A in some places gcc/ChangeLog: PR target/119372 * config/aarch64/aarch64.cc (aarch64_expand_epilogue): Use TARGET_PAUTH. * config/aarch64/aarch64.md: Update comment. (cherry-picked from commit 20385cb92cbd4a1934661ab97a162c1e25935836) diff --git a/gcc/config/aarch64/aarch64.cc b/gcc/config/aarch64/aarch64.cc index 72d737d6222..74d74976f4a 100644 --- a/gcc/config/aarch64/aarch64.cc +++ b/gcc/config/aarch64/aarch64.cc @@ -10048,12 +10048,12 @@ aarch64_expand_epilogue (bool for_sibcall) 1) Sibcalls don't return in a normal way, so if we're about to call one we must authenticate. - 2) The RETAA instruction is not available before ARMv8.3-A, so if we are - generating code for !TARGET_ARMV8_3 we can't use it and must + 2) The RETAA instruction is not available without FEAT_PAuth, so if we + are generating code for !TARGET_PAUTH we can't use it and must explicitly authenticate. */ if (aarch64_return_address_signing_enabled () - && (for_sibcall || !TARGET_ARMV8_3)) + && (for_sibcall || !TARGET_PAUTH)) { switch (aarch64_ra_sign_key) { diff --git a/gcc/config/aarch64/aarch64.md b/gcc/config/aarch64/aarch64.md index 47b70feff02..0b205b43d97 100644 --- a/gcc/config/aarch64/aarch64.md +++ b/gcc/config/aarch64/aarch64.md @@ -7124,11 +7124,11 @@ [(set_attr "type" "f_cvtf2i")] ) -;; Pointer authentication patterns are always provided. In architecture -;; revisions prior to ARMv8.3-A these HINT instructions operate as NOPs. +;; Pointer authentication patterns are always provided. On targets that +;; don't implement FEAT_PAuth these HINT instructions operate as NOPs. ;; This lets the user write portable software which authenticates pointers -;; when run on something which implements ARMv8.3-A, and which runs -;; correctly, but does not authenticate pointers, where ARMv8.3-A is not +;; when run on something which implements FEAT_PAuth, and which runs +;; correctly, but does not authenticate pointers, where FEAT_PAuth is not ;; implemented. ;; Signing/Authenticating R30 using SP as the salt. commit 813ca3de452bcd50cfc00027f6d1e19b6f82a81e Author: GCC Administrator Date: Wed Apr 2 00:21:11 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index bdda9d5b047..6a43292b7ad 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,11 @@ +2025-04-01 Andrew Carlotti + + PR target/119372 + * config/aarch64/aarch64.cc + (aarch64_expand_epilogue): Use TARGET_PAUTH. + * config/aarch64/aarch64.md: Update comment. + (cherry-picked from commit 20385cb92cbd4a1934661ab97a162c1e25935836) + 2025-03-11 Richard Sandiford Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 9908187aed0..2508b1fc5e7 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250401 +20250402 commit 8250987299d01ff318b73a02e4ba245afd46b64b Author: Jonathan Wakely Date: Thu Feb 8 13:59:42 2024 +0000 libstdc++: Avoid aliasing violation in std::valarray [PR99117] The call to __valarray_copy constructs an _Array object to refer to this->_M_data but that means that accesses to this->_M_data are through a restrict-qualified pointer. This leads to undefined behaviour when copying from an _Expr object that actually aliases this->_M_data. Replace the call to __valarray_copy with a plain loop. I think this removes the only use of that overload of __valarray_copy, so it could probably be removed. I haven't done that here. libstdc++-v3/ChangeLog: PR libstdc++/99117 * include/std/valarray (valarray::operator=(const _Expr&)): Use loop to copy instead of __valarray_copy with _Array. * testsuite/26_numerics/valarray/99117.cc: New test. (cherry picked from commit b58f0e5216a3053486e7f1aa96c3f2443b14d630) diff --git a/libstdc++-v3/include/std/valarray b/libstdc++-v3/include/std/valarray index 32e6e7e8a76..d4995977251 100644 --- a/libstdc++-v3/include/std/valarray +++ b/libstdc++-v3/include/std/valarray @@ -838,7 +838,13 @@ _GLIBCXX_BEGIN_NAMESPACE_VERSION // _GLIBCXX_RESOLVE_LIB_DEFECTS // 630. arrays of valarray. if (_M_size == __e.size()) - std::__valarray_copy(__e, _M_size, _Array<_Tp>(_M_data)); + { + // Copy manually instead of using __valarray_copy, because __e might + // alias _M_data and the _Array param type of __valarray_copy uses + // restrict which doesn't allow aliasing. + for (size_t __i = 0; __i < _M_size; ++__i) + _M_data[__i] = __e[__i]; + } else { if (_M_data) diff --git a/libstdc++-v3/testsuite/26_numerics/valarray/99117.cc b/libstdc++-v3/testsuite/26_numerics/valarray/99117.cc new file mode 100644 index 00000000000..81621bd079a --- /dev/null +++ b/libstdc++-v3/testsuite/26_numerics/valarray/99117.cc @@ -0,0 +1,17 @@ +// { dg-do run { target c++11 } } + +// PR libstdc++/99117 cannot accumulate std::valarray + +#include +#include +#include + +int main() +{ + std::vector> v = {{1,1}, {2,2}}; + std::valarray sum(2); + for (const auto& e : v) + sum = sum + e; + VERIFY(sum[0]==3); + VERIFY(sum[1]==3); +} commit f3878885b1aa4ab96b6905d9485a1261934e291b Author: Jonathan Wakely Date: Fri Mar 31 13:44:04 2023 +0100 libstdc++: Teach optimizer that empty COW strings are empty [PR107087] The compiler doesn't know about the invariant that the _S_empty_rep() object is immutable and so _M_length and _M_refcount are always zero. This means that we get warnings about writing possibly-non-zero length strings into buffers that can't hold them. If we teach the compiler that the empty rep is always zero length, it knows it can be copied into any buffer. For Stage 1 we might want to also consider adding this to capacity(): if (_S_empty_rep()._M_capacity != 0) __builtin_unreachable(); And this to _Rep::_M_is_leaked() and _Rep::_M_is_shared(): if (_S_empty_rep()._M_refcount != 0) __builtin_unreachable(); libstdc++-v3/ChangeLog: PR tree-optimization/107087 * include/bits/cow_string.h (basic_string::size()): Add optimizer hint that _S_empty_rep()._M_length is always zero. (basic_string::length()): Call size(). (cherry picked from commit 4969dcd2b7a94ce6c0d07225b21b5f3c040a4902) diff --git a/libstdc++-v3/include/bits/cow_string.h b/libstdc++-v3/include/bits/cow_string.h index 6dd23883429..f79fd879582 100644 --- a/libstdc++-v3/include/bits/cow_string.h +++ b/libstdc++-v3/include/bits/cow_string.h @@ -909,17 +909,24 @@ _GLIBCXX_BEGIN_NAMESPACE_VERSION public: // Capacity: + /// Returns the number of characters in the string, not including any /// null-termination. size_type size() const _GLIBCXX_NOEXCEPT - { return _M_rep()->_M_length; } + { +#if _GLIBCXX_FULLY_DYNAMIC_STRING == 0 && __OPTIMIZE__ + if (_S_empty_rep()._M_length != 0) + __builtin_unreachable(); +#endif + return _M_rep()->_M_length; + } /// Returns the number of characters in the string, not including any /// null-termination. size_type length() const _GLIBCXX_NOEXCEPT - { return _M_rep()->_M_length; } + { return size(); } /// Returns the size() of the largest possible %string. size_type commit 3c4fbdbacd386e2bee5c826a0f75ccc2b2d34f3f Author: GCC Administrator Date: Thu Apr 3 00:20:14 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 2508b1fc5e7..75f45468e7e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250402 +20250403 diff --git a/libstdc++-v3/ChangeLog b/libstdc++-v3/ChangeLog index 176f0d3de3f..0e0629219c9 100644 --- a/libstdc++-v3/ChangeLog +++ b/libstdc++-v3/ChangeLog @@ -1,3 +1,23 @@ +2025-04-02 Jonathan Wakely + + Backported from master: + 2023-03-31 Jonathan Wakely + + PR tree-optimization/107087 + * include/bits/cow_string.h (basic_string::size()): Add + optimizer hint that _S_empty_rep()._M_length is always zero. + (basic_string::length()): Call size(). + +2025-04-02 Jonathan Wakely + + Backported from master: + 2024-02-15 Jonathan Wakely + + PR libstdc++/99117 + * include/std/valarray (valarray::operator=(const _Expr&)): + Use loop to copy instead of __valarray_copy with _Array. + * testsuite/26_numerics/valarray/99117.cc: New test. + 2024-10-07 Jonathan Wakely * include/std/string_view (basic_string_view::copy) Use explicit commit d4d63be9e9517a6ffd01ddb6ea19e50952a79134 Author: GCC Administrator Date: Fri Apr 4 00:19:26 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 75f45468e7e..c36e5b25617 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250403 +20250404 commit b65f4fe10996c247751b0636773009659d92ab6c Author: Eric Botcazou Date: Fri Apr 4 11:45:23 2025 +0200 Ada: Fix thinko in Eigensystem for complex Hermitian matrices The implementation solves the eigensystem for a NxN complex Hermitian matrix by first solving it for a 2Nx2N real symmetric matrix and then interpreting the 2Nx1 real vectors as Nx1 complex ones, but the last step does not work. The patch fixes the last step and also performs a small cleanup throughout the implementation, mostly in the commentary and without functional changes. gcc/ada/ * libgnat/a-ngcoar.adb (Eigensystem): Adjust notation and fix the layout of the real symmetric matrix in the main comment. Adjust the layout of the associated code accordingly and correctly turn the 2Nx1 real vectors into Nx1 complex ones. (Eigenvalues): Minor similar tweaks. * libgnat/a-ngrear.adb (Jacobi): Minor tweaks in the main comment. Adjust notation and corresponding parameter names of functions. Fix call to Unit_Matrix routine. Adjust the comment describing the various kinds of iterations to match the implementation. diff --git a/gcc/ada/libgnat/a-ngcoar.adb b/gcc/ada/libgnat/a-ngcoar.adb index 8dfbc3b174a..cbbc6d274aa 100644 --- a/gcc/ada/libgnat/a-ngcoar.adb +++ b/gcc/ada/libgnat/a-ngcoar.adb @@ -1058,19 +1058,21 @@ package body Ada.Numerics.Generic_Complex_Arrays is is N : constant Natural := Length (A); - -- For a Hermitian matrix C, we convert the eigenvalue problem to a - -- real symmetric one: if C = A + i * B, then the (N, N) complex + -- For a Hermitian matrix A, we convert the eigenvalue problem to a + -- real symmetric one: if A = X + i * Y, then the (N, N) complex -- eigenvalue problem: - -- (A + i * B) * (u + i * v) = Lambda * (u + i * v) + -- + -- (X + i * Y) * (u + i * v) = Lambda * (u + i * v) -- -- is equivalent to the (2 * N, 2 * N) real eigenvalue problem: - -- [ A, B ] [ u ] = Lambda * [ u ] - -- [ -B, A ] [ v ] [ v ] -- - -- Note that the (2 * N, 2 * N) matrix above is symmetric, as - -- Transpose (A) = A and Transpose (B) = -B if C is Hermitian. + -- [ X, -Y ] [ u ] = Lambda * [ u ] + -- [ Y, X ] [ v ] [ v ] + -- + -- Note that the (2 * N, 2 * N) matrix M above is symmetric, because + -- Transpose (X) = X and Transpose (Y) = -Y as A is Hermitian. - -- We solve this eigensystem using the real-valued algorithms. The final + -- We solve this eigensystem using the real-valued algorithm. The final -- result will have every eigenvalue twice, so in the sorted output we -- just pick every second value, with associated eigenvector u + i * v. @@ -1085,10 +1087,8 @@ package body Ada.Numerics.Generic_Complex_Arrays is C : constant Complex := (A (A'First (1) + (J - 1), A'First (2) + (K - 1))); begin - M (J, K) := Re (C); - M (J + N, K + N) := Re (C); - M (J + N, K) := Im (C); - M (J, K + N) := -Im (C); + M (J, K) := Re (C); M (J, K + N) := -Im (C); + M (J + N, K) := Im (C); M (J + N, K + N) := Re (C); end; end loop; end loop; @@ -1103,10 +1103,9 @@ package body Ada.Numerics.Generic_Complex_Arrays is for K in 1 .. N loop declare - Row : constant Integer := Vectors'First (2) + (K - 1); + Row : constant Integer := Vectors'First (1) + (K - 1); begin - Vectors (Row, Col) - := (Vecs (J * 2, Col), Vecs (J * 2, Col + N)); + Vectors (Row, Col) := (Vecs (K, 2 * J), Vecs (K + N, 2 * J)); end; end loop; end; @@ -1118,13 +1117,14 @@ package body Ada.Numerics.Generic_Complex_Arrays is ----------------- function Eigenvalues (A : Complex_Matrix) return Real_Vector is - -- See Eigensystem for a description of the algorithm - N : constant Natural := Length (A); - R : Real_Vector (A'Range (1)); + + -- See Eigensystem for a description of the algorithm M : Real_Matrix (1 .. 2 * N, 1 .. 2 * N); + R : Real_Vector (A'Range (1)); Vals : Real_Vector (1 .. 2 * N); + begin for J in 1 .. N loop for K in 1 .. N loop @@ -1132,10 +1132,8 @@ package body Ada.Numerics.Generic_Complex_Arrays is C : constant Complex := (A (A'First (1) + (J - 1), A'First (2) + (K - 1))); begin - M (J, K) := Re (C); - M (J + N, K + N) := Re (C); - M (J + N, K) := Im (C); - M (J, K + N) := -Im (C); + M (J, K) := Re (C); M (J, K + N) := -Im (C); + M (J + N, K) := Im (C); M (J + N, K + N) := Re (C); end; end loop; end loop; diff --git a/gcc/ada/libgnat/a-ngrear.adb b/gcc/ada/libgnat/a-ngrear.adb index 844d6264ee7..1d07a8a9b5f 100644 --- a/gcc/ada/libgnat/a-ngrear.adb +++ b/gcc/ada/libgnat/a-ngrear.adb @@ -85,7 +85,7 @@ package body Ada.Numerics.Generic_Real_Arrays is function Is_Symmetric (A : Real_Matrix) return Boolean is (Transpose (A) = A); - -- Return True iff A is symmetric, see RM G.3.1 (90). + -- Return True iff A is symmetric, see RM G.3.1 (90) function Is_Tiny (Value, Compared_To : Real) return Boolean is (abs Compared_To + 100.0 * abs (Value) = abs Compared_To); @@ -104,7 +104,7 @@ package body Ada.Numerics.Generic_Real_Arrays is -- not a square matrix, and otherwise returns its length. procedure Rotate (X, Y : in out Real; Sin, Tau : Real); - -- Perform a Givens rotation + -- Perform a Givens rotation of angle Theta given by sin and sin/(1 + cos) procedure Sort_Eigensystem (Values : in out Real_Vector; @@ -525,13 +525,13 @@ package body Ada.Numerics.Generic_Real_Arrays is Vectors : out Real_Matrix; Compute_Vectors : Boolean) is - -- This subprogram uses Carl Gustav Jacob Jacobi's iterative method - -- for computing eigenvalues and eigenvectors and is based on + -- This subprogram uses Carl Gustav Jacob Jacobi's cyclic iterative + -- method for computing eigenvalues and eigenvectors and is based on -- Rutishauser's implementation. -- The given real symmetric matrix is transformed iteratively to -- diagonal form through a sequence of appropriately chosen elementary - -- orthogonal transformations, called Jacobi rotations here. + -- orthogonal transformations, called Jacobi rotations. -- The Jacobi method produces a systematic decrease of the sum of the -- squares of off-diagonal elements. Convergence to zero is quadratic, @@ -542,41 +542,44 @@ package body Ada.Numerics.Generic_Real_Arrays is -- best choice here, even though for large matrices other methods will -- be significantly more efficient in both time and space. - -- While the eigensystem computations are absolutely foolproof for all + -- While the eigensystem computation is absolutely foolproof for all -- real symmetric matrices, in presence of invalid values, or similar - -- exceptional situations it might not. In such cases the results cannot - -- be trusted and Constraint_Error is raised. + -- exceptional situations, it may not be. In such cases, the results + -- cannot be trusted and Constraint_Error is raised. -- Note: this implementation needs temporary storage for 2 * N + N**2 -- values of type Real. - Max_Iterations : constant := 50; - N : constant Natural := Length (A); + Max_Iterations : constant := 50; + N : constant Natural := Length (A); subtype Square_Matrix is Real_Matrix (1 .. N, 1 .. N); - -- In order to annihilate the M (Row, Col) element, the - -- rotation parameters Cos and Sin are computed as - -- follows: + -- In order to annihilate the M (Row, Col) element, the rotation angle + -- Theta is chosen as follows: - -- Theta = Cot (2.0 * Phi) - -- = (Diag (Col) - Diag (Row)) / (2.0 * M (Row, Col)) + -- Cot (2.0 * Theta) = (Diag (Col) - Diag (Row)) / (2.0 * M (Row, Col)) - -- Then Tan (Phi) as the smaller root (in modulus) of + -- If C = Cot (2.0 * Theta), then Tan (Theta) is computed as the smaller + -- root (in modulus) of: - -- T**2 + 2 * T * Theta = 1 (or 0.5 / Theta, if Theta is large) + -- X**2 + 2 * C * X - 1 = 0 - function Compute_Tan (Theta : Real) return Real is - (Real'Copy_Sign (1.0 / (abs Theta + Sqrt (1.0 + Theta**2)), Theta)); + -- or else as 0.5 / C, if C is large. + + function Compute_Tan (C : Real) return Real is + (Real'Copy_Sign (1.0 / (abs C + Sqrt (1.0 + C**2)), C)); function Compute_Tan (P, H : Real) return Real is - (if Is_Tiny (P, Compared_To => H) then P / H - else Compute_Tan (Theta => H / (2.0 * P))); + (if Is_Tiny (P, Compared_To => H) + then P / H + else Compute_Tan (C => H / (2.0 * P))); pragma Annotate (CodePeer, False_Positive, "divide by zero", "H, P /= 0"); function Sum_Strict_Upper (M : Square_Matrix) return Real; - -- Return the sum of all elements in the strict upper triangle of M + -- Return the sum of the absolute value of all the elements in the + -- strict upper triangle of M. ---------------------- -- Sum_Strict_Upper -- @@ -595,11 +598,13 @@ package body Ada.Numerics.Generic_Real_Arrays is return Sum; end Sum_Strict_Upper; + -- Local variables + M : Square_Matrix := A; -- Work space for solving eigensystem - Threshold : Real; - Sum : Real; Diag : Real_Vector (1 .. N); Diag_Adj : Real_Vector (1 .. N); + Sum : Real; + Threshold : Real; -- The vector Diag_Adj indicates the amount of change in each value, -- while Diag tracks the value itself and Values holds the values as @@ -621,22 +626,24 @@ package body Ada.Numerics.Generic_Real_Arrays is raise Constraint_Error with "matrix not symmetric"; end if; + Values := Diagonal (M); + -- Note: Only the locally declared matrix M and vectors (Diag, Diag_Adj) -- have lower bound equal to 1. The Vectors matrix may have -- different bounds, so take care indexing elements. Assignment -- as a whole is fine as sliding is automatic in that case. - Vectors := (if not Compute_Vectors then [1 .. 0 => [1 .. 0 => 0.0]] - else Unit_Matrix (Vectors'Length (1), Vectors'Length (2))); - Values := Diagonal (M); + Vectors := (if Compute_Vectors + then Unit_Matrix (N) + else [1 .. 0 => [1 .. 0 => 0.0]]); Sweep : for Iteration in 1 .. Max_Iterations loop - -- The first three iterations, perform rotation for any non-zero - -- element. After this, rotate only for those that are not much - -- smaller than the average off-diagnal element. After the fifth - -- iteration, additionally zero out off-diagonal elements that are - -- very small compared to elements on the diagonal with the same + -- During the first three iterations, perform the rotation only for + -- elements that are not much smaller than the average off-diagonal + -- element. After this, rotate for any non-zero elements. After the + -- fifth iteration, additionally zero out off-diagonal elements that + -- are very small compared to elements on the diagonal with the same -- column or row index. Sum := Sum_Strict_Upper (M); @@ -645,8 +652,8 @@ package body Ada.Numerics.Generic_Real_Arrays is Threshold := (if Iteration < 4 then 0.2 * Sum / Real (N**2) else 0.0); - -- Iterate over all off-diagonal elements, rotating any that have - -- an absolute value that exceeds the threshold. + -- Iterate over all off-diagonal elements, rotating any that have an + -- absolute value that exceeds the threshold. Diag := Values; Diag_Adj := [others => 0.0]; -- Accumulates adjustments to Diag @@ -654,11 +661,11 @@ package body Ada.Numerics.Generic_Real_Arrays is for Row in 1 .. N - 1 loop for Col in Row + 1 .. N loop - -- If, before the rotation M (Row, Col) is tiny compared to + -- If, before the rotation, M (Row, Col) is tiny compared to -- Diag (Row) and Diag (Col), rotation is skipped. This is -- meaningful, as it produces no larger error than would be -- produced anyhow if the rotation had been performed. - -- Suppress this optimization in the first four sweeps, so + -- Suppress this optimization in the first four iterations, so -- that this procedure can be used for computing eigenvectors -- of perturbed diagonal matrices. @@ -670,8 +677,8 @@ package body Ada.Numerics.Generic_Real_Arrays is elsif abs M (Row, Col) > Threshold then Perform_Rotation : declare - Tan : constant Real := Compute_Tan (M (Row, Col), - Diag (Col) - Diag (Row)); + Tan : constant Real := + Compute_Tan (M (Row, Col), Diag (Col) - Diag (Row)); Cos : constant Real := 1.0 / Sqrt (1.0 + Tan**2); Sin : constant Real := Tan * Cos; Tau : constant Real := Sin / (1.0 + Cos); @@ -710,7 +717,7 @@ package body Ada.Numerics.Generic_Real_Arrays is Values := Values + Diag_Adj; end loop Sweep; - -- All normal matrices with valid values should converge perfectly. + -- All normal matrices with valid values should converge perfectly if Sum /= 0.0 then raise Constraint_Error with "eigensystem solution does not converge"; commit 8eca273335750e100efff8f45aee285290be3c46 Author: Richard Biener Date: Mon Feb 3 09:55:50 2025 +0100 tree-optimization/118717 - store commoning vs. abnormals When we sink common stores in cselim or the sink pass we have to make sure to not introduce overlapping lifetimes for abnormals used in the ref. The easiest is to avoid sinking stmts which reference abnormals at all which is what the following does. PR tree-optimization/118717 * tree-ssa-phiopt.cc (cond_if_else_store_replacement_1): Do not common stores referencing abnormal SSA names. * tree-ssa-sink.cc (sink_common_stores_to_bb): Likewise. * gcc.dg/torture/pr118717.c: New testcase. (cherry picked from commit fbcbbfe2bf83eb8b1347144eeca37b06be5a8bb5) diff --git a/gcc/testsuite/gcc.dg/torture/pr118717.c b/gcc/testsuite/gcc.dg/torture/pr118717.c new file mode 100644 index 00000000000..42dc5ec84f2 --- /dev/null +++ b/gcc/testsuite/gcc.dg/torture/pr118717.c @@ -0,0 +1,41 @@ +/* { dg-do compile } */ + +void jj(void); +int ff1(void) __attribute__((__returns_twice__)); +struct s2 { + int prev; +}; +typedef struct s1 { + unsigned interrupt_flag; + unsigned interrupt_mask; + int tag; + int state; +}s1; +int ff(void); +static inline +int mm(s1 *ec) { + if (ff()) + if (ec->interrupt_flag & ~(ec)->interrupt_mask) + return 0; +} +void ll(s1 *ec) { + int t = 1; + int state; + if (t) + { + { + s1 *const _ec = ec; + struct s2 _tag = {0}; + if (ff1()) + state = ec->state; + else + state = 0; + if (!state) + mm (ec); + _ec->tag = _tag.prev; + } + if (state) + __builtin_exit(0); + } + jj(); +} diff --git a/gcc/tree-ssa-phiopt.cc b/gcc/tree-ssa-phiopt.cc index 558d5b4b57d..3ef7d6b28fc 100644 --- a/gcc/tree-ssa-phiopt.cc +++ b/gcc/tree-ssa-phiopt.cc @@ -3293,7 +3293,9 @@ cond_if_else_store_replacement_1 (basic_block then_bb, basic_block else_bb, || else_assign == NULL || !gimple_assign_single_p (else_assign) || gimple_clobber_p (else_assign) - || gimple_has_volatile_ops (else_assign)) + || gimple_has_volatile_ops (else_assign) + || stmt_references_abnormal_ssa_name (then_assign) + || stmt_references_abnormal_ssa_name (else_assign)) return false; lhs = gimple_assign_lhs (then_assign); diff --git a/gcc/tree-ssa-sink.cc b/gcc/tree-ssa-sink.cc index 8ce4403ddc8..f747f8adc34 100644 --- a/gcc/tree-ssa-sink.cc +++ b/gcc/tree-ssa-sink.cc @@ -35,6 +35,7 @@ along with GCC; see the file COPYING3. If not see #include "tree-cfg.h" #include "cfgloop.h" #include "tree-eh.h" +#include "tree-dfa.h" /* TODO: 1. Sinking store only using scalar promotion (IE without moving the RHS): @@ -523,7 +524,8 @@ sink_common_stores_to_bb (basic_block bb) gimple *def = SSA_NAME_DEF_STMT (arg); if (! is_gimple_assign (def) || stmt_can_throw_internal (cfun, def) - || (gimple_phi_arg_edge (phi, i)->flags & EDGE_ABNORMAL)) + || (gimple_phi_arg_edge (phi, i)->flags & EDGE_ABNORMAL) + || stmt_references_abnormal_ssa_name (def)) { /* ??? We could handle some cascading with the def being another PHI. We'd have to insert multiple PHIs for commit 5509544dad81adc21c2e6d9782b0ef1390aa4f7c Author: Richard Biener Date: Thu Dec 5 10:47:13 2024 +0100 tree-optimization/117912 - bogus address equivalences for __builtin_object_size VN again is the culprit for exploiting address equivalences before __builtin_object_size got the chance to do its job. This time it isn't about union members but adjacent structure fields where an address to one after the last element of an array field can spill over to the next field. The following protects all out-of-bound accesses on the upper bound side (singling out TYPE_MAX_VALUE + 1 is more expensive). It ignores other out-of-bound addresses that would invoke UB. Zero-sized arrays are a bit awkward because the C++ represents them with a -1U upper bound. There's a similar issue for zero-sized components whose address can be the same as the adjacent field in C. PR tree-optimization/117912 * tree-ssa-sccvn.cc (copy_reference_ops_from_ref): For addresses of zero-sized components do not set ->off if the object size pass didn't run. For OOB ARRAY_REF accesses in address expressions avoid setting ->off if the object size pass didn't run. (valueize_refs_1): Likewise. * c-c++-common/torture/pr117912-1.c: New testcase. * c-c++-common/torture/pr117912-2.c: Likewise. * c-c++-common/torture/pr117912-3.c: Likewise. (cherry picked from commit 233972ab3b5338d7a5d1d7af9108c1f366170e44) diff --git a/gcc/testsuite/c-c++-common/torture/pr117912-1.c b/gcc/testsuite/c-c++-common/torture/pr117912-1.c new file mode 100644 index 00000000000..2750585c7f7 --- /dev/null +++ b/gcc/testsuite/c-c++-common/torture/pr117912-1.c @@ -0,0 +1,28 @@ +/* { dg-do run } */ + +struct S { int a; int b[24]; int c[24]; int d; }; +volatile int *p; + +void __attribute__((noipa)) +bar (int *q) +{ + p = q; +} + +__SIZE_TYPE__ __attribute__((noipa)) +foo (struct S *p) +{ + bar (&p->b[24]); + bar (&p->c[0]); + return __builtin_object_size (&p->c[0], 1); +} + +int +main() +{ + struct S s; + __SIZE_TYPE__ x = foo (&s); + if (x < sizeof (int) * 24) + __builtin_abort (); + return 0; +} diff --git a/gcc/testsuite/c-c++-common/torture/pr117912-2.c b/gcc/testsuite/c-c++-common/torture/pr117912-2.c new file mode 100644 index 00000000000..a3a62157563 --- /dev/null +++ b/gcc/testsuite/c-c++-common/torture/pr117912-2.c @@ -0,0 +1,28 @@ +/* { dg-do run } */ + +struct S { int a; int b[0]; int c[24]; int d; }; +volatile int *p; + +void __attribute__((noipa)) +bar (int *q) +{ + p = q; +} + +__SIZE_TYPE__ __attribute__((noipa)) +foo (struct S *p) +{ + bar (&p->b[0]); + bar (&p->c[0]); + return __builtin_object_size (&p->c[0], 1); +} + +int +main() +{ + struct S s; + __SIZE_TYPE__ x = foo (&s); + if (x < sizeof (int) * 24) + __builtin_abort (); + return 0; +} diff --git a/gcc/testsuite/c-c++-common/torture/pr117912-3.c b/gcc/testsuite/c-c++-common/torture/pr117912-3.c new file mode 100644 index 00000000000..64e981d2a5e --- /dev/null +++ b/gcc/testsuite/c-c++-common/torture/pr117912-3.c @@ -0,0 +1,61 @@ +/* { dg-do run } */ +/* { dg-additional-options "-std=gnu++20" { target c++ } } */ + +struct B {}; +struct A { int a; +#ifdef __cplusplus + [[no_unique_address]] +#endif + struct B b; + char c[]; }; +volatile void *p; + +void __attribute__((noipa)) +bar (void *q) +{ + p = q; +} + +__SIZE_TYPE__ __attribute__((noipa)) +foo (struct A *p) +{ + bar (&p->b); + bar (&p->c); + return __builtin_object_size (&p->c, 1); +} + +__SIZE_TYPE__ __attribute__((noipa)) +baz (void) +{ + struct A *p = (struct A *) __builtin_malloc (__builtin_offsetof (struct A, c) + 64); + bar (&p->b); + bar (&p->c); + return __builtin_object_size (&p->c, 1); +} + +__SIZE_TYPE__ __attribute__((noipa)) +qux (struct A *p) +{ + bar (&p->b); + bar (&p->c); + return __builtin_object_size (&p->c, 3); +} + +__SIZE_TYPE__ __attribute__((noipa)) +boo (void) +{ + struct A *p = (struct A *) __builtin_malloc (__builtin_offsetof (struct A, c) + 64); + bar (&p->b); + bar (&p->c); + return __builtin_object_size (&p->c, 3); +} + +int +main () +{ + static struct A a = { .a = 1, .b = {}, .c = { 1, 2, 3, 4, 0 } }; + if (foo (&a) < 5) + __builtin_abort (); + if (baz () < 64) + __builtin_abort (); +} diff --git a/gcc/tree-ssa-sccvn.cc b/gcc/tree-ssa-sccvn.cc index cda03e0aa68..60a4826e580 100644 --- a/gcc/tree-ssa-sccvn.cc +++ b/gcc/tree-ssa-sccvn.cc @@ -943,11 +943,14 @@ copy_reference_ops_from_ref (tree ref, vec *result) poly_offset_int off = (wi::to_poly_offset (this_offset) + (wi::to_offset (bit_offset) >> LOG2_BITS_PER_UNIT)); - /* Probibit value-numbering zero offset components + /* Prohibit value-numbering zero offset components of addresses the same before the pass folding - __builtin_object_size had a chance to run. */ + __builtin_object_size had a chance to run. Likewise + for components of zero size at arbitrary offset. */ if (TREE_CODE (orig) != ADDR_EXPR - || maybe_ne (off, 0) + || (TYPE_SIZE (temp.type) + && integer_nonzerop (TYPE_SIZE (temp.type)) + && maybe_ne (off, 0)) || (cfun->curr_properties & PROP_objsz)) off.to_shwi (&temp.off); } @@ -968,9 +971,31 @@ copy_reference_ops_from_ref (tree ref, vec *result) if (! temp.op2) temp.op2 = size_binop (EXACT_DIV_EXPR, TYPE_SIZE_UNIT (eltype), size_int (TYPE_ALIGN_UNIT (eltype))); + /* Prohibit value-numbering addresses of one-after-the-last + element ARRAY_REFs the same as addresses of other components + before the pass folding __builtin_object_size had a chance + to run. */ + bool avoid_oob = true; + if (TREE_CODE (orig) != ADDR_EXPR + || cfun->curr_properties & PROP_objsz) + avoid_oob = false; + else if (poly_int_tree_p (temp.op0)) + { + tree ub = array_ref_up_bound (ref); + if (ub + && poly_int_tree_p (ub) + /* ??? The C frontend for T[0] uses [0:] and the + C++ frontend [0:-1U]. See layout_type for how + awkward this is. */ + && !integer_minus_onep (ub) + && known_le (wi::to_poly_offset (temp.op0), + wi::to_poly_offset (ub))) + avoid_oob = false; + } if (poly_int_tree_p (temp.op0) && poly_int_tree_p (temp.op1) - && TREE_CODE (temp.op2) == INTEGER_CST) + && TREE_CODE (temp.op2) == INTEGER_CST + && !avoid_oob) { poly_offset_int off = ((wi::to_poly_offset (temp.op0) - wi::to_poly_offset (temp.op1)) @@ -1706,6 +1731,24 @@ re_valueize: && poly_int_tree_p (vro->op1) && TREE_CODE (vro->op2) == INTEGER_CST) { + /* Prohibit value-numbering addresses of one-after-the-last + element ARRAY_REFs the same as addresses of other components + before the pass folding __builtin_object_size had a chance + to run. */ + if (!(cfun->curr_properties & PROP_objsz) + && (*orig)[0].opcode == ADDR_EXPR) + { + tree dom = TYPE_DOMAIN ((*orig)[i + 1].type); + if (!dom + || !TYPE_MAX_VALUE (dom) + || !poly_int_tree_p (TYPE_MAX_VALUE (dom)) + || integer_minus_onep (TYPE_MAX_VALUE (dom))) + continue; + if (!known_le (wi::to_poly_offset (vro->op0), + wi::to_poly_offset (TYPE_MAX_VALUE (dom)))) + continue; + } + poly_offset_int off = ((wi::to_poly_offset (vro->op0) - wi::to_poly_offset (vro->op1)) * wi::to_offset (vro->op2) commit 8eeaaec278275db2be81792c833713f64c9d20eb Author: Richard Biener Date: Fri Nov 15 11:56:14 2024 +0100 tree-optimization/117574 - bougs niter lt-to-ne When trying to change a IV from IV0 < IV1 to IV0' != IV1' we apply fancy adjustments to the may_be_zero condition we compute rather than using the obvious IV0->base >= IV1->base expression (to be able to use > instead of >=?). This doesn't seem to go well. PR tree-optimization/117574 * tree-ssa-loop-niter.cc (number_of_iterations_lt_to_ne): Use the obvious may_be_zero condition. * gcc.dg/torture/pr117574-1.c: New testcase. (cherry picked from commit ff5a14abeb31cd6bd0ca55e7043d05c8141a8c7f) diff --git a/gcc/testsuite/gcc.dg/torture/pr117574-1.c b/gcc/testsuite/gcc.dg/torture/pr117574-1.c new file mode 100644 index 00000000000..2e99cec13b6 --- /dev/null +++ b/gcc/testsuite/gcc.dg/torture/pr117574-1.c @@ -0,0 +1,20 @@ +/* { dg-do run } */ + +void abort (void); +int a, c; +long b; +short d; +static long e(long f, long h, long i) { + for (long g = f; g <= h; g += i) + b += g; + return b; +} +int main() { + c = 1; + for (; c >= 0; c--) + ; + for (; e(d + 40, d + 76, c + 51) < 4;) + ; + if (a != 0) + abort (); +} diff --git a/gcc/tree-ssa-loop-niter.cc b/gcc/tree-ssa-loop-niter.cc index 0353ffd3022..77798a0fd48 100644 --- a/gcc/tree-ssa-loop-niter.cc +++ b/gcc/tree-ssa-loop-niter.cc @@ -1214,17 +1214,6 @@ number_of_iterations_lt_to_ne (tree type, affine_iv *iv0, affine_iv *iv1, if (integer_zerop (assumption)) goto end; } - if (mpz_cmp (mmod, bnds->below) < 0) - noloop = boolean_false_node; - else if (POINTER_TYPE_P (type)) - noloop = fold_build2 (GT_EXPR, boolean_type_node, - iv0->base, - fold_build_pointer_plus (iv1->base, tmod)); - else - noloop = fold_build2 (GT_EXPR, boolean_type_node, - iv0->base, - fold_build2 (PLUS_EXPR, type1, - iv1->base, tmod)); } else { @@ -1240,21 +1229,15 @@ number_of_iterations_lt_to_ne (tree type, affine_iv *iv0, affine_iv *iv1, if (integer_zerop (assumption)) goto end; } - if (mpz_cmp (mmod, bnds->below) < 0) - noloop = boolean_false_node; - else if (POINTER_TYPE_P (type)) - noloop = fold_build2 (GT_EXPR, boolean_type_node, - fold_build_pointer_plus (iv0->base, - fold_build1 (NEGATE_EXPR, - type1, tmod)), - iv1->base); - else - noloop = fold_build2 (GT_EXPR, boolean_type_node, - fold_build2 (MINUS_EXPR, type1, - iv0->base, tmod), - iv1->base); } + /* IV0 < IV1 does not loop if IV0->base >= IV1->base. */ + if (mpz_cmp (mmod, bnds->below) < 0) + noloop = boolean_false_node; + else + noloop = fold_build2 (GE_EXPR, boolean_type_node, + iv0->base, iv1->base); + if (!integer_nonzerop (assumption)) niter->assumptions = fold_build2 (TRUTH_AND_EXPR, boolean_type_node, niter->assumptions, commit c5e4e7617e6e6453e6b6995fb2e31bfe1dd28fdf Author: Richard Biener Date: Mon Jan 13 09:12:23 2025 +0100 tree-optimization/117119 - ICE with int128 IV in dataref analysis Here's another fix for a missing check that an IV value fits in a HIW. It's originally from Stefan. PR tree-optimization/117119 * tree-data-ref.cc (initialize_matrix_A): Check whether an INTEGER_CST fits in HWI, otherwise return chrec_dont_know. * gcc.dg/torture/pr117119.c: New testcase. Co-Authored-By: Stefan Schulze Frielinghaus (cherry picked from commit d3904a3ad9d7b4c8e5e536e5166b89548510fd48) diff --git a/gcc/testsuite/gcc.dg/torture/pr117119.c b/gcc/testsuite/gcc.dg/torture/pr117119.c new file mode 100644 index 00000000000..0ec4ac1b180 --- /dev/null +++ b/gcc/testsuite/gcc.dg/torture/pr117119.c @@ -0,0 +1,10 @@ +/* { dg-do compile } */ +/* { dg-require-effective-target int128 } */ + +unsigned __int128 g_728; +int func_1_l_5011[8]; +void func_1() { + for (;; g_728 += 1) + func_1_l_5011[g_728] ^= func_1_l_5011[g_728 + 5]; +} +void main() {} diff --git a/gcc/tree-data-ref.cc b/gcc/tree-data-ref.cc index ea55f81a92f..996f8217e28 100644 --- a/gcc/tree-data-ref.cc +++ b/gcc/tree-data-ref.cc @@ -4032,7 +4032,7 @@ initialize_matrix_A (lambda_matrix A, tree chrec, unsigned index, int mult) } case INTEGER_CST: - return chrec; + return cst_and_fits_in_hwi (chrec) ? chrec : chrec_dont_know; default: gcc_unreachable (); commit c62cbe7164765e03a7bc41377452b80b6237d93d Author: Richard Biener Date: Thu Sep 26 15:41:59 2024 +0200 tree-optimization/116850 - corrupt post-dom info Path isolation computes post-dominators on demand but can end up splitting blocks after that, wrecking it. We can delay splitting of blocks until we no longer need the post-dom info which is what the following patch does to solve the issue. PR tree-optimization/116850 * gimple-ssa-isolate-paths.cc (bb_split_points): New global. (insert_trap): Delay BB splitting if post-doms are computed. (find_explicit_erroneous_behavior): Process delayed BB splitting after releasing post dominators. (gimple_ssa_isolate_erroneous_paths): Do not free post-dom info here. * gcc.dg/pr116850.c: New testcase. (cherry picked from commit 64163657ba7e70347087a63bb2b32d83b52ea7d9) diff --git a/gcc/gimple-ssa-isolate-paths.cc b/gcc/gimple-ssa-isolate-paths.cc index cc0ed9760a0..7613f83da6a 100644 --- a/gcc/gimple-ssa-isolate-paths.cc +++ b/gcc/gimple-ssa-isolate-paths.cc @@ -62,6 +62,8 @@ check_loadstore (gimple *stmt, tree op, tree, void *data) return false; } +static vec *bb_split_points; + /* Insert a trap after SI and split the block after the trap. */ static void @@ -104,14 +106,20 @@ insert_trap (gimple_stmt_iterator *si_p, tree op) gsi_insert_after (si_p, seq, GSI_NEW_STMT); if (stmt_ends_bb_p (stmt)) { - split_block (gimple_bb (stmt), stmt); + if (dom_info_available_p (CDI_POST_DOMINATORS)) + bb_split_points->safe_push (stmt); + else + split_block (gimple_bb (stmt), stmt); return; } } else gsi_insert_before (si_p, seq, GSI_NEW_STMT); - split_block (gimple_bb (new_stmt), new_stmt); + if (dom_info_available_p (CDI_POST_DOMINATORS)) + bb_split_points->safe_push (new_stmt); + else + split_block (gimple_bb (new_stmt), new_stmt); *si_p = gsi_for_stmt (stmt); } @@ -840,6 +848,8 @@ static void find_explicit_erroneous_behavior (void) { basic_block bb; + auto_vec local_bb_split_points; + bb_split_points = &local_bb_split_points; FOR_EACH_BB_FN (bb, cfun) { @@ -881,6 +891,14 @@ find_explicit_erroneous_behavior (void) warn_return_addr_local (bb, return_stmt); } } + + free_dominance_info (CDI_POST_DOMINATORS); + + /* Perform delayed splitting of blocks. */ + for (gimple *stmt : local_bb_split_points) + split_block (gimple_bb (stmt), stmt); + + bb_split_points = NULL; } /* Search the function for statements which, if executed, would cause @@ -937,7 +955,6 @@ gimple_ssa_isolate_erroneous_paths (void) /* We scramble the CFG and loop structures a bit, clean up appropriately. We really should incrementally update the loop structures, in theory it shouldn't be that hard. */ - free_dominance_info (CDI_POST_DOMINATORS); if (cfg_altered) { free_dominance_info (CDI_DOMINATORS); diff --git a/gcc/testsuite/gcc.dg/pr116850.c b/gcc/testsuite/gcc.dg/pr116850.c new file mode 100644 index 00000000000..7ab5da1848b --- /dev/null +++ b/gcc/testsuite/gcc.dg/pr116850.c @@ -0,0 +1,12 @@ +/* { dg-do compile } */ +/* { dg-options "-Os -w" } */ + +int a, b; +int *c() +{ + int d, *e = 0, *f = &d, *g = &a; + if (b) + g = 0; + *e = *g; + return f; +} commit f1989b50a9141b5bf62275d92b2fa28e9d82ca8b Author: Richard Biener Date: Sun Oct 13 11:42:27 2024 +0200 tree-optimization/116481 - avoid building function_type[] The following avoids building an array type with function or method element type during diagnosing an array bound violation as this will result in an error, rejecting a program with a not too useful error message. Instead build such array type manually. PR tree-optimization/116481 * pointer-query.cc (build_printable_array_type): Build an array types with function or method element type manually to avoid bogus diagnostic. * gcc.dg/pr116481.c: New testcase. (cherry picked from commit 1506027347776a2f6ec5b92d56ef192e85944e2e) diff --git a/gcc/pointer-query.cc b/gcc/pointer-query.cc index 9a12bb27c34..fb63546d563 100644 --- a/gcc/pointer-query.cc +++ b/gcc/pointer-query.cc @@ -2574,6 +2574,17 @@ array_elt_at_offset (tree artype, HOST_WIDE_INT off, tree build_printable_array_type (tree eltype, unsigned HOST_WIDE_INT nelts) { + /* Cannot build an array type of functions or methods without + an error diagnostic. */ + if (FUNC_OR_METHOD_TYPE_P (eltype)) + { + tree arrtype = make_node (ARRAY_TYPE); + TREE_TYPE (arrtype) = eltype; + TYPE_SIZE (arrtype) = bitsize_zero_node; + TYPE_SIZE_UNIT (arrtype) = size_zero_node; + return arrtype; + } + if (TYPE_SIZE_UNIT (eltype) && TREE_CODE (TYPE_SIZE_UNIT (eltype)) == INTEGER_CST && !integer_zerop (TYPE_SIZE_UNIT (eltype)) diff --git a/gcc/testsuite/gcc.dg/pr116481.c b/gcc/testsuite/gcc.dg/pr116481.c new file mode 100644 index 00000000000..3ee6d747087 --- /dev/null +++ b/gcc/testsuite/gcc.dg/pr116481.c @@ -0,0 +1,13 @@ +/* { dg-do compile } */ +/* { dg-options "-O2 -Warray-bounds" } */ + +extern void tramp (); + +int is_trampoline (void* function) /* { dg-bogus "arrays of functions are not meaningful" } */ +{ + void* tramp_address = tramp; + if (!(((unsigned long)function & 3) == 2)) + return 0; + return (((long *) ((char*)function - 2))[0] + == ((long *) ((char*)tramp_address-2))[0]); /* { dg-warning "outside array bounds" } */ +} commit b444c3061a14d12d164d82345a1836f5fbd21939 Author: Richard Biener Date: Thu Jul 18 13:35:33 2024 +0200 middle-end/115641 - invalid address construction fold_truth_andor_1 via make_bit_field_ref builds an address of a CALL_EXPR which isn't valid GENERIC and later causes an ICE. The following simply avoids the folding for f ().a != 1 || f ().b != 2 as it is a premature optimization anyway. The alternative would have been to build a TARGET_EXPR around the call. To get this far f () has to be const as otherwise the two calls are not semantically equivalent for the optimization. PR middle-end/115641 * fold-const.cc (decode_field_reference): If the inner reference isn't something we can take the address of, fail. * gcc.dg/torture/pr115641.c: New testcase. (cherry picked from commit 3670c70c561656a19f6bff36dd229f18120af127) diff --git a/gcc/fold-const.cc b/gcc/fold-const.cc index 391f1109540..4e88aa4ef52 100644 --- a/gcc/fold-const.cc +++ b/gcc/fold-const.cc @@ -4767,6 +4767,9 @@ decode_field_reference (location_t loc, tree *exp_, HOST_WIDE_INT *pbitsize, || *pbitsize < 0 || offset != 0 || TREE_CODE (inner) == PLACEHOLDER_EXPR + /* We eventually want to build a larger reference and need to take + the address of this. */ + || (!REFERENCE_CLASS_P (inner) && !DECL_P (inner)) /* Reject out-of-bound accesses (PR79731). */ || (! AGGREGATE_TYPE_P (TREE_TYPE (inner)) && compare_tree_int (TYPE_SIZE (TREE_TYPE (inner)), diff --git a/gcc/testsuite/gcc.dg/torture/pr115641.c b/gcc/testsuite/gcc.dg/torture/pr115641.c new file mode 100644 index 00000000000..65fb09ca64f --- /dev/null +++ b/gcc/testsuite/gcc.dg/torture/pr115641.c @@ -0,0 +1,29 @@ +/* { dg-do run } */ + +typedef struct { + char hours, day, month; + short year; +} T; + +T g (void) +{ + T now; + now.hours = 1; + now.day = 2; + now.month = 3; + now.year = 4; + return now; +} + +__attribute__((const)) T f (void) +{ + T virk = g (); + return virk; +} + +int main () +{ + if (f ().hours != 1 || f ().day != 2 || f ().month != 3 || f ().year != 4) + __builtin_abort (); + return 0; +} commit bfb1b0f8bb9fe76071fb02840002a17ed00a8999 Author: Richard Biener Date: Wed Mar 6 09:02:31 2024 +0100 tree-optimization/114246 - invalid call argument from DSE The following makes sure to strip type conversions added by build_fold_addr_expr before placing the result in a call argument. PR tree-optimization/114246 * tree-ssa-dse.cc (increment_start_addr): Strip useless type conversions from the adjusted address. * gcc.dg/torture/pr114246.c: New testcase. (cherry picked from commit 0249744a9fe0775c2c895727aeebec4c59fd5f95) diff --git a/gcc/testsuite/gcc.dg/torture/pr114246.c b/gcc/testsuite/gcc.dg/torture/pr114246.c new file mode 100644 index 00000000000..eb20db594cd --- /dev/null +++ b/gcc/testsuite/gcc.dg/torture/pr114246.c @@ -0,0 +1,11 @@ +/* { dg-do compile } */ +/* { dg-additional-options "-w" } */ + +int a, b; + +void +foo (void) +{ + __builtin_memcpy (&a, (char *)&b - 1, 2); + __builtin_memcpy (&a, &b, 1); +} diff --git a/gcc/tree-ssa-dse.cc b/gcc/tree-ssa-dse.cc index c373dacd806..749756d8197 100644 --- a/gcc/tree-ssa-dse.cc +++ b/gcc/tree-ssa-dse.cc @@ -45,6 +45,7 @@ along with GCC; see the file COPYING3. If not see #include "ipa-modref.h" #include "target.h" #include "tree-ssa-loop-niter.h" +#include "tree-ssa.h" /* This file implements dead store elimination. @@ -614,6 +615,7 @@ increment_start_addr (gimple *stmt, tree *where, int increment) *where, build_int_cst (ptr_type_node, increment))); + STRIP_USELESS_TYPE_CONVERSION (*where); } /* STMT is builtin call that writes bytes in bitmap ORIG, some bytes are dead commit b33dad2d3ef8c18c17a1c01879c725d04d8b5e82 Author: GCC Administrator Date: Sat Apr 5 00:19:15 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 6a43292b7ad..0d267a8805b 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,86 @@ +2025-04-04 Richard Biener + + Backported from master: + 2024-03-06 Richard Biener + + PR tree-optimization/114246 + * tree-ssa-dse.cc (increment_start_addr): Strip useless + type conversions from the adjusted address. + +2025-04-04 Richard Biener + + Backported from master: + 2024-07-18 Richard Biener + + PR middle-end/115641 + * fold-const.cc (decode_field_reference): If the inner + reference isn't something we can take the address of, fail. + +2025-04-04 Richard Biener + + Backported from master: + 2024-10-13 Richard Biener + + PR tree-optimization/116481 + * pointer-query.cc (build_printable_array_type): + Build an array types with function or method element type + manually to avoid bogus diagnostic. + +2025-04-04 Richard Biener + + Backported from master: + 2024-09-30 Richard Biener + + PR tree-optimization/116850 + * gimple-ssa-isolate-paths.cc (bb_split_points): New global. + (insert_trap): Delay BB splitting if post-doms are computed. + (find_explicit_erroneous_behavior): Process delayed BB + splitting after releasing post dominators. + (gimple_ssa_isolate_erroneous_paths): Do not free post-dom + info here. + +2025-04-04 Richard Biener + + Backported from master: + 2025-01-13 Richard Biener + Stefan Schulze Frielinghaus + + PR tree-optimization/117119 + * tree-data-ref.cc (initialize_matrix_A): Check whether + an INTEGER_CST fits in HWI, otherwise return chrec_dont_know. + +2025-04-04 Richard Biener + + Backported from master: + 2024-11-20 Richard Biener + + PR tree-optimization/117574 + * tree-ssa-loop-niter.cc (number_of_iterations_lt_to_ne): + Use the obvious may_be_zero condition. + +2025-04-04 Richard Biener + + Backported from master: + 2024-12-10 Richard Biener + + PR tree-optimization/117912 + * tree-ssa-sccvn.cc (copy_reference_ops_from_ref): For addresses + of zero-sized components do not set ->off if the object size pass + didn't run. + For OOB ARRAY_REF accesses in address expressions avoid setting + ->off if the object size pass didn't run. + (valueize_refs_1): Likewise. + +2025-04-04 Richard Biener + + Backported from master: + 2025-02-03 Richard Biener + + PR tree-optimization/118717 + * tree-ssa-phiopt.cc (cond_if_else_store_replacement_1): + Do not common stores referencing abnormal SSA names. + * tree-ssa-sink.cc (sink_common_stores_to_bb): Likewise. + 2025-04-01 Andrew Carlotti PR target/119372 diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c36e5b25617..d5996ab4116 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250404 +20250405 diff --git a/gcc/ada/ChangeLog b/gcc/ada/ChangeLog index b0085cf918b..2c1842a01fb 100644 --- a/gcc/ada/ChangeLog +++ b/gcc/ada/ChangeLog @@ -1,3 +1,15 @@ +2025-04-04 Eric Botcazou + + * libgnat/a-ngcoar.adb (Eigensystem): Adjust notation and fix the + layout of the real symmetric matrix in the main comment. Adjust + the layout of the associated code accordingly and correctly turn + the 2Nx1 real vectors into Nx1 complex ones. + (Eigenvalues): Minor similar tweaks. + * libgnat/a-ngrear.adb (Jacobi): Minor tweaks in the main comment. + Adjust notation and corresponding parameter names of functions. + Fix call to Unit_Matrix routine. Adjust the comment describing + the various kinds of iterations to match the implementation. + 2025-01-05 Estevan Castilho (Tevo) * libgnarl/s-taprop__dummy.adb: Remove use clause for diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 2cc52c8d6d8..338ee3a1ddb 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,70 @@ +2025-04-04 Richard Biener + + Backported from master: + 2024-03-06 Richard Biener + + PR tree-optimization/114246 + * gcc.dg/torture/pr114246.c: New testcase. + +2025-04-04 Richard Biener + + Backported from master: + 2024-07-18 Richard Biener + + PR middle-end/115641 + * gcc.dg/torture/pr115641.c: New testcase. + +2025-04-04 Richard Biener + + Backported from master: + 2024-10-13 Richard Biener + + PR tree-optimization/116481 + * gcc.dg/pr116481.c: New testcase. + +2025-04-04 Richard Biener + + Backported from master: + 2024-09-30 Richard Biener + + PR tree-optimization/116850 + * gcc.dg/pr116850.c: New testcase. + +2025-04-04 Richard Biener + + Backported from master: + 2025-01-13 Richard Biener + Stefan Schulze Frielinghaus + + PR tree-optimization/117119 + * gcc.dg/torture/pr117119.c: New testcase. + +2025-04-04 Richard Biener + + Backported from master: + 2024-11-20 Richard Biener + + PR tree-optimization/117574 + * gcc.dg/torture/pr117574-1.c: New testcase. + +2025-04-04 Richard Biener + + Backported from master: + 2024-12-10 Richard Biener + + PR tree-optimization/117912 + * c-c++-common/torture/pr117912-1.c: New testcase. + * c-c++-common/torture/pr117912-2.c: Likewise. + * c-c++-common/torture/pr117912-3.c: Likewise. + +2025-04-04 Richard Biener + + Backported from master: + 2025-02-03 Richard Biener + + PR tree-optimization/118717 + * gcc.dg/torture/pr118717.c: New testcase. + 2025-03-11 Iain Buclaw Backported from master: commit f9d954f937c19ea538be49e9984d69137a0a8e95 Author: GCC Administrator Date: Sun Apr 6 00:18:50 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d5996ab4116..b362d0af100 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250405 +20250406 commit 0097de8f0eca90f906e57774127073ba4ea057d3 Author: GCC Administrator Date: Mon Apr 7 00:19:02 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index b362d0af100..64b49caa89a 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250406 +20250407 commit 82bca9ad01b82f1051edf669f897be61a8c60123 Author: GCC Administrator Date: Tue Apr 8 00:20:16 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 64b49caa89a..fd8dd2cde24 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250407 +20250408 commit 1a722dc7881eee6fb021fa033a290a1a24a3d3f6 Author: GCC Administrator Date: Wed Apr 9 00:19:41 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index fd8dd2cde24..cb38c2c30be 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250408 +20250409 commit 21d644b871f4cfd015356b11aa172d2cd3c4dcb4 Author: GCC Administrator Date: Thu Apr 10 00:19:05 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index cb38c2c30be..4013552f78d 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250409 +20250410 commit e616244b488b70049fbf92886799f6726258c128 Author: GCC Administrator Date: Fri Apr 11 00:18:50 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 4013552f78d..534b600c823 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250410 +20250411 commit bdd49524fe47b8c163d3d0e2f7acb0bce0a0b7e3 Author: GCC Administrator Date: Sat Apr 12 00:18:48 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 534b600c823..d986e55ceef 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250411 +20250412 commit 00210d187018a40c2fddcf46fcc9669d6e8d7c53 Author: GCC Administrator Date: Sun Apr 13 00:18:50 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index d986e55ceef..2061764a097 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250412 +20250413 commit 3232e151654f3998926c8e82a2ffd911c6e74df4 Author: Vladimir N. Makarov Date: Wed Feb 5 14:23:23 2025 -0500 [PR115568][LRA]: Use more strict output reload check in rematerialization In this PR case LRA rematerialized a value from inheritance insn instead of output reload one. This resulted in considering a rematerilization candidate value available when it was actually not. As a consequence an insn after rematerliazation used the unexpected value and this use resulted in fp exception. The patch fixes this bug. gcc/ChangeLog: PR rtl-optimization/115568 * lra-remat.cc (create_cands): Check that output reload insn is adjacent to given insn. Update a comment. gcc/testsuite/ChangeLog: PR rtl-optimization/115568 * gcc.target/i386/pr115568.c: New. (cherry picked from commit 98545441308c2ae4d535f14b108ad6551fd927d5) diff --git a/gcc/lra-remat.cc b/gcc/lra-remat.cc index 5079eb50f28..f565fe1d939 100644 --- a/gcc/lra-remat.cc +++ b/gcc/lra-remat.cc @@ -459,7 +459,8 @@ create_cands (void) if (insn2 != NULL && dst_regno >= FIRST_PSEUDO_REGISTER && reg_renumber[dst_regno] < 0 - && BLOCK_FOR_INSN (insn2) == BLOCK_FOR_INSN (insn)) + && BLOCK_FOR_INSN (insn2) == BLOCK_FOR_INSN (insn) + && insn2 == prev_nonnote_insn (insn)) { create_cand (insn2, regno_potential_cand[src_regno].nop, dst_regno, insn); @@ -473,9 +474,10 @@ create_cands (void) gcc_assert (REG_P (*id->operand_loc[nop])); int regno = REGNO (*id->operand_loc[nop]); gcc_assert (regno >= FIRST_PSEUDO_REGISTER); - /* If we're setting an unrenumbered pseudo, make a candidate immediately. - If it's an output reload register, save it for later; the code above - looks for output reload insns later on. */ + /* If we're setting an unrenumbered pseudo, make a candidate + immediately. If it's a potential output reload register, save + it for later; the code above looks for output reload insns later + on. */ if (reg_renumber[regno] < 0) create_cand (insn, nop, regno); else if (regno >= lra_constraint_new_regno_start) diff --git a/gcc/testsuite/gcc.target/i386/pr115568.c b/gcc/testsuite/gcc.target/i386/pr115568.c new file mode 100644 index 00000000000..cedc7ac3843 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr115568.c @@ -0,0 +1,38 @@ +/* { dg-do run } */ +/* { dg-options "-O2 -fno-tree-sink -fno-tree-ter -fschedule-insns" } */ + +int a, c, d = 1, e, f = 1, h, i, j; +unsigned b = 1, g; +int main() { + for (; h < 2; h++) { + int k = ~(b || 0), l = ((~e - j) ^ a % b) % k, m = (b ^ -1) + e; + unsigned o = ~a % ~1; + if (f) { + l = d; + m = 10; + i = e; + d = -(~e + b); + g = o % m; + e = -1; + n: + a = a % ~i; + b = ~k; + if (!g) { + b = e + o % -1; + continue; + } + if (!l) + break; + } + int q = (~d + g) << ~e, p = (~d - q) & a >> b; + unsigned s = ~((g & e) + (p | (b ^ (d + k)))); + int r = (e & s) + p, u = d | ~a, + t = ((~(q + (~a + (s + e)))) & u) | (-g & (c << d ^ p)); + if (t) + if (!r) + goto n; + g = m; + e = i; + } + return 0; +} commit 6675cf3abd09731ec8360ba8ac8928b63b33b7bb Author: Richard Biener Date: Wed Apr 9 14:36:19 2025 +0200 rtl-optimization/119689 - compare-debug failure with LRA The previous change to fix LRA rematerialization broke compare-debug for i586 bootstrap. Fixed by using prev_nonnote_nondebug_insn instead of prev_nonnote_insn. PR rtl-optimization/119689 PR rtl-optimization/115568 * lra-remat.cc (create_cands): Use prev_nonnote_nondebug_insn to check whether insn2 is directly before insn. * g++.target/i386/pr119689.C: New testcase. (cherry picked from commit 088887de7717a22b1503760e9b79dfbe22a0f428) diff --git a/gcc/lra-remat.cc b/gcc/lra-remat.cc index f565fe1d939..2545fef3f9a 100644 --- a/gcc/lra-remat.cc +++ b/gcc/lra-remat.cc @@ -460,7 +460,7 @@ create_cands (void) && dst_regno >= FIRST_PSEUDO_REGISTER && reg_renumber[dst_regno] < 0 && BLOCK_FOR_INSN (insn2) == BLOCK_FOR_INSN (insn) - && insn2 == prev_nonnote_insn (insn)) + && insn2 == prev_nonnote_nondebug_insn (insn)) { create_cand (insn2, regno_potential_cand[src_regno].nop, dst_regno, insn); diff --git a/gcc/testsuite/g++.target/i386/pr119689.C b/gcc/testsuite/g++.target/i386/pr119689.C new file mode 100644 index 00000000000..cdc6d2dade5 --- /dev/null +++ b/gcc/testsuite/g++.target/i386/pr119689.C @@ -0,0 +1,44 @@ +// { dg-do compile } +// { dg-options "-O2 -fcompare-debug" } +// { dg-additional-options "-march=i586 -mtune=generic" { target ia32 } } +// { dg-additional-options "-fPIC" { target { fpic } } } + +enum gimple_code { GIMPLE_ASSIGN, GIMPLE_RETURN }; +bool is_gimple_call(); +int m_sig, m_exp, sreal_new_exp; +struct sreal { + sreal(long long sig) { + long long __trans_tmp_6 = sig >= 0 ? sig : -(unsigned long long)sig; + sig = __trans_tmp_6 <<= sreal_new_exp -= m_exp = __trans_tmp_6; + m_sig = sig; + } + void operator/(sreal); +}; +struct ipa_predicate { + ipa_predicate(bool = true); + void operator&=(ipa_predicate); + void operator&(ipa_predicate); +}; +void add_condition(); +gimple_code eliminated_by_inlining_prob_code; +static int eliminated_by_inlining_prob() { + switch (eliminated_by_inlining_prob_code) { + case GIMPLE_RETURN: + return 2; + case GIMPLE_ASSIGN: + return 1; + } + return 0; +} +void fp_expression_p() { + ipa_predicate bb_predicate; + for (;;) { + int prob = eliminated_by_inlining_prob(); + ipa_predicate sra_predicate; + sra_predicate &= add_condition; + if (is_gimple_call()) + sreal(prob) / 2; + if (prob != 2) + bb_predicate & sra_predicate; + } +} commit ad66eca255b1e244805fd6197d140dae352185a2 Author: GCC Administrator Date: Mon Apr 14 00:19:22 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 0d267a8805b..0cdb6d268e7 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,22 @@ +2025-04-13 Richard Biener + + Backported from master: + 2025-04-09 Richard Biener + + PR rtl-optimization/119689 + PR rtl-optimization/115568 + * lra-remat.cc (create_cands): Use prev_nonnote_nondebug_insn + to check whether insn2 is directly before insn. + +2025-04-13 Vladimir N. Makarov + + Backported from master: + 2025-02-05 Vladimir N. Makarov + + PR rtl-optimization/115568 + * lra-remat.cc (create_cands): Check that output reload insn is + adjacent to given insn. Update a comment. + 2025-04-04 Richard Biener Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 2061764a097..1e2b10c9cd9 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250413 +20250414 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 338ee3a1ddb..1002a907daf 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,20 @@ +2025-04-13 Richard Biener + + Backported from master: + 2025-04-09 Richard Biener + + PR rtl-optimization/119689 + PR rtl-optimization/115568 + * g++.target/i386/pr119689.C: New testcase. + +2025-04-13 Vladimir N. Makarov + + Backported from master: + 2025-02-05 Vladimir N. Makarov + + PR rtl-optimization/115568 + * gcc.target/i386/pr115568.c: New. + 2025-04-04 Richard Biener Backported from master: commit d48c5ec8f19b71c4591ea0bb4b05a960e13fc685 Author: GCC Administrator Date: Tue Apr 15 00:21:18 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 1e2b10c9cd9..6a8ff35d9ec 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250414 +20250415 commit 31d7e0751e58ce006038fa5100a79da6ec6ddb7e Author: Jonathan Wakely Date: Fri Apr 11 11:08:34 2025 +0100 libstdc++: Document thread-safety for COW std::string [PR21334] The gcc4-compatible copy-on-write std::string does not conform to the C++11 requirements on data race avoidance in standard containers. Specifically, calling non-const member functions such as begin() and data() needs to do the "copy on write" operation and so is most definitely a modification of the object. As such, those non-const members must not be called concurrently with any other uses of the string object. libstdc++-v3/ChangeLog: PR libstdc++/21334 * doc/xml/manual/using.xml: Document that container data race avoidance rules do not apply to COW std::string. * doc/html/*: Regenerate. (cherry picked from commit dd35f66287b7cca196a720c9641e463255dceb1c) diff --git a/libstdc++-v3/doc/html/manual/using_concurrency.html b/libstdc++-v3/doc/html/manual/using_concurrency.html index e09fc5e15bc..f99cca414ec 100644 --- a/libstdc++-v3/doc/html/manual/using_concurrency.html +++ b/libstdc++-v3/doc/html/manual/using_concurrency.html @@ -126,6 +126,16 @@ gcc version 4.1.2 20070925 (Red Hat 4.1.2-33) the container the iterator refers to (for example incrementing a list iterator must access the pointers between nodes, which are part of the container and so conflict with other accesses to the container). +

+ The Copy-On-Write std::string implementation + used before GCC 5 (and with + _GLIBCXX_USE_CXX11_ABI=0) + is not a standard container and does not conform to the data race + avoidance rules described above. For the Copy-On-Write + std::string, non-const member functions such as + begin() are considered to be modifying accesses + and so must not be used concurrently with any other accesses to the + same object.

Programs which follow the rules above will not encounter data races in library code, even when using library types which share state between distinct objects. In the example below the diff --git a/libstdc++-v3/doc/xml/manual/using.xml b/libstdc++-v3/doc/xml/manual/using.xml index 1c075084a56..04aff0c8b8d 100644 --- a/libstdc++-v3/doc/xml/manual/using.xml +++ b/libstdc++-v3/doc/xml/manual/using.xml @@ -1808,6 +1808,18 @@ gcc version 4.1.2 20070925 (Red Hat 4.1.2-33) of the container and so conflict with other accesses to the container). + + The Copy-On-Write std::string implementation + used before GCC 5 (and with + _GLIBCXX_USE_CXX11_ABI=0) + is not a standard container and does not conform to the data race + avoidance rules described above. For the Copy-On-Write + std::string, non-const member functions such as + begin() are considered to be modifying accesses + and so must not be used concurrently with any other accesses to the + same object. + + Programs which follow the rules above will not encounter data races in library code, even when using library types which share state between distinct objects. In the example below the commit 4889daddae1062528bd8d3578635af20cc31fcfc Author: GCC Administrator Date: Wed Apr 16 00:20:39 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 6a8ff35d9ec..c9d404d186e 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250415 +20250416 diff --git a/libstdc++-v3/ChangeLog b/libstdc++-v3/ChangeLog index 0e0629219c9..f7da4e52bc8 100644 --- a/libstdc++-v3/ChangeLog +++ b/libstdc++-v3/ChangeLog @@ -1,3 +1,13 @@ +2025-04-15 Jonathan Wakely + + Backported from master: + 2025-04-14 Jonathan Wakely + + PR libstdc++/21334 + * doc/xml/manual/using.xml: Document that container data race + avoidance rules do not apply to COW std::string. + * doc/html/*: Regenerate. + 2025-04-02 Jonathan Wakely Backported from master: commit 5f3811fd50113571127b849325ab8f6c674158af Author: Jonathan Wakely Date: Thu Nov 14 01:14:44 2024 +0000 libstdc++: Add missing parts of LWG 3480 for directory iterators [PR117560] It looks like I only read half the resolution of LWG 3480 and decided we already supported it. As well as making the non-member overloads of end take their parameters by value, we need some specializations of the enable_borrowed_range and enable_view variable templates. libstdc++-v3/ChangeLog: PR libstdc++/117560 * include/bits/fs_dir.h (enable_borrowed_range, enable_view): Define specializations for directory iterators, as per LWG 3480. * testsuite/27_io/filesystem/iterators/lwg3480.cc: New test. (cherry picked from commit eec6e8923586b9a54e37f32cef112d26d86e8f01) diff --git a/libstdc++-v3/include/bits/fs_dir.h b/libstdc++-v3/include/bits/fs_dir.h index 968a5cf52a0..822508ae83e 100644 --- a/libstdc++-v3/include/bits/fs_dir.h +++ b/libstdc++-v3/include/bits/fs_dir.h @@ -39,6 +39,7 @@ #if __cplusplus >= 202002L # include // std::strong_ordering # include // std::default_sentinel_t +# include // enable_view, enable_borrowed_range #endif namespace std _GLIBCXX_VISIBILITY(default) @@ -626,6 +627,27 @@ _GLIBCXX_END_NAMESPACE_CXX11 extern template class __shared_ptr; +#ifdef __cpp_lib_concepts // >= C++20 +// _GLIBCXX_RESOLVE_LIB_DEFECTS +// 3480. directory_iterator and recursive_directory_iterator are not ranges +namespace ranges +{ + template<> + inline constexpr bool + enable_borrowed_range = true; + template<> + inline constexpr bool + enable_borrowed_range = true; + + template<> + inline constexpr bool + enable_view = true; + template<> + inline constexpr bool + enable_view = true; +} // namespace ranges +#endif // concepts + _GLIBCXX_END_NAMESPACE_VERSION } // namespace std diff --git a/libstdc++-v3/testsuite/27_io/filesystem/iterators/lwg3480.cc b/libstdc++-v3/testsuite/27_io/filesystem/iterators/lwg3480.cc new file mode 100644 index 00000000000..15e0286fff6 --- /dev/null +++ b/libstdc++-v3/testsuite/27_io/filesystem/iterators/lwg3480.cc @@ -0,0 +1,16 @@ +// { dg-do compile { target c++20 } } +// { dg-require-filesystem-ts "" } + +// LWG 3480 +// directory_iterator and recursive_directory_iterator are not C++20 ranges + +#include + +namespace fs = std::filesystem; +namespace rg = std::ranges; + +static_assert( rg::borrowed_range ); +static_assert( rg::borrowed_range ); + +static_assert( rg::view ); +static_assert( rg::view ); commit 4cb91df31d65c3bca5c6d4c2efcd51f454def884 Author: Jonathan Wakely Date: Mon Sep 2 11:29:13 2024 +0100 libstdc++: Specialize std::disable_sized_sentinel_for for std::move_iterator [PR116549] LWG 3736 added a partial specialization of this variable template for two std::move_iterator types. This is needed for the case where the types satisfy std::sentinel_for and are subtractable, but do not model the semantics requirements of std::sized_sentinel_for. libstdc++-v3/ChangeLog: PR libstdc++/116549 * include/bits/stl_iterator.h (disable_sized_sentinel_for): Define specialization for two move_iterator types, as per LWG 3736. * testsuite/24_iterators/move_iterator/lwg3736.cc: New test. (cherry picked from commit 819deae0a5bee079a7d5582fafaa098c26144ae8) diff --git a/libstdc++-v3/include/bits/stl_iterator.h b/libstdc++-v3/include/bits/stl_iterator.h index 2d4fc3435e1..2de447ce752 100644 --- a/libstdc++-v3/include/bits/stl_iterator.h +++ b/libstdc++-v3/include/bits/stl_iterator.h @@ -1829,6 +1829,14 @@ _GLIBCXX_BEGIN_NAMESPACE_VERSION { return _ReturnType(__i); } #if __cplusplus > 201703L && __cpp_lib_concepts + // _GLIBCXX_RESOLVE_LIB_DEFECTS + // 3736. move_iterator missing disable_sized_sentinel_for specialization + template + requires (!sized_sentinel_for<_Iterator1, _Iterator2>) + inline constexpr bool + disable_sized_sentinel_for, + move_iterator<_Iterator2>> = true; + // [iterators.common] Common iterators namespace __detail diff --git a/libstdc++-v3/testsuite/24_iterators/move_iterator/lwg3736.cc b/libstdc++-v3/testsuite/24_iterators/move_iterator/lwg3736.cc new file mode 100644 index 00000000000..eaf791b3089 --- /dev/null +++ b/libstdc++-v3/testsuite/24_iterators/move_iterator/lwg3736.cc @@ -0,0 +1,52 @@ +// { dg-do compile { target c++20 } } + +// 3736. move_iterator missing disable_sized_sentinel_for specialization + +#include + +template using MoveIter = std::move_iterator; + +using std::sized_sentinel_for; +using std::disable_sized_sentinel_for; + +// These assertions always passed, even without LWG 3736: +static_assert(sized_sentinel_for, MoveIter>); +static_assert(sized_sentinel_for, MoveIter>); +static_assert(not sized_sentinel_for, MoveIter>); +static_assert(not sized_sentinel_for, std::default_sentinel_t>); +static_assert(not disable_sized_sentinel_for, MoveIter>); + +// These types don't satisfy sized_sentinel_for anyway (because the subtraction +// is ill-formed) but LWG 3736 makes the variable template explicitly false: +static_assert(disable_sized_sentinel_for, MoveIter>); + +struct Iter +{ + using iterator_category = std::random_access_iterator_tag; + using value_type = int; + using pointer = int*; + using reference = int&; + using difference_type = long; + + Iter() = default; + Iter& operator++(); + Iter operator++(int); + Iter& operator--(); + Iter operator--(int); + reference operator*() const; + pointer operator->() const; + Iter& operator+=(difference_type); + Iter& operator-=(difference_type); + friend Iter operator+(Iter, difference_type); + friend Iter operator+(difference_type, Iter); + friend Iter operator-(Iter, difference_type); + friend difference_type operator-(Iter, Iter); + bool operator==(Iter) const; +}; + +// Specialize the variable template so that Iter is not its own sized sentinel: +template<> constexpr bool std::disable_sized_sentinel_for = true; +static_assert( not sized_sentinel_for ); + +// LWG 3736 means that affects std::move_iterator as well: +static_assert( not sized_sentinel_for, MoveIter> ); commit 9357f7e0fbfbf0d7d5adb2cd1588a765f40b6162 Author: Jonathan Wakely Date: Fri Feb 9 17:06:20 2024 +0000 libstdc++: Constrain std::vector default constructor [PR113841] This is needed to avoid errors outside the immediate context when evaluating is_default_constructible_v> when A is not default constructible. To avoid diagnostic regressions for 23_containers/vector/48101_neg.cc we need to make the std::allocator partial specializations default constructible, which they probably should have been anyway. libstdc++-v3/ChangeLog: PR libstdc++/113841 * include/bits/allocator.h (allocator): Add default constructor to partial specializations for cv-qualified types. * include/bits/stl_vector.h (_Vector_impl::_Vector_impl()): Constrain so that it's only present if the allocator is default constructible. * include/bits/stl_bvector.h (_Bvector_impl::_Bvector_impl()): Likewise. * testsuite/23_containers/vector/cons/113841.cc: New test. (cherry picked from commit 142cc4c223d695e515ed2504501b91d8a7ac6eb8) diff --git a/libstdc++-v3/include/bits/allocator.h b/libstdc++-v3/include/bits/allocator.h index aec0b374fd1..a27b7d92503 100644 --- a/libstdc++-v3/include/bits/allocator.h +++ b/libstdc++-v3/include/bits/allocator.h @@ -247,6 +247,7 @@ _GLIBCXX_BEGIN_NAMESPACE_VERSION { public: typedef _Tp value_type; + allocator() { } template allocator(const allocator<_Up>&) { } }; @@ -255,6 +256,7 @@ _GLIBCXX_BEGIN_NAMESPACE_VERSION { public: typedef _Tp value_type; + allocator() { } template allocator(const allocator<_Up>&) { } }; @@ -263,6 +265,7 @@ _GLIBCXX_BEGIN_NAMESPACE_VERSION { public: typedef _Tp value_type; + allocator() { } template allocator(const allocator<_Up>&) { } }; diff --git a/libstdc++-v3/include/bits/stl_bvector.h b/libstdc++-v3/include/bits/stl_bvector.h index e1559c253d8..aa8c4026d23 100644 --- a/libstdc++-v3/include/bits/stl_bvector.h +++ b/libstdc++-v3/include/bits/stl_bvector.h @@ -548,6 +548,9 @@ _GLIBCXX_BEGIN_NAMESPACE_CONTAINER _GLIBCXX20_CONSTEXPR _Bvector_impl() _GLIBCXX_NOEXCEPT_IF( is_nothrow_default_constructible<_Bit_alloc_type>::value) +#if __cpp_concepts + requires is_default_constructible_v<_Bit_alloc_type> +#endif : _Bit_alloc_type() { } diff --git a/libstdc++-v3/include/bits/stl_vector.h b/libstdc++-v3/include/bits/stl_vector.h index 977474acd7f..df1010c07af 100644 --- a/libstdc++-v3/include/bits/stl_vector.h +++ b/libstdc++-v3/include/bits/stl_vector.h @@ -136,6 +136,9 @@ _GLIBCXX_BEGIN_NAMESPACE_CONTAINER _GLIBCXX20_CONSTEXPR _Vector_impl() _GLIBCXX_NOEXCEPT_IF( is_nothrow_default_constructible<_Tp_alloc_type>::value) +#if __cpp_lib_concepts + requires is_default_constructible_v<_Tp_alloc_type> +#endif : _Tp_alloc_type() { } diff --git a/libstdc++-v3/testsuite/23_containers/vector/cons/113841.cc b/libstdc++-v3/testsuite/23_containers/vector/cons/113841.cc new file mode 100644 index 00000000000..a7721d27f79 --- /dev/null +++ b/libstdc++-v3/testsuite/23_containers/vector/cons/113841.cc @@ -0,0 +1,34 @@ +// { dg-do compile { target c++20 } } + +#include + +template +struct Alloc +{ + using value_type = T; + + Alloc(int) { } // not default constructible + + template Alloc(const Alloc&) { } + + T* allocate(std::size_t n) { return std::allocator().allocate(n); } + void deallocate(T* p, std::size_t n) { std::allocator().deallocate(p, n); } +}; + +template struct wrap { T t; }; + +template void do_adl(T&) { } + +void test_pr113841() +{ + using test_type = std::vector>; + std::pair>* h = nullptr; + do_adl(h); +} + +void test_pr113841_bool() +{ + using test_type = std::vector>; + std::pair>* h = nullptr; + do_adl(h); +} commit 7b0f505650310c11455ee7d34f95008658d3e655 Author: Jonathan Wakely Date: Mon Dec 9 17:35:24 2024 +0000 libstdc++: Skip redundant assertions in std::span construction [PR117966] As PR c++/117966 shows, the Debug Mode checks cause a compilation error for a global constexpr std::span. Those debug checks are redundant when constructing from an array or a range, because we already know we have a valid range and we know its size. Instead of delegating to the std::span(contiguous_iterator, contiguous_iterator) constructor, just initialize the data members directly. libstdc++-v3/ChangeLog: PR libstdc++/117966 * include/std/span (span(T (&)[N])): Do not delegate to constructor that performs redundant checks. (span(array&), span(const array&)): Likewise. (span(Range&&), span(const span&)): Likewise. * testsuite/23_containers/span/117966.cc: New test. (cherry picked from commit e95bda027e0b81922c1bf44770674190bdf787e8) diff --git a/libstdc++-v3/include/std/span b/libstdc++-v3/include/std/span index 251fed91abf..2b68ac42ba8 100644 --- a/libstdc++-v3/include/std/span +++ b/libstdc++-v3/include/std/span @@ -183,21 +183,21 @@ _GLIBCXX_BEGIN_NAMESPACE_VERSION requires (_Extent == dynamic_extent || _ArrayExtent == _Extent) constexpr span(type_identity_t (&__arr)[_ArrayExtent]) noexcept - : span(static_cast(__arr), _ArrayExtent) + : _M_ptr(__arr), _M_extent(_ArrayExtent) { } template requires __is_compatible_array<_Tp, _ArrayExtent>::value constexpr span(array<_Tp, _ArrayExtent>& __arr) noexcept - : span(static_cast(__arr.data()), _ArrayExtent) + : _M_ptr(__arr.data()), _M_extent(_ArrayExtent) { } template requires __is_compatible_array::value constexpr span(const array<_Tp, _ArrayExtent>& __arr) noexcept - : span(static_cast(__arr.data()), _ArrayExtent) + : _M_ptr(__arr.data()), _M_extent(_ArrayExtent) { } template @@ -211,7 +211,7 @@ _GLIBCXX_BEGIN_NAMESPACE_VERSION span(_Range&& __range) noexcept(noexcept(ranges::data(__range)) && noexcept(ranges::size(__range))) - : span(ranges::data(__range), ranges::size(__range)) + : _M_ptr(ranges::data(__range)), _M_extent(ranges::size(__range)) { if constexpr (extent != dynamic_extent) { @@ -229,7 +229,7 @@ _GLIBCXX_BEGIN_NAMESPACE_VERSION constexpr explicit(extent != dynamic_extent && _OExtent == dynamic_extent) span(const span<_OType, _OExtent>& __s) noexcept - : _M_extent(__s.size()), _M_ptr(__s.data()) + : _M_ptr(__s.data()), _M_extent(__s.size()) { if constexpr (extent != dynamic_extent) { diff --git a/libstdc++-v3/testsuite/23_containers/span/117966.cc b/libstdc++-v3/testsuite/23_containers/span/117966.cc new file mode 100644 index 00000000000..8bbb5ca1e07 --- /dev/null +++ b/libstdc++-v3/testsuite/23_containers/span/117966.cc @@ -0,0 +1,13 @@ +// { dg-options "-D_GLIBCXX_DEBUG" } +// { dg-do compile { target c++20 } } + +// Bug 117966 +// constexpr std::span construction fails to compile with D_GLIBCXX_DEBUG + +#include +#include + +struct A { + constexpr A(std::span) {} +}; +constexpr A val{std::array{0x11, 0x22}}; commit c0d805192b0092235b2ef32a184b17a76ebac401 Author: Jonathan Wakely Date: Mon Dec 9 17:35:24 2024 +0000 libstdc++: Skip redundant assertions in std::array equality [PR106212] As PR c++/106212 shows, the Debug Mode checks cause a compilation error for equality comparisons involving std::array prvalues in constant expressions. Those Debug Mode checks are redundant when comparing two std::array objects, because we already know we have a valid range. We can also avoid the unnecessary step of using std::__niter_base to do __normal_iterator unwrapping, which isn't needed because our std::array iterators are just pointers. Using std::__equal_aux1 instead of std::equal avoids the redundant checks in std::equal and std::__equal_aux. libstdc++-v3/ChangeLog: PR libstdc++/106212 * include/std/array (operator==): Use std::__equal_aux1 instead of std::equal. * testsuite/23_containers/array/comparison_operators/106212.cc: New test. (cherry picked from commit 3aeb2edee2f9fc39ab77c7e020f09d7204b167ac) diff --git a/libstdc++-v3/include/std/array b/libstdc++-v3/include/std/array index d8337f482ff..7cab84688ea 100644 --- a/libstdc++-v3/include/std/array +++ b/libstdc++-v3/include/std/array @@ -302,7 +302,7 @@ _GLIBCXX_BEGIN_NAMESPACE_VERSION _GLIBCXX20_CONSTEXPR inline bool operator==(const array<_Tp, _Nm>& __one, const array<_Tp, _Nm>& __two) - { return std::equal(__one.begin(), __one.end(), __two.begin()); } + { return std::__equal_aux1(__one.begin(), __one.end(), __two.begin()); } #if __cpp_lib_three_way_comparison && __cpp_lib_concepts template diff --git a/libstdc++-v3/testsuite/23_containers/array/comparison_operators/106212.cc b/libstdc++-v3/testsuite/23_containers/array/comparison_operators/106212.cc new file mode 100644 index 00000000000..f7b12bd04ef --- /dev/null +++ b/libstdc++-v3/testsuite/23_containers/array/comparison_operators/106212.cc @@ -0,0 +1,15 @@ +// { dg-options "-D_GLIBCXX_DEBUG" } +// { dg-do compile { target c++20 } } + +// Bug libstdc++/106212 - Code becomes non-constexpr with _GLIBCXX_DEBUG + +#include + +struct A +{ + constexpr A(int i) : e{i} {} + constexpr bool operator==(const A& a) const = default; + std::array e; +}; + +static_assert(A{1} != A{2}, ""); commit 6d8ee74e424c9d42b8164296dda0ec604a971253 Author: Jonathan Wakely Date: Thu Nov 14 17:31:43 2024 +0000 libstdc++: Fix get<0> constraint for lvalue ranges::subrange (LWG 3589) Approved at October 2021 plenary. libstdc++-v3/ChangeLog: * include/bits/ranges_util.h (subrange::begin): Fix constraint, as per LWG 3589. * testsuite/std/ranges/subrange/lwg3589.cc: New test. (cherry picked from commit 4a3a0be34f723df192361e43bb48b9292dfe3a54) diff --git a/libstdc++-v3/include/bits/ranges_util.h b/libstdc++-v3/include/bits/ranges_util.h index 37d7bc862f9..5299ce193c9 100644 --- a/libstdc++-v3/include/bits/ranges_util.h +++ b/libstdc++-v3/include/bits/ranges_util.h @@ -399,8 +399,11 @@ namespace ranges __detail::__make_unsigned_like_t>) -> subrange, sentinel_t<_Rng>, subrange_kind::sized>; + // _GLIBCXX_RESOLVE_LIB_DEFECTS + // 3589. The const lvalue reference overload of get for subrange does not + // constrain I to be copyable when N == 0 template - requires (_Num < 2) + requires ((_Num == 0 && copyable<_It>) || _Num == 1) constexpr auto get(const subrange<_It, _Sent, _Kind>& __r) { diff --git a/libstdc++-v3/testsuite/std/ranges/subrange/lwg3589.cc b/libstdc++-v3/testsuite/std/ranges/subrange/lwg3589.cc new file mode 100644 index 00000000000..1ccc52d81f8 --- /dev/null +++ b/libstdc++-v3/testsuite/std/ranges/subrange/lwg3589.cc @@ -0,0 +1,30 @@ +// { dg-do compile { target c++20 } } + +// LWG 3589. The const lvalue reference overload of get for subrange does not +// constrain I to be copyable when N == 0 + +#include +#include + +void +test_lwg3589() +{ + int a[2]{}; + __gnu_test::test_range r(a); + + // Use a generic lambda so we have a dependent context. + auto test = [](auto& x) + { + // This was wrong before the LWG 3589 change: + if constexpr (requires { std::ranges::get<0>(x); }) + (void) std::ranges::get<0>(x); + + // These always worked unconditionally: + (void) std::ranges::get<1>(x); + (void) std::ranges::get<0>(std::move(x)); + (void) std::ranges::get<1>(std::move(x)); + }; + + std::ranges::subrange sr(r.begin(), r.end()); + test(sr); +} commit cab0305aea0ca591b032adf09e97acd5e34c9f10 Author: Jonathan Wakely Date: Wed Apr 16 18:38:03 2025 +0100 libstdc++: Add dg-options "-std=gnu++20" to backported tests These tests were backported from gcc-14 where the testsuite automatically adds -std=gnu++20 as needed. That doesn't happen on the older release branches, so an explicit dg-options directive is needed to ensure the tests are run by default. Otherwise they'll only be run when somebody uses a custom --target_board that includes -std=gnu++20. For 29_atomics/headers/stdatomic.h/115807.cc we need to compile with -std=gnu++23 instead. libstdc++-v3/ChangeLog: * testsuite/20_util/integer_sequence/112473.cc: Compile with -std=gnu++20. * testsuite/21_strings/char_traits/requirements/113200.cc: Likewise. * testsuite/23_containers/array/comparison_operators/106212.cc: Likewise. * testsuite/23_containers/span/117966.cc: Likewise. * testsuite/23_containers/vector/cons/113841.cc: Compile with -std=gnu++20. * testsuite/24_iterators/move_iterator/lwg3736.cc: Likewise. * testsuite/25_algorithms/lexicographical_compare_three_way/113960.cc: Likewise. * testsuite/27_io/filesystem/iterators/lwg3480.cc: Likewise. * testsuite/29_atomics/headers/stdatomic.h/115807.cc: Compile with -std=gnu++23. * testsuite/std/ranges/subrange/lwg3589.cc: Likewise. * testsuite/std/time/month/2.cc: Likewise. * testsuite/std/time/weekday/2.cc: Likewise. (cherry picked from commit 4182fa87758d5a94c1f6d71f471842207e56f933) diff --git a/libstdc++-v3/testsuite/20_util/integer_sequence/112473.cc b/libstdc++-v3/testsuite/20_util/integer_sequence/112473.cc index 14abfbc8149..1c8035c6b1e 100644 --- a/libstdc++-v3/testsuite/20_util/integer_sequence/112473.cc +++ b/libstdc++-v3/testsuite/20_util/integer_sequence/112473.cc @@ -1,3 +1,4 @@ +// { dg-options "-std=gnu++20" } // { dg-do compile { target c++20 } } // PR libstdc++/112473 - integer_sequence accepts non-integer types diff --git a/libstdc++-v3/testsuite/21_strings/char_traits/requirements/113200.cc b/libstdc++-v3/testsuite/21_strings/char_traits/requirements/113200.cc index 0fe765d53bc..6a1b03293ef 100644 --- a/libstdc++-v3/testsuite/21_strings/char_traits/requirements/113200.cc +++ b/libstdc++-v3/testsuite/21_strings/char_traits/requirements/113200.cc @@ -1,3 +1,4 @@ +// { dg-options "-std=gnu++20" } // { dg-do compile { target c++20 } } // PR libstdc++/113200 diff --git a/libstdc++-v3/testsuite/23_containers/array/comparison_operators/106212.cc b/libstdc++-v3/testsuite/23_containers/array/comparison_operators/106212.cc index f7b12bd04ef..d6d9bf464b2 100644 --- a/libstdc++-v3/testsuite/23_containers/array/comparison_operators/106212.cc +++ b/libstdc++-v3/testsuite/23_containers/array/comparison_operators/106212.cc @@ -1,4 +1,4 @@ -// { dg-options "-D_GLIBCXX_DEBUG" } +// { dg-options "-D_GLIBCXX_DEBUG -std=gnu++20" } // { dg-do compile { target c++20 } } // Bug libstdc++/106212 - Code becomes non-constexpr with _GLIBCXX_DEBUG diff --git a/libstdc++-v3/testsuite/23_containers/span/117966.cc b/libstdc++-v3/testsuite/23_containers/span/117966.cc index 8bbb5ca1e07..cf30fcadd8f 100644 --- a/libstdc++-v3/testsuite/23_containers/span/117966.cc +++ b/libstdc++-v3/testsuite/23_containers/span/117966.cc @@ -1,4 +1,4 @@ -// { dg-options "-D_GLIBCXX_DEBUG" } +// { dg-options "-D_GLIBCXX_DEBUG -std=gnu++20" } // { dg-do compile { target c++20 } } // Bug 117966 diff --git a/libstdc++-v3/testsuite/23_containers/vector/cons/113841.cc b/libstdc++-v3/testsuite/23_containers/vector/cons/113841.cc index a7721d27f79..9c9c7d2d1c4 100644 --- a/libstdc++-v3/testsuite/23_containers/vector/cons/113841.cc +++ b/libstdc++-v3/testsuite/23_containers/vector/cons/113841.cc @@ -1,3 +1,4 @@ +// { dg-options "-std=gnu++20" } // { dg-do compile { target c++20 } } #include diff --git a/libstdc++-v3/testsuite/24_iterators/move_iterator/lwg3736.cc b/libstdc++-v3/testsuite/24_iterators/move_iterator/lwg3736.cc index eaf791b3089..49fb686aa06 100644 --- a/libstdc++-v3/testsuite/24_iterators/move_iterator/lwg3736.cc +++ b/libstdc++-v3/testsuite/24_iterators/move_iterator/lwg3736.cc @@ -1,3 +1,4 @@ +// { dg-options "-std=gnu++20" } // { dg-do compile { target c++20 } } // 3736. move_iterator missing disable_sized_sentinel_for specialization diff --git a/libstdc++-v3/testsuite/25_algorithms/lexicographical_compare_three_way/113960.cc b/libstdc++-v3/testsuite/25_algorithms/lexicographical_compare_three_way/113960.cc index d51ae1a3d50..388a535b904 100644 --- a/libstdc++-v3/testsuite/25_algorithms/lexicographical_compare_three_way/113960.cc +++ b/libstdc++-v3/testsuite/25_algorithms/lexicographical_compare_three_way/113960.cc @@ -1,3 +1,4 @@ +// { dg-options "-std=gnu++20" } // { dg-do run { target c++20 } } // PR libstdc++/113960 diff --git a/libstdc++-v3/testsuite/27_io/filesystem/iterators/lwg3480.cc b/libstdc++-v3/testsuite/27_io/filesystem/iterators/lwg3480.cc index 15e0286fff6..98e97a7775f 100644 --- a/libstdc++-v3/testsuite/27_io/filesystem/iterators/lwg3480.cc +++ b/libstdc++-v3/testsuite/27_io/filesystem/iterators/lwg3480.cc @@ -1,3 +1,4 @@ +// { dg-options "-std=gnu++20" } // { dg-do compile { target c++20 } } // { dg-require-filesystem-ts "" } diff --git a/libstdc++-v3/testsuite/29_atomics/headers/stdatomic.h/115807.cc b/libstdc++-v3/testsuite/29_atomics/headers/stdatomic.h/115807.cc index 14f320fe835..7f9d742838d 100644 --- a/libstdc++-v3/testsuite/29_atomics/headers/stdatomic.h/115807.cc +++ b/libstdc++-v3/testsuite/29_atomics/headers/stdatomic.h/115807.cc @@ -1,3 +1,4 @@ +// { dg-options "-std=gnu++23" } // { dg-do compile { target c++23 } } #include namespace other { diff --git a/libstdc++-v3/testsuite/std/ranges/subrange/lwg3589.cc b/libstdc++-v3/testsuite/std/ranges/subrange/lwg3589.cc index 1ccc52d81f8..4447b3186eb 100644 --- a/libstdc++-v3/testsuite/std/ranges/subrange/lwg3589.cc +++ b/libstdc++-v3/testsuite/std/ranges/subrange/lwg3589.cc @@ -1,3 +1,4 @@ +// { dg-options "-std=gnu++20" } // { dg-do compile { target c++20 } } // LWG 3589. The const lvalue reference overload of get for subrange does not diff --git a/libstdc++-v3/testsuite/std/time/month/2.cc b/libstdc++-v3/testsuite/std/time/month/2.cc index 3bcefa60003..4ef0adfbce3 100644 --- a/libstdc++-v3/testsuite/std/time/month/2.cc +++ b/libstdc++-v3/testsuite/std/time/month/2.cc @@ -1,3 +1,4 @@ +// { dg-options "-std=gnu++20" } // { dg-do run { target c++20 } } // Class month [time.cal.month] diff --git a/libstdc++-v3/testsuite/std/time/weekday/2.cc b/libstdc++-v3/testsuite/std/time/weekday/2.cc index 924709321e5..6980f78fbe1 100644 --- a/libstdc++-v3/testsuite/std/time/weekday/2.cc +++ b/libstdc++-v3/testsuite/std/time/weekday/2.cc @@ -1,3 +1,4 @@ +// { dg-options "-std=gnu++20" } // { dg-do run { target c++20 } } // Class weekday [time.cal.wd] commit 5347b382a38d664216bd7484dd5438a1863d02a3 Author: Andrew Pinski Date: Mon Dec 2 08:35:23 2024 -0800 phiopt: Reset the number of iterations information of a loop when changing an exit from the loop [PR117243] After r12-5300-gf98f373dd822b3, phiopt could get the following bb structure: | middle-bb -----| | | | |----| | phi<1, 2> | | cond | | | | | |--------+---| Which was considered 2 loops. The inner loop had esimtate of upper_bound to be 8, due to the original `for (b = 0; b <= 7; b++)`. The outer loop was already an infinite one. So phiopt would come along and change the condition to be unconditionally true, we change the inner loop to being an infinite one but don't reset the estimate on the loop and cleanup cfg comes along and changes it into one loop but also does not reset the estimate of the loop. Then the loop unrolling uses the old estimate and decides to add an unreachable there.o So the fix is when phiopt changes an exit to a loop, reset the estimates, similar to how cleanupcfg does it when merging some basic blocks. Bootstrapped and tested on x86_64-linux-gnu. PR tree-optimization/117243 PR tree-optimization/116749 gcc/ChangeLog: * tree-ssa-phiopt.cc (replace_phi_edge_with_variable): Reset loop estimates if the cond_block was an exit to a loop. gcc/testsuite/ChangeLog: * gcc.dg/torture/pr117243-1.c: New test. * gcc.dg/torture/pr117243-2.c: New test. Signed-off-by: Andrew Pinski (cherry picked from commit b7c69cc072ef0da36439ebc55c513b48e68391b7) diff --git a/gcc/testsuite/gcc.dg/torture/pr117243-1.c b/gcc/testsuite/gcc.dg/torture/pr117243-1.c new file mode 100644 index 00000000000..c4bbc31467c --- /dev/null +++ b/gcc/testsuite/gcc.dg/torture/pr117243-1.c @@ -0,0 +1,30 @@ +/* { dg-do compile } */ +/* { dg-options "-fdump-tree-optimized" } */ +/* { dg-skip-if "" { *-*-* } { "-fno-fat-lto-objects" } { "" } } */ + +/* PR tree-optimization/117243 */ +/* foo should be an infinite but sometimes it gets optimized incorrectly into + an __builtin_unreachable(); which is not valid. */ +void +foo (unsigned int a, unsigned char b) +{ + lbl: + for (b = 0; b <= 7; b++) + { + unsigned char c[1][1]; + int i, j; + for (i = 0; i < 1; i++) + for (j = 0; j < 1; j++) + c[i][j] = 1; + if (b) + goto lbl; + } +} + +int +main () +{ + foo (1, 2); +} + +/* { dg-final { scan-tree-dump-not "__builtin_unreachable " "optimized"} } */ diff --git a/gcc/testsuite/gcc.dg/torture/pr117243-2.c b/gcc/testsuite/gcc.dg/torture/pr117243-2.c new file mode 100644 index 00000000000..d9b0d3eeb98 --- /dev/null +++ b/gcc/testsuite/gcc.dg/torture/pr117243-2.c @@ -0,0 +1,34 @@ +/* { dg-do compile } */ +/* { dg-options "-fno-tree-ch -fdump-tree-optimized" } */ +/* { dg-skip-if "" { *-*-* } { "-fno-fat-lto-objects" } { "" } } */ + +/* PR tree-optimization/117243 */ +/* PR tree-optimization/116749 */ + +/* main1 should be an infinite but sometimes it gets optimized incorrectly into + an __builtin_unreachable(); which is not valid. */ +int main1 (void) +{ + int g=0; + int l1[1]; + int *l2 = &g; + int i; + for (i=0; i<1; i++) + l1[i] = (1); + for (g=0; g; ++g) + { + int *l3[1] = {&l1[0]}; + } + *l2 = *l1; +b: + for (i=0; i<2; ++i) + { + if (i) + goto b; + if (g) + continue; + } + return 0; +} + +/* { dg-final { scan-tree-dump-not "__builtin_unreachable " "optimized"} } */ diff --git a/gcc/tree-ssa-phiopt.cc b/gcc/tree-ssa-phiopt.cc index 3ef7d6b28fc..167b0b3be74 100644 --- a/gcc/tree-ssa-phiopt.cc +++ b/gcc/tree-ssa-phiopt.cc @@ -53,6 +53,7 @@ along with GCC; see the file COPYING3. If not see #include "gimple-match.h" #include "dbgcnt.h" #include "tree-ssa-propagate.h" +#include "tree-ssa-loop-niter.h" static unsigned int tree_ssa_phiopt_worker (bool, bool, bool); static bool two_value_replacement (basic_block, basic_block, edge, gphi *, @@ -429,6 +430,17 @@ replace_phi_edge_with_variable (basic_block cond_block, edge_to_remove = EDGE_SUCC (cond_block, 1); else edge_to_remove = EDGE_SUCC (cond_block, 0); + + /* If we are removing the cond on a loop exit, + reset number of iteration information of the loop. */ + if (loop_exits_from_bb_p (cond_block->loop_father, cond_block)) + { + auto loop = cond_block->loop_father; + free_numbers_of_iterations_estimates (loop); + loop->any_upper_bound = false; + loop->any_likely_upper_bound = false; + } + if (EDGE_COUNT (edge_to_remove->dest->preds) == 1) { e->flags |= EDGE_FALLTHRU; commit a5929ef809558a3117e7a6e2e66316a828b50f29 Author: Andrew Pinski Date: Tue Oct 1 14:48:19 2024 -0700 backprop: Fix deleting of a phi node [PR116922] The problem here is remove_unused_var is called on a name that is defined by a phi node but it deletes it like removing a normal statement. remove_phi_node should be called rather than gsi_remove for phinodes. Note there is a possibility of using simple_dce_from_worklist instead but that is for another day. Bootstrapped and tested on x86_64-linux-gnu. PR tree-optimization/116922 gcc/ChangeLog: * gimple-ssa-backprop.cc (remove_unused_var): Handle phi nodes correctly. gcc/testsuite/ChangeLog: * gcc.dg/torture/pr116922.c: New test. Signed-off-by: Andrew Pinski (cherry picked from commit cea87c84eacdb422caeada734ba5138c994d7022) diff --git a/gcc/gimple-ssa-backprop.cc b/gcc/gimple-ssa-backprop.cc index 68ea403e847..f8d5841db42 100644 --- a/gcc/gimple-ssa-backprop.cc +++ b/gcc/gimple-ssa-backprop.cc @@ -657,8 +657,14 @@ remove_unused_var (tree var) print_gimple_stmt (dump_file, stmt, 0, TDF_SLIM); } gimple_stmt_iterator gsi = gsi_for_stmt (stmt); - gsi_remove (&gsi, true); - release_defs (stmt); + if (gimple_code (stmt) == GIMPLE_PHI) + remove_phi_node (&gsi, true); + else + { + unlink_stmt_vdef (stmt); + gsi_remove (&gsi, true); + release_defs (stmt); + } } /* Note that we're replacing OLD_RHS with NEW_RHS in STMT. */ diff --git a/gcc/testsuite/gcc.dg/torture/pr116922.c b/gcc/testsuite/gcc.dg/torture/pr116922.c new file mode 100644 index 00000000000..0fcf912930f --- /dev/null +++ b/gcc/testsuite/gcc.dg/torture/pr116922.c @@ -0,0 +1,19 @@ +/* { dg-do compile } */ +/* { dg-additional-options "-ffast-math" } */ +/* PR tree-optimization/116922 */ + + +static int g; + +void +foo (int c, double v, double *r) +{ +b: + do + v /= g - v; + while (c); + *r = v; + + double x; + foo (5, (double)0, &x); +} commit df76d7b05693dc511a114cb6ece2cbc5eef25dea Author: Andrew Pinski Date: Sun Oct 27 13:16:22 2024 -0700 vec-lowering: Fix ABSU lowering [PR111285] ABSU_EXPR lowering incorrectly used the resulting type for the new expression but in the case of ABSU the resulting type is an unsigned type and with ABSU is folded away. The fix is to use a signed type for the expression instead. Bootstrapped and tested on x86_64-linux-gnu. PR middle-end/111285 gcc/ChangeLog: * tree-vect-generic.cc (do_unop): Use a signed type for the operand if the operation was ABSU_EXPR. gcc/testsuite/ChangeLog: * g++.dg/torture/vect-absu-1.C: New test. Signed-off-by: Andrew Pinski (cherry picked from commit ad0084337e901ddaedd48c14e7a5dad9fc2a093e) diff --git a/gcc/testsuite/g++.dg/torture/vect-absu-1.C b/gcc/testsuite/g++.dg/torture/vect-absu-1.C new file mode 100644 index 00000000000..0b2035f638f --- /dev/null +++ b/gcc/testsuite/g++.dg/torture/vect-absu-1.C @@ -0,0 +1,29 @@ +// { dg-do run } +// PR middle-end/111285 + +// The lowering of vect absu was done incorrectly + +#define vect1 __attribute__((vector_size(sizeof(int)))) + +#define negabs(a) a < 0 ? a : -a + +__attribute__((noinline)) +int s(int a) +{ + return negabs(a); +} +__attribute__((noinline)) +vect1 int v(vect1 int a) +{ + return negabs(a); +} + +int main(void) +{ + for(int i = -10; i < 10; i++) + { + vect1 int t = {i}; + if (v(t)[0] != s(i)) + __builtin_abort(); + } +} diff --git a/gcc/tree-vect-generic.cc b/gcc/tree-vect-generic.cc index 9a3ca26d414..5a6256a0b70 100644 --- a/gcc/tree-vect-generic.cc +++ b/gcc/tree-vect-generic.cc @@ -202,7 +202,15 @@ do_unop (gimple_stmt_iterator *gsi, tree inner_type, tree a, tree b ATTRIBUTE_UNUSED, tree bitpos, tree bitsize, enum tree_code code, tree type ATTRIBUTE_UNUSED) { - a = tree_vec_extract (gsi, inner_type, a, bitsize, bitpos); + tree rhs_type = inner_type; + + /* For ABSU_EXPR, use the signed type for the rhs if the rhs was signed. */ + if (code == ABSU_EXPR + && ANY_INTEGRAL_TYPE_P (TREE_TYPE (a)) + && !TYPE_UNSIGNED (TREE_TYPE (a))) + rhs_type = signed_type_for (rhs_type); + + a = tree_vec_extract (gsi, rhs_type, a, bitsize, bitpos); return gimplify_build1 (gsi, code, inner_type, a); } commit 9d0c2f769553c65a39c12df3c2d0891c53fafd6a Author: Andrew Pinski Date: Sat Mar 8 22:43:54 2025 -0800 phiopt: Fix value_replacement for middle bb having phi nodes [PR118922] After r12-5300-gf98f373dd822b3, value_replacement would be able to look at the following cfg structure: ``` [local count: 1014686024]: if (h_6 != 0) goto ; [94.50%] else goto ; [5.50%] [local count: 114863530]: # h_6 = PHI <0(4), 1(5)> [local count: 1073741824]: # f_8 = PHI <0(5), h_6(6)> _9 = f_8 ^ 1; a.0_10 = a; _11 = _9 + a.0_10; if (_11 != -117) goto ; [94.50%] else goto ; [5.50%] ``` value_replacement would incorrectly think the middle bb (6) was empty and so it decides to remove condition in bb5 and replacing it with 0 as the function thought it was `h_6 ? 0 : h_6`. But since the there is an incoming phi node to bb6 defining h_6 that is incorrect. The fix is to check if there is phi nodes in the middle bb and set empty_or_with_defined_p to false. This was not needed before r12-5300-gf98f373dd822b3 because the phi would have been dead otherwise due to other checks. Bootstrapped and tested on x86_64-linux-gnu. PR tree-optimization/118922 gcc/ChangeLog: * tree-ssa-phiopt.cc (value_replacement): Set empty_or_with_defined_p to false when there is phi nodes for the middle bb. gcc/testsuite/ChangeLog: * gcc.dg/torture/pr118922-1.c: New test. Signed-off-by: Andrew Pinski (cherry picked from commit 7232c005afb5002cdfd0a2dbd0e8b8f2d80250ce) diff --git a/gcc/testsuite/gcc.dg/torture/pr118922-1.c b/gcc/testsuite/gcc.dg/torture/pr118922-1.c new file mode 100644 index 00000000000..27e8c78c0e4 --- /dev/null +++ b/gcc/testsuite/gcc.dg/torture/pr118922-1.c @@ -0,0 +1,57 @@ +/* { dg-do run } */ +/* PR tree-optimization/118922 */ + +/* Phi-opt would convert: + [local count: 1014686024]: + if (h_6 != 0) + goto ; [94.50%] + else + goto ; [5.50%] + + [local count: 114863530]: + # h_6 = PHI <0(4), 1(5)> + + [local count: 1073741824]: + # f_8 = PHI <0(5), h_6(6)> + _9 = f_8 ^ 1; + a.0_10 = a; + _11 = _9 + a.0_10; + if (_11 != -117) + goto ; [94.50%] + else + goto ; [5.50%] + +into: + + [local count: 59055799]: + c = d_3; + + [local count: 1073741824]: + # f_8 = PHI <0(5), 0(4)> + _9 = f_8 ^ 1; + a.0_10 = a; + _11 = _9 + a.0_10; + if (_11 != -117) + goto ; [94.50%] + else + goto ; [5.50%] + +as it thought the middle bb was empty as there was only a phi node there. */ + + +int a = -117, b, c, e; +void g(int h) { + int f = 0; + while (!f + a - -117) { + f = h == 0; + if (h == 0) + h = 1; + } +} +int main() { + int d = 8; + for (; e;) + d = 0; + c = d; + g(0); +} diff --git a/gcc/tree-ssa-phiopt.cc b/gcc/tree-ssa-phiopt.cc index 167b0b3be74..aec0350010b 100644 --- a/gcc/tree-ssa-phiopt.cc +++ b/gcc/tree-ssa-phiopt.cc @@ -1322,6 +1322,9 @@ value_replacement (basic_block cond_bb, basic_block middle_bb, && jump_function_from_stmt (&arg1, stmt))) empty_or_with_defined_p = false; } + /* The middle bb is not empty if there are any phi nodes. */ + if (phi_nodes (middle_bb)) + empty_or_with_defined_p = false; cond = last_stmt (cond_bb); code = gimple_cond_code (cond); commit 8aff886ffeb435c37fc68861f6e5501834aa3603 Author: Andrew Pinski Date: Mon Aug 19 08:06:36 2024 -0700 match: Reject non-ssa name/min invariants in gimple_extract [PR116412] After the conversion for phiopt's conditional operand to use maybe_push_res_to_seq, it was found that gimple_extract will extract out from REALPART_EXPR/IMAGPART_EXPR/VCE and BIT_FIELD_REF, a memory load. But that extraction was not needed as memory loads are not simplified in match and simplify. So gimple_extract should return false in those cases. Changes since v1: * Move the rejection to gimple_extract from factor_out_conditional_operation. GCC13: the function is in gimple-match-head.cc rather than gimple-match-exports.cc. Bootstrapped and tested on x86_64-linux-gnu. PR tree-optimization/116412 gcc/ChangeLog: * gimple-match-head.cc (gimple_extract): Return false if op0 was not a SSA name nor a min invariant for REALPART_EXPR/IMAGPART_EXPR/VCE and BIT_FIELD_REF. gcc/testsuite/ChangeLog: * gcc.dg/torture/pr116412-1.c: New test. Signed-off-by: Andrew Pinski (cherry picked from commit c7b76a076cb2c6ded7ae208464019b04cb0531a2) diff --git a/gcc/gimple-match-head.cc b/gcc/gimple-match-head.cc index 2fd27fcbacc..451d736ffe4 100644 --- a/gcc/gimple-match-head.cc +++ b/gcc/gimple-match-head.cc @@ -943,6 +943,9 @@ gimple_extract (gimple *stmt, gimple_match_op *res_op, || code == VIEW_CONVERT_EXPR) { tree op0 = TREE_OPERAND (gimple_assign_rhs1 (stmt), 0); + /* op0 needs to be a SSA name or an min invariant. */ + if (TREE_CODE (op0) != SSA_NAME && !is_gimple_min_invariant (op0)) + return false; res_op->set_op (code, type, valueize_op (op0)); return true; } @@ -950,6 +953,9 @@ gimple_extract (gimple *stmt, gimple_match_op *res_op, { tree rhs1 = gimple_assign_rhs1 (stmt); tree op0 = valueize_op (TREE_OPERAND (rhs1, 0)); + /* op0 needs to be a SSA name or an min invariant. */ + if (TREE_CODE (op0) != SSA_NAME && !is_gimple_min_invariant (op0)) + return false; res_op->set_op (code, type, op0, TREE_OPERAND (rhs1, 1), TREE_OPERAND (rhs1, 2), diff --git a/gcc/testsuite/gcc.dg/torture/pr116412-1.c b/gcc/testsuite/gcc.dg/torture/pr116412-1.c new file mode 100644 index 00000000000..3bc26ecd8b8 --- /dev/null +++ b/gcc/testsuite/gcc.dg/torture/pr116412-1.c @@ -0,0 +1,6 @@ +/* { dg-do compile } */ +double f(_Complex double a, _Complex double *b, int c) +{ + if (c) return __real__ a; + return __real__ *b; +} commit 75fc02d56f0be5c4d8a26b2f48fcaa482a849094 Author: Andrew Pinski Date: Mon Apr 14 08:40:24 2025 -0700 testcase: Add testcase for already fixed PR [PR118476] This testcase was fixed by r15-3052-gc7b76a076cb2c6ded but is a testcase that failed in a different fashion and a much older failure than the one added with r15-3052. Pushed as obvious after a quick test. PR tree-optimization/118476 gcc/testsuite/ChangeLog: * gcc.dg/torture/pr118476-1.c: New test. Signed-off-by: Andrew Pinski (cherry picked from commit d45a6502d1ec87d43f1a39f87cca58f1e28369c8) diff --git a/gcc/testsuite/gcc.dg/torture/pr118476-1.c b/gcc/testsuite/gcc.dg/torture/pr118476-1.c new file mode 100644 index 00000000000..33509403b61 --- /dev/null +++ b/gcc/testsuite/gcc.dg/torture/pr118476-1.c @@ -0,0 +1,14 @@ +/* { dg-do compile } */ + +/* PR tree-optimization/118476 */ + +typedef unsigned long long poly64x1 __attribute__((__vector_size__(1*sizeof(long long)))); + +poly64x1 vext_p64(poly64x1 a, poly64x1 b, const int n) +{ + poly64x1 r = a; + unsigned src = (unsigned)n; + long long t = b[0]; + r[0] = (src < 1) ? a[src] : t; + return r; +} commit b517d0cef3c73815e294650146cd399f15471fc4 Author: GCC Administrator Date: Thu Apr 17 00:20:34 2025 +0000 Daily bump. diff --git a/gcc/ChangeLog b/gcc/ChangeLog index 0cdb6d268e7..f473d0c116c 100644 --- a/gcc/ChangeLog +++ b/gcc/ChangeLog @@ -1,3 +1,50 @@ +2025-04-16 Andrew Pinski + + Backported from master: + 2024-08-20 Andrew Pinski + + PR tree-optimization/116412 + * gimple-match-head.cc (gimple_extract): Return false if op0 + was not a SSA name nor a min invariant for REALPART_EXPR/IMAGPART_EXPR/VCE + and BIT_FIELD_REF. + +2025-04-16 Andrew Pinski + + Backported from master: + 2025-03-09 Andrew Pinski + + PR tree-optimization/118922 + * tree-ssa-phiopt.cc (value_replacement): Set empty_or_with_defined_p + to false when there is phi nodes for the middle bb. + +2025-04-16 Andrew Pinski + + Backported from master: + 2024-10-28 Andrew Pinski + + PR middle-end/111285 + * tree-vect-generic.cc (do_unop): Use a signed type for the + operand if the operation was ABSU_EXPR. + +2025-04-16 Andrew Pinski + + Backported from master: + 2024-10-02 Andrew Pinski + + PR tree-optimization/116922 + * gimple-ssa-backprop.cc (remove_unused_var): Handle phi + nodes correctly. + +2025-04-16 Andrew Pinski + + Backported from master: + 2024-12-04 Andrew Pinski + + PR tree-optimization/117243 + PR tree-optimization/116749 + * tree-ssa-phiopt.cc (replace_phi_edge_with_variable): Reset loop + estimates if the cond_block was an exit to a loop. + 2025-04-13 Richard Biener Backported from master: diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index c9d404d186e..f2901859e28 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250416 +20250417 diff --git a/gcc/testsuite/ChangeLog b/gcc/testsuite/ChangeLog index 1002a907daf..f09b200e3a6 100644 --- a/gcc/testsuite/ChangeLog +++ b/gcc/testsuite/ChangeLog @@ -1,3 +1,53 @@ +2025-04-16 Andrew Pinski + + Backported from master: + 2025-04-14 Andrew Pinski + + PR tree-optimization/118476 + * gcc.dg/torture/pr118476-1.c: New test. + +2025-04-16 Andrew Pinski + + Backported from master: + 2024-08-20 Andrew Pinski + + PR tree-optimization/116412 + * gcc.dg/torture/pr116412-1.c: New test. + +2025-04-16 Andrew Pinski + + Backported from master: + 2025-03-09 Andrew Pinski + + PR tree-optimization/118922 + * gcc.dg/torture/pr118922-1.c: New test. + +2025-04-16 Andrew Pinski + + Backported from master: + 2024-10-28 Andrew Pinski + + PR middle-end/111285 + * g++.dg/torture/vect-absu-1.C: New test. + +2025-04-16 Andrew Pinski + + Backported from master: + 2024-10-02 Andrew Pinski + + PR tree-optimization/116922 + * gcc.dg/torture/pr116922.c: New test. + +2025-04-16 Andrew Pinski + + Backported from master: + 2024-12-04 Andrew Pinski + + PR tree-optimization/117243 + PR tree-optimization/116749 + * gcc.dg/torture/pr117243-1.c: New test. + * gcc.dg/torture/pr117243-2.c: New test. + 2025-04-13 Richard Biener Backported from master: diff --git a/libstdc++-v3/ChangeLog b/libstdc++-v3/ChangeLog index f7da4e52bc8..1532f63000e 100644 --- a/libstdc++-v3/ChangeLog +++ b/libstdc++-v3/ChangeLog @@ -1,3 +1,95 @@ +2025-04-16 Jonathan Wakely + + Backported from master: + 2025-04-16 Jonathan Wakely + + * testsuite/20_util/integer_sequence/112473.cc: Compile with + -std=gnu++20. + * testsuite/21_strings/char_traits/requirements/113200.cc: + Likewise. + * testsuite/23_containers/array/comparison_operators/106212.cc: + Likewise. + * testsuite/23_containers/span/117966.cc: Likewise. + * testsuite/23_containers/vector/cons/113841.cc: Compile with + -std=gnu++20. + * testsuite/24_iterators/move_iterator/lwg3736.cc: Likewise. + * testsuite/25_algorithms/lexicographical_compare_three_way/113960.cc: + Likewise. + * testsuite/27_io/filesystem/iterators/lwg3480.cc: Likewise. + * testsuite/29_atomics/headers/stdatomic.h/115807.cc: Compile + with -std=gnu++23. + * testsuite/std/ranges/subrange/lwg3589.cc: Likewise. + * testsuite/std/time/month/2.cc: Likewise. + * testsuite/std/time/weekday/2.cc: Likewise. + +2025-04-16 Jonathan Wakely + + Backported from master: + 2024-11-14 Jonathan Wakely + + * include/bits/ranges_util.h (subrange::begin): Fix constraint, + as per LWG 3589. + * testsuite/std/ranges/subrange/lwg3589.cc: New test. + +2025-04-16 Jonathan Wakely + + Backported from master: + 2024-12-11 Jonathan Wakely + + PR libstdc++/106212 + * include/std/array (operator==): Use std::__equal_aux1 instead + of std::equal. + * testsuite/23_containers/array/comparison_operators/106212.cc: + New test. + +2025-04-16 Jonathan Wakely + + Backported from master: + 2024-12-11 Jonathan Wakely + + PR libstdc++/117966 + * include/std/span (span(T (&)[N])): Do not delegate to + constructor that performs redundant checks. + (span(array&), span(const array&)): Likewise. + (span(Range&&), span(const span&)): Likewise. + * testsuite/23_containers/span/117966.cc: New test. + +2025-04-16 Jonathan Wakely + + Backported from master: + 2024-03-22 Jonathan Wakely + + PR libstdc++/113841 + * include/bits/allocator.h (allocator): Add default + constructor to partial specializations for cv-qualified types. + * include/bits/stl_vector.h (_Vector_impl::_Vector_impl()): + Constrain so that it's only present if the allocator is default + constructible. + * include/bits/stl_bvector.h (_Bvector_impl::_Bvector_impl()): + Likewise. + * testsuite/23_containers/vector/cons/113841.cc: New test. + +2025-04-16 Jonathan Wakely + + Backported from master: + 2024-09-03 Jonathan Wakely + + PR libstdc++/116549 + * include/bits/stl_iterator.h (disable_sized_sentinel_for): + Define specialization for two move_iterator types, as per LWG + 3736. + * testsuite/24_iterators/move_iterator/lwg3736.cc: New test. + +2025-04-16 Jonathan Wakely + + Backported from master: + 2024-11-14 Jonathan Wakely + + PR libstdc++/117560 + * include/bits/fs_dir.h (enable_borrowed_range, enable_view): + Define specializations for directory iterators, as per LWG 3480. + * testsuite/27_io/filesystem/iterators/lwg3480.cc: New test. + 2025-04-15 Jonathan Wakely Backported from master: commit d14333852ebd77b898b3bf1030f9bf5152497345 Author: Jonathan Wakely Date: Mon Dec 16 17:42:24 2024 +0000 libstdc++: Fix std::deque::insert(pos, first, last) undefined behaviour [PR118035] Inserting an empty range into a std::deque results in undefined calls to either std::copy, std::copy_backward, std::move, or std::move_backward. We call those algos with invalid arguments where the output range is the same as the input range, e.g. std::copy(first, last, first) which violates the preconditions for the algorithms. This fix simply returns early if there's nothing to insert. Most callers already ensure that we don't even call _M_range_insert_aux with an empty range, but some callers don't. Rather than checking for n == 0 in each of the callers, this just does the check once and uses __builtin_expect to treat empty insertions as unlikely. libstdc++-v3/ChangeLog: PR libstdc++/118035 * include/bits/deque.tcc (_M_range_insert_aux): Return immediately if inserting an empty range. * testsuite/23_containers/deque/modifiers/insert/118035.cc: New test. (cherry picked from commit b273e25e11c842a5729d0e03c85088cf5ba8e06c) diff --git a/libstdc++-v3/include/bits/deque.tcc b/libstdc++-v3/include/bits/deque.tcc index 03e0a505e14..2ba5d7b8602 100644 --- a/libstdc++-v3/include/bits/deque.tcc +++ b/libstdc++-v3/include/bits/deque.tcc @@ -601,6 +601,9 @@ _GLIBCXX_BEGIN_NAMESPACE_CONTAINER std::forward_iterator_tag) { const size_type __n = std::distance(__first, __last); + if (__builtin_expect(__n == 0, 0)) + return; + if (__pos._M_cur == this->_M_impl._M_start._M_cur) { iterator __new_start = _M_reserve_elements_at_front(__n); diff --git a/libstdc++-v3/testsuite/23_containers/deque/modifiers/insert/118035.cc b/libstdc++-v3/testsuite/23_containers/deque/modifiers/insert/118035.cc new file mode 100644 index 00000000000..a37d3dc3d04 --- /dev/null +++ b/libstdc++-v3/testsuite/23_containers/deque/modifiers/insert/118035.cc @@ -0,0 +1,26 @@ +// { dg-do run } + +#include +#include + +struct Sparks +{ + Sparks& operator=(const Sparks& s) + { + VERIFY( this != &s ); // This town ain't big enough for the both of us. + return *this; + } +}; + +void +test_pr118035() +{ + std::deque d(3, Sparks()); + Sparks s[1]; + d.insert(d.begin() + 1, s, s); +} + +int main() +{ + test_pr118035(); +} commit 1bfcb247ef2f2b7afb9f4b682740ceb332db86ce Author: Jonathan Wakely Date: Fri Mar 28 22:00:38 2025 +0000 libstdc++: Fix bogus -Wstringop-overflow in std::vector::insert [PR117983] This was fixed on trunk by r15-4473-g3abe751ea86e34, but that isn't suitable for backporting. Instead, just add another unreachable condition in std::vector::_M_range_insert so the compiler knows this memcpy doesn't use a length originating from a negative ptrdiff_t converted to a very positive size_t. libstdc++-v3/ChangeLog: PR libstdc++/117983 * include/bits/vector.tcc (vector::_M_range_insert): Add unreachable condition to tell the compiler begin() <= end(). * testsuite/23_containers/vector/modifiers/insert/117983.cc: New test. Reviewed-by: Tomasz Kamiński (cherry picked from commit 878812b6f6905774ab37cb78903e3e11bf1c508c) diff --git a/libstdc++-v3/include/bits/vector.tcc b/libstdc++-v3/include/bits/vector.tcc index 27ef1a4ee7f..d949e2d418e 100644 --- a/libstdc++-v3/include/bits/vector.tcc +++ b/libstdc++-v3/include/bits/vector.tcc @@ -793,6 +793,8 @@ _GLIBCXX_BEGIN_NAMESPACE_CONTAINER // reachable. pointer __old_start = this->_M_impl._M_start; pointer __old_finish = this->_M_impl._M_finish; + if ((__old_finish - __old_start) < 0) + __builtin_unreachable(); const size_type __len = _M_check_len(__n, "vector::_M_range_insert"); diff --git a/libstdc++-v3/testsuite/23_containers/vector/modifiers/insert/117983.cc b/libstdc++-v3/testsuite/23_containers/vector/modifiers/insert/117983.cc new file mode 100644 index 00000000000..e6027a677ee --- /dev/null +++ b/libstdc++-v3/testsuite/23_containers/vector/modifiers/insert/117983.cc @@ -0,0 +1,17 @@ +// { dg-options "-O3 -Werror=stringop-overflow" } +// { dg-do compile } + +// PR libstdc++/117983 +// -Wstringop-overflow false positive for __builtin_memmove from vector::insert + +#include + +typedef std::vector bytes; + +void push(bytes chunk, bytes& data) { + if (data.empty()) { + data.swap(chunk); + } else { + data.insert(data.end(), chunk.begin(), chunk.end()); + } +} commit d143630ff7242ebaf9667667ecc1deb6bb678023 Author: Jonathan Wakely Date: Thu Apr 4 10:33:33 2024 +0100 libstdc++: Fix infinite loop in std::istream::ignore(n, delim) [PR93672] A negative delim value passed to std::istream::ignore can never match any character in the stream, because the comparison is done using traits_type::eq_int_type(sb->sgetc(), delim) and sgetc() never returns negative values (except at EOF). The optimized version of ignore for the std::istream specialization uses traits_type::find to locate the delim character in the streambuf, which _can_ match a negative delim on platforms where char is signed, but then we do another comparison using eq_int_type which fails. The code then keeps looping forever, with traits_type::find locating the character and traits_type::eq_int_type saying it's not a match, so traits_type::find is used again and finds the same character again. A possible fix would be to check with eq_int_type after a successful find, to see whether we really have a match. However, that would be suboptimal since we know that a negative delimiter will never match using eq_int_type. So a better fix is to adjust the check at the top of the function that handles delim==eof(), so that we treat all negative delim values as equivalent to EOF. That way we don't bother using find to search for something that will never match with eq_int_type. The version of ignore in the primary template doesn't need a change, because it doesn't use traits_type::find, instead characters are extracted one-by-one and always matched using eq_int_type. That avoids the inconsistency between find and eq_int_type. The specialization for std::wistream does use traits_type::find, but traits_type::to_int_type is equivalent to an implicit conversion from wchar_t to wint_t, so passing a wchar_t directly to ignore without using to_int_type works. libstdc++-v3/ChangeLog: PR libstdc++/93672 * src/c++98/istream.cc (istream::ignore(streamsize, int_type)): Treat all negative delimiter values as eof(). * testsuite/27_io/basic_istream/ignore/char/93672.cc: New test. * testsuite/27_io/basic_istream/ignore/wchar_t/93672.cc: New test. (cherry picked from commit 2d694414ada8e3b58f504c1b175d31088529632e) diff --git a/libstdc++-v3/src/c++98/istream.cc b/libstdc++-v3/src/c++98/istream.cc index 4d1e34fee84..33a511e58a4 100644 --- a/libstdc++-v3/src/c++98/istream.cc +++ b/libstdc++-v3/src/c++98/istream.cc @@ -112,8 +112,17 @@ _GLIBCXX_BEGIN_NAMESPACE_VERSION basic_istream:: ignore(streamsize __n, int_type __delim) { - if (traits_type::eq_int_type(__delim, traits_type::eof())) - return ignore(__n); + { + // If conversion to int_type changes the value then __delim does not + // correspond to a value of type char_type, and so will never match + // a character extracted from the input sequence. Just use ignore(n). + const int_type chk_delim = traits_type::to_int_type(__delim); + const bool matchable = traits_type::eq_int_type(chk_delim, __delim); + if (__builtin_expect(!matchable, 0)) + return ignore(__n); + // Now we know that __delim is a valid char_type value, so it's safe + // for the code below to use traits_type::find to search for it. + } _M_gcount = 0; sentry __cerb(*this, true); diff --git a/libstdc++-v3/testsuite/27_io/basic_istream/ignore/char/93672.cc b/libstdc++-v3/testsuite/27_io/basic_istream/ignore/char/93672.cc new file mode 100644 index 00000000000..96737485b83 --- /dev/null +++ b/libstdc++-v3/testsuite/27_io/basic_istream/ignore/char/93672.cc @@ -0,0 +1,101 @@ +// { dg-do run } + +#include +#include +#include + +void +test_pr93672() // std::basic_istream::ignore hangs if delim MSB is set +{ + std::istringstream in(".\xfc..\xfd...\xfe."); + + // This should find '\xfd' even on platforms where char is signed, + // because the delimiter is correctly converted to the stream's int_type. + in.ignore(100, std::char_traits::to_int_type('\xfc')); + VERIFY( in.gcount() == 2 ); + VERIFY( ! in.eof() ); + + // This should work equivalently to traits_type::to_int_type + in.ignore(100, (unsigned char)'\xfd'); + VERIFY( in.gcount() == 3 ); + VERIFY( ! in.eof() ); + + // This only works if char is unsigned. + in.ignore(100, '\xfe'); + if (std::numeric_limits::is_signed) + { + // When char is signed, '\xfe' != traits_type::to_int_type('\xfe') + // so the delimiter does not match the character in the input sequence, + // and ignore consumes all input until EOF. + VERIFY( in.gcount() == 5 ); + VERIFY( in.eof() ); + } + else + { + // When char is unsigned, '\xfe' == to_int_type('\xfe') so the delimiter + // matches the character in the input sequence, and doesn't reach EOF. + VERIFY( in.gcount() == 4 ); + VERIFY( ! in.eof() ); + } + + in.clear(); + in.str(".a."); + in.ignore(100, 'a' + 256); // Should not match 'a' + VERIFY( in.gcount() == 3 ); + VERIFY( in.eof() ); +} + +// Custom traits type that inherits all behaviour from std::char_traits. +struct traits : std::char_traits { }; + +void +test_primary_template() +{ + // Check that the primary template for std::basic_istream::ignore + // works the same as the std::istream::ignore specialization. + // The infinite loop bug was never present in the primary template, + // because it doesn't use traits_type::find to search the input sequence. + + std::basic_istringstream in(".\xfc..\xfd...\xfe."); + + // This should find '\xfd' even on platforms where char is signed, + // because the delimiter is correctly converted to the stream's int_type. + in.ignore(100, std::char_traits::to_int_type('\xfc')); + VERIFY( in.gcount() == 2 ); + VERIFY( ! in.eof() ); + + // This should work equivalently to traits_type::to_int_type + in.ignore(100, (unsigned char)'\xfd'); + VERIFY( in.gcount() == 3 ); + VERIFY( ! in.eof() ); + + // This only works if char is unsigned. + in.ignore(100, '\xfe'); + if (std::numeric_limits::is_signed) + { + // When char is signed, '\xfe' != traits_type::to_int_type('\xfe') + // so the delimiter does not match the character in the input sequence, + // and ignore consumes all input until EOF. + VERIFY( in.gcount() == 5 ); + VERIFY( in.eof() ); + } + else + { + // When char is unsigned, '\xfe' == to_int_type('\xfe') so the delimiter + // matches the character in the input sequence, and doesn't reach EOF. + VERIFY( in.gcount() == 4 ); + VERIFY( ! in.eof() ); + } + + in.clear(); + in.str(".a."); + in.ignore(100, 'a' + 256); // Should not match 'a' + VERIFY( in.gcount() == 3 ); + VERIFY( in.eof() ); +} + +int main() +{ + test_pr93672(); + test_primary_template(); +} diff --git a/libstdc++-v3/testsuite/27_io/basic_istream/ignore/wchar_t/93672.cc b/libstdc++-v3/testsuite/27_io/basic_istream/ignore/wchar_t/93672.cc new file mode 100644 index 00000000000..5ce9155e02c --- /dev/null +++ b/libstdc++-v3/testsuite/27_io/basic_istream/ignore/wchar_t/93672.cc @@ -0,0 +1,34 @@ +// { dg-do run } + +#include +#include +#include +#include + +// PR 93672 was a bug in std::istream that never affected std::wistream. +// This test ensures that the bug doesn't get introduced to std::wistream. +void +test_pr93672() +{ + std::wstring str = L".x..x."; + str[1] = (wchar_t)-2; + str[4] = (wchar_t)-3; + std::wistringstream in(str); + + // This should find the character even on platforms where wchar_t is signed, + // because the delimiter is correctly converted to the stream's int_type. + in.ignore(100, std::char_traits::to_int_type((wchar_t)-2)); + VERIFY( in.gcount() == 2 ); + VERIFY( ! in.eof() ); + + // This also works, because std::char_traits::to_int_type(wc) is + // equivalent to (int_type)wc so using to_int_type isn't needed. + in.ignore(100, (wchar_t)-3); + VERIFY( in.gcount() == 3 ); + VERIFY( ! in.eof() ); +} + +int main() +{ + test_pr93672(); +} commit 2ea39e7b58376b2fb57cd74098a374604b627266 Author: Jonathan Wakely Date: Fri Jun 23 13:50:01 2023 +0100 libstdc++: Qualify calls to debug mode helpers These functions should be qualified to disable unwanted ADL. The overload of __check_singular_aux for safe iterators was previously being found by ADL, because it wasn't declared before __check_singular. Add a declaration so that it can be found by qualified lookup. libstdc++-v3/ChangeLog: * include/debug/helper_functions.h (__get_distance) (__check_singular, __valid_range_aux, __valid_range): Qualify calls to disable ADL. (__check_singular_aux(const _Safe_iterator_base*)): Declare overload that was previously found via ADL. (cherry picked from commit fa98bc4270dcb4ec78b5b1c0f4c067094c84bae6) diff --git a/libstdc++-v3/include/debug/helper_functions.h b/libstdc++-v3/include/debug/helper_functions.h index d28e4b51bb9..a641fc2eaa0 100644 --- a/libstdc++-v3/include/debug/helper_functions.h +++ b/libstdc++-v3/include/debug/helper_functions.h @@ -111,12 +111,19 @@ namespace __gnu_debug _GLIBCXX_CONSTEXPR inline typename _Distance_traits<_Iterator>::__type __get_distance(_Iterator __lhs, _Iterator __rhs) - { return __get_distance(__lhs, __rhs, std::__iterator_category(__lhs)); } + { + return __gnu_debug::__get_distance(__lhs, __rhs, + std::__iterator_category(__lhs)); + } // An arbitrary iterator pointer is not singular. inline bool __check_singular_aux(const void*) { return false; } + // Defined in + bool + __check_singular_aux(const class _Safe_iterator_base*); + // We may have an iterator that derives from _Safe_iterator_base but isn't // a _Safe_iterator. template @@ -125,7 +132,7 @@ namespace __gnu_debug __check_singular(_Iterator const& __x) { return ! std::__is_constant_evaluated() - && __check_singular_aux(std::__addressof(__x)); + && __gnu_debug::__check_singular_aux(std::__addressof(__x)); } /** Non-NULL pointers are nonsingular. */ @@ -163,7 +170,8 @@ namespace __gnu_debug std::input_iterator_tag) { return __first == __last - || (!__check_singular(__first) && !__check_singular(__last)); + || (!__gnu_debug::__check_singular(__first) + && !__gnu_debug::__check_singular(__last)); } template @@ -172,8 +180,8 @@ namespace __gnu_debug __valid_range_aux(_InputIterator __first, _InputIterator __last, std::random_access_iterator_tag) { - return - __valid_range_aux(__first, __last, std::input_iterator_tag()) + return __gnu_debug::__valid_range_aux(__first, __last, + std::input_iterator_tag()) && __first <= __last; } @@ -186,8 +194,8 @@ namespace __gnu_debug __valid_range_aux(_InputIterator __first, _InputIterator __last, std::__false_type) { - return __valid_range_aux(__first, __last, - std::__iterator_category(__first)); + return __gnu_debug::__valid_range_aux(__first, __last, + std::__iterator_category(__first)); } template @@ -197,10 +205,11 @@ namespace __gnu_debug typename _Distance_traits<_InputIterator>::__type& __dist, std::__false_type) { - if (!__valid_range_aux(__first, __last, std::input_iterator_tag())) + if (!__gnu_debug::__valid_range_aux(__first, __last, + std::input_iterator_tag())) return false; - __dist = __get_distance(__first, __last); + __dist = __gnu_debug::__get_distance(__first, __last); switch (__dist.second) { case __dp_none: @@ -231,7 +240,8 @@ namespace __gnu_debug typename _Distance_traits<_InputIterator>::__type& __dist) { typedef typename std::__is_integer<_InputIterator>::__type _Integral; - return __valid_range_aux(__first, __last, __dist, _Integral()); + return __gnu_debug::__valid_range_aux(__first, __last, __dist, + _Integral()); } template @@ -254,7 +264,7 @@ namespace __gnu_debug __valid_range(_InputIterator __first, _InputIterator __last) { typedef typename std::__is_integer<_InputIterator>::__type _Integral; - return __valid_range_aux(__first, __last, _Integral()); + return __gnu_debug::__valid_range_aux(__first, __last, _Integral()); } template commit 9fac6eeafa4573870ad1f197b134cd5fc4809fee Author: GCC Administrator Date: Fri Apr 18 00:20:19 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index f2901859e28..18aa6a5fa2d 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250417 +20250418 diff --git a/libstdc++-v3/ChangeLog b/libstdc++-v3/ChangeLog index 1532f63000e..4ab3df7ad91 100644 --- a/libstdc++-v3/ChangeLog +++ b/libstdc++-v3/ChangeLog @@ -1,3 +1,48 @@ +2025-04-17 Jonathan Wakely + + Backported from master: + 2023-06-26 Jonathan Wakely + + * include/debug/helper_functions.h (__get_distance) + (__check_singular, __valid_range_aux, __valid_range): Qualify + calls to disable ADL. + (__check_singular_aux(const _Safe_iterator_base*)): Declare + overload that was previously found via ADL. + +2025-04-17 Jonathan Wakely + + Backported from master: + 2024-04-15 Jonathan Wakely + + PR libstdc++/93672 + * src/c++98/istream.cc (istream::ignore(streamsize, int_type)): + Treat all negative delimiter values as eof(). + * testsuite/27_io/basic_istream/ignore/char/93672.cc: New test. + * testsuite/27_io/basic_istream/ignore/wchar_t/93672.cc: New + test. + +2025-04-17 Jonathan Wakely + + Backported from master: + 2025-03-28 Jonathan Wakely + + PR libstdc++/117983 + * include/bits/vector.tcc (vector::_M_range_insert): Add + unreachable condition to tell the compiler begin() <= end(). + * testsuite/23_containers/vector/modifiers/insert/117983.cc: New + test. + +2025-04-17 Jonathan Wakely + + Backported from master: + 2024-12-17 Jonathan Wakely + + PR libstdc++/118035 + * include/bits/deque.tcc (_M_range_insert_aux): Return + immediately if inserting an empty range. + * testsuite/23_containers/deque/modifiers/insert/118035.cc: New + test. + 2025-04-16 Jonathan Wakely Backported from master: commit 382798a5ad546abf38972e082bc2a8e2c3ebf64c Author: GCC Administrator Date: Sat Apr 19 00:21:01 2025 +0000 Daily bump. diff --git a/gcc/DATESTAMP b/gcc/DATESTAMP index 18aa6a5fa2d..f0d1b43cb20 100644 --- a/gcc/DATESTAMP +++ b/gcc/DATESTAMP @@ -1 +1 @@ -20250418 +20250419