[PATCH v2 04/13] ifn: optab: aarch64: Add optabs and internal functions for hssr operations.

Alfie Richards <[email protected]>
Newsgroups gmane.comp.gcc.patches
Message-ID <[email protected]>
Also removes pfalse ldff1 tests as the calls can no longer be removed
due to virtual operation which represents the HSSR state.

gcc/ChangeLog:

	* config/aarch64/aarch64-sve-builtins-base.cc:
	(svldxf1_impl::fold): Update to use new optabs.
	(svldxf1_impl::expand): Update to use new optabs.
	(aarch64_wrffr::fold): Update to use new IFN's.
	(aarch64_wrffr::expand): Update to use new optabs.
	(svsetffr_impl::fold): Update to use new IFN's.
	(svsetffr_impl::expand): Update to use new optabs.
	(svwrffr_impl::fold): Update to use new optabs.
	(svwrffr_impl::expand): Update to use new IFN's.
	* config/aarch64/aarch64-sve.md
	(@set_hssr_state<mode>): Rename aarch64_wrffr and add new modes.
	(aarch64_rdffr<mode>): Update for new modes.
	(aarch64_ld<fn>f1<mode>): Add support for extending loads.
	(mask_ff_hssr_load<mode><vpred>): New define expand.
	* internal-fn.cc (read_hssr_state_direct): New macro.
	(set_hssr_state_direct): New macro.
	(detect_hssr_fault_direct): New macro.
	(expand_read_hssr_state_optab_fn): New macro.
	(expand_detect_hssr_fault_optab_fn): New macro.
	(expand_set_hssr_state_optab_fn): New function.
	(direct_read_hssr_state_optab_supported_p): New macro.
	(direct_set_hssr_state_optab_supported_p): New macro.
	(direct_detect_hssr_fault_optab_supported_p): New macro.
	(internal_load_fn_p): Add support for IFN_MASK_FF_HSSR_LOAD.
	(internal_fn_else_index): Add support for IFN_MASK_FF_HSSR_LOAD.
	(internal_fn_mask_index): Add support for IFN_MASK_FF_HSSR_LOAD.
	(internal_fn_alias_ptr_index): Add support for IFN_MASK_FF_HSSR_LOAD.
	* internal-fn.def (MASK_FF_HSSR_LOAD): New IFN.
	(SET_HSSR_STATE): New IFN.
	(READ_HSSR_STATE): New IFN.
	(DETECT_HSSR_FAULT): New IFN.
	* optabs.def (mask_ff_hssr_load_optab): New optab.
	(read_hssr_state_optab): New optab.
	(set_hssr_state_optab): New optab.
	(detect_hssr_fault_optab): New optab.
	* tree-data-ref.cc (get_references_in_stmt):
	Add support for IFN_MASK_FF_HSSR_LOAD.
	* tree-ssa-alias.cc (ref_maybe_used_by_call_p_1):
	Add support for IFN_MASK_FF_HSSR_LOAD.
	* tree-ssa-loop-ivopts.cc (get_mem_type_for_internal_fn):
	Add support for IFN_MASK_FF_HSSR_LOAD.
	(get_alias_ptr_type_for_ptr_address):
	Add support for IFN_MASK_FF_HSSR_LOAD.
	* tree-vect-slp.cc (vect_get_operand_map):
	Add support for IFN_MASK_FF_HSSR_LOAD.

gcc/testsuite/ChangeLog:

	* gcc.target/aarch64/sve/pfalse-load.c: Remove ldff1 tests.

Co-Authored-By: Alex Coplan <[email protected]>

---
 .../aarch64/aarch64-sve-builtins-base.cc      |  74 ++++--
 gcc/config/aarch64/aarch64-sve.md             | 221 ++++++++++++++++--
 gcc/internal-fn.cc                            |  25 ++
 gcc/internal-fn.def                           |  18 ++
 gcc/optabs.def                                |   8 +
 .../gcc.target/aarch64/sve/pfalse-load.c      |   7 +-
 gcc/tree-data-ref.cc                          |   2 +
 gcc/tree-ssa-alias.cc                         |   1 +
 gcc/tree-ssa-loop-ivopts.cc                   |   2 +
 gcc/tree-vect-slp.cc                          |   3 +
 10 files changed, 315 insertions(+), 46 deletions(-)

diff --git a/gcc/config/aarch64/aarch64-sve-builtins-base.cc b/gcc/config/aarch64/aarch64-sve-builtins-base.cc
index 800541a9262..febcfe163c6 100644
--- a/gcc/config/aarch64/aarch64-sve-builtins-base.cc
+++ b/gcc/config/aarch64/aarch64-sve-builtins-base.cc
@@ -2046,7 +2046,10 @@ public:
 class svldxf1_impl : public full_width_access
 {
 public:
-  constexpr svldxf1_impl (unspec unspec) : m_unspec (unspec) {}
+  constexpr svldxf1_impl (unspec unspec) : m_first (unspec == UNSPEC_LDFF1)
+  {
+    gcc_checking_assert (m_first || unspec == UNSPEC_LDNF1);
+  }
 
   unsigned int
   call_properties (const function_instance &) const override
@@ -2054,19 +2057,33 @@ public:
     return CP_READ_MEMORY | CP_READ_FFR | CP_WRITE_FFR;
   }
 
-  rtx
-  expand (function_expander &e) const override
+  gimple *fold (gimple_folder &f) const override
   {
-    /* See the block comment in aarch64-sve.md for details about the
-       FFR handling.  */
-    emit_insn (gen_aarch64_update_ffr_for_load ());
+    if (m_first)
+      return f.fold_contiguous_load (IFN_MASK_FF_HSSR_LOAD);
+    else
+      return nullptr;
+  }
 
-    machine_mode mode = e.vector_mode (0);
-    return e.use_contiguous_load_insn (code_for_aarch64_ldf1 (m_unspec, mode));
+  rtx expand (function_expander &e) const override
+  {
+    if (m_first)
+      {
+	auto icode = convert_optab_handler (mask_ff_hssr_load_optab,
+					    e.vector_mode (0), e.gp_mode (0));
+	return e.use_contiguous_load_insn (icode, true);
+      }
+    else
+      {
+	emit_insn (gen_aarch64_update_ffr_for_load ());
+	machine_mode mode = e.vector_mode (0);
+	return e.use_contiguous_load_insn (code_for_aarch64_ldf1 (UNSPEC_LDNF1,
+								  mode),
+					   true);
+      }
   }
 
-  /* The unspec associated with the load.  */
-  unspec m_unspec;
+  bool m_first;
 };
 
 /* Implements extending contiguous forms of svldff1 and svldnf1.  */
@@ -2848,15 +2865,24 @@ public:
     return CP_READ_FFR;
   }
 
-  rtx
-  expand (function_expander &e) const override
+  gimple *fold (gimple_folder &f) const override
+  {
+    if (f.pred == PRED_z)
+      return nullptr;
+
+    gcall *new_call = gimple_build_call_internal (IFN_READ_HSSR_STATE, 0);
+    gimple_call_set_lhs (new_call, f.lhs);
+    return new_call;
+  }
+
+  rtx expand (function_expander &e) const override
   {
     /* See the block comment in aarch64-sve.md for details about the
        FFR handling.  */
     emit_insn (gen_aarch64_copy_ffr_to_ffrt ());
     rtx result = e.use_exact_insn (e.pred == PRED_z
-				   ? CODE_FOR_aarch64_rdffr_z
-				   : CODE_FOR_aarch64_rdffr);
+				   ? CODE_FOR_aarch64_rdffr_zvnx16bi
+				   : CODE_FOR_aarch64_rdffrvnx16bi);
     emit_insn (gen_aarch64_update_ffrt ());
     return result;
   }
@@ -3041,11 +3067,17 @@ public:
     return CP_WRITE_FFR;
   }
 
-  rtx
-  expand (function_expander &e) const override
+  gimple *fold (gimple_folder &f ATTRIBUTE_UNUSED) const override
+  {
+    tree bool_type = acle_vector_types[0][VECTOR_TYPE_svbool_t];
+    return gimple_build_call_internal (IFN_SET_HSSR_STATE, 1,
+				       build_all_ones_cst (bool_type));
+  }
+
+  rtx expand (function_expander &e) const override
   {
     e.args.quick_push (CONSTM1_RTX (VNx16BImode));
-    return e.use_exact_insn (CODE_FOR_aarch64_wrffr);
+    return e.use_exact_insn (code_for_set_hssr_state (VNx16BImode));
   }
 };
 
@@ -3468,10 +3500,16 @@ public:
     return CP_WRITE_FFR;
   }
 
+  gimple *fold (gimple_folder &f) const override
+  {
+    return gimple_build_call_internal (IFN_SET_HSSR_STATE, 1,
+				       gimple_call_arg (f.call, 0));
+  }
+
   rtx
   expand (function_expander &e) const override
   {
-    return e.use_exact_insn (CODE_FOR_aarch64_wrffr);
+    return e.use_exact_insn (code_for_set_hssr_state (VNx16BImode));
   }
 };
 
diff --git a/gcc/config/aarch64/aarch64-sve.md b/gcc/config/aarch64/aarch64-sve.md
index 878dadf6f61..eb56bc850b4 100644
--- a/gcc/config/aarch64/aarch64-sve.md
+++ b/gcc/config/aarch64/aarch64-sve.md
@@ -1131,9 +1131,23 @@ (define_insn_and_split "mov<mode>"
 ;; [W1 in the block comment above about FFR handling]
 ;;
 ;; Write to the FFR and start a new FFRT scheduling region.
-(define_insn "aarch64_wrffr"
+(define_insn "@set_hssr_state<mode>"
   [(set (reg:VNx16BI FFR_REGNUM)
-	(match_operand:VNx16BI 0 "aarch64_simd_reg_or_minus_one"))
+	(match_operand:VNx16BI_ONLY 0 "aarch64_simd_reg_or_minus_one"))
+   (set (reg:VNx16BI FFRT_REGNUM)
+	(unspec:VNx16BI [(match_dup 0)] UNSPEC_WRFFR))]
+  "TARGET_SVE && TARGET_NON_STREAMING"
+  {@ [ cons: 0 ]
+     [ Dm      ] setffr
+     [ Upa     ] wrffr\t%0.b
+  }
+  [(set_attr "sve_type" "sve_ffr")]
+)
+
+(define_insn "@set_hssr_state<mode>"
+  [(set (reg:VNx16BI FFR_REGNUM)
+	(subreg:VNx16BI
+	  (match_operand:PRED_HSD 0 "aarch64_simd_reg_or_minus_one") 0))
    (set (reg:VNx16BI FFRT_REGNUM)
 	(unspec:VNx16BI [(match_dup 0)] UNSPEC_WRFFR))]
   "TARGET_SVE && TARGET_NON_STREAMING"
@@ -1176,20 +1190,63 @@ (define_insn "aarch64_copy_ffr_to_ffrt"
 ;; [R2 in the block comment above about FFR handling]
 ;;
 ;; Read the FFR via the FFRT.
-(define_insn "aarch64_rdffr"
-  [(set (match_operand:VNx16BI 0 "register_operand" "=Upa")
+(define_insn "aarch64_rdffr<mode>"
+  [(set (match_operand:VNx16BI_ONLY 0 "register_operand" "=Upa")
 	(reg:VNx16BI FFRT_REGNUM))]
   "TARGET_SVE && TARGET_NON_STREAMING"
   "rdffr\t%0.b"
   [(set_attr "sve_type" "sve_ffr")]
 )
 
+(define_insn "aarch64_rdffr<mode>"
+  [(set (match_operand:PRED_HSD 0 "register_operand" "=Upa")
+	(subreg:PRED_HSD (reg:VNx16BI FFRT_REGNUM) 0))]
+  "TARGET_SVE && TARGET_NON_STREAMING"
+  "rdffr\t%0.b"
+  [(set_attr "sve_type" "sve_ffr")]
+)
+
+;; [R2 in the block comment above about FFR handling]
+;;
+(define_expand "read_hssr_state<mode>"
+  [(set (reg:VNx16BI FFRT_REGNUM) ; copy_ffr_to_ffrt
+	(reg:VNx16BI FFR_REGNUM))
+   (set (match_operand:VNx16BI_ONLY 0 "register_operand" "=Upa") ; rdffr
+	(reg:VNx16BI FFRT_REGNUM))
+   (set (reg:VNx16BI FFRT_REGNUM) ; update ffrt
+	(unspec:VNx16BI [(reg:VNx16BI FFRT_REGNUM)] UNSPEC_UPDATE_FFRT))]
+  "TARGET_SVE && TARGET_NON_STREAMING"
+)
+
+(define_expand "read_hssr_state<mode>"
+  [(set (reg:VNx16BI FFRT_REGNUM) ; copy_ffr_to_ffrt
+	(reg:VNx16BI FFR_REGNUM))
+   (set (match_operand:PRED_HSD 0 "register_operand" "=Upa") ; rdffr
+	(subreg:PRED_HSD (reg:VNx16BI FFRT_REGNUM) 0))
+   (set (reg:VNx16BI FFRT_REGNUM) ; update ffrt
+	(unspec:VNx16BI [(reg:VNx16BI FFRT_REGNUM)] UNSPEC_UPDATE_FFRT))]
+  "TARGET_SVE && TARGET_NON_STREAMING"
+)
+
 ;; Likewise with zero predication.
-(define_insn "aarch64_rdffr_z"
-  [(set (match_operand:VNx16BI 0 "register_operand")
-	(and:VNx16BI
+(define_insn "aarch64_rdffr_z<mode>"
+  [(set (match_operand:VNx16BI_ONLY 0 "register_operand")
+	(and:VNx16BI_ONLY
 	  (reg:VNx16BI FFRT_REGNUM)
-	  (match_operand:VNx16BI 1 "register_operand")))]
+	  (match_operand:VNx16BI_ONLY 1 "register_operand")))]
+  "TARGET_SVE && TARGET_NON_STREAMING"
+  {@ [ cons: =0, 1   ; attrs: pred_clobber ]
+     [ &Upa    , Upa ; yes                 ] rdffr\t%0.b, %1/z
+     [ ?Upa    , 0Upa; yes                 ] ^
+     [ Upa     , Upa ; no                  ] ^
+  }
+  [(set_attr "sve_type" "sve_ffr")]
+)
+(define_insn "aarch64_rdffr_z<mode>"
+  [(set (match_operand:PRED_HSD 0 "register_operand")
+	(and:PRED_HSD
+	  (subreg:PRED_HSD (reg:VNx16BI FFRT_REGNUM) 0)
+	  (match_operand:PRED_HSD 1 "register_operand")))]
   "TARGET_SVE && TARGET_NON_STREAMING"
   {@ [ cons: =0, 1   ; attrs: pred_clobber ]
      [ &Upa    , Upa ; yes                 ] rdffr\t%0.b, %1/z
@@ -1200,17 +1257,36 @@ (define_insn "aarch64_rdffr_z"
 )
 
 ;; Read the FFR to test for a fault, without using the predicate result.
-(define_insn "*aarch64_rdffr_z_ptest"
+(define_insn "*aarch64_rdffr_z_ptest<mode>"
   [(set (reg:CC_NZC CC_REGNUM)
 	(unspec:CC_NZC
 	  [(match_operand:VNx16BI 1 "register_operand")
 	   (match_dup 1)
 	   (match_operand:SI 2 "aarch64_sve_ptrue_flag")
-	   (and:VNx16BI
+	   (and:VNx16BI_ONLY
 	     (reg:VNx16BI FFRT_REGNUM)
 	     (match_dup 1))]
 	  UNSPEC_PTEST))
-   (clobber (match_scratch:VNx16BI 0))]
+   (clobber (match_scratch:VNx16BI_ONLY 0))]
+  "TARGET_SVE && TARGET_NON_STREAMING"
+  {@ [ cons: =0, 1   ; attrs: pred_clobber ]
+     [ &Upa    , Upa ; yes                 ] rdffrs\t%0.b, %1/z
+     [ ?Upa    , 0Upa; yes                 ] ^
+     [ Upa     , Upa ; no                  ] ^
+  }
+  [(set_attr "sve_type" "sve_ffr")]
+)
+(define_insn "*aarch64_rdffr_z_ptest<mode>"
+  [(set (reg:CC_NZC CC_REGNUM)
+	(unspec:CC_NZC
+	  [(match_operand:VNx16BI 1 "register_operand")
+	   (match_operand:PRED_HSD 3)
+	   (match_operand:SI 2 "aarch64_sve_ptrue_flag")
+	   (and:PRED_HSD
+	     (subreg:PRED_HSD (reg:VNx16BI FFRT_REGNUM) 0)
+	     (match_dup 3))]
+	  UNSPEC_PTEST))
+   (clobber (match_scratch:PRED_HSD 0))]
   "TARGET_SVE && TARGET_NON_STREAMING"
   {@ [ cons: =0, 1   ; attrs: pred_clobber ]
      [ &Upa    , Upa ; yes                 ] rdffrs\t%0.b, %1/z
@@ -1221,15 +1297,32 @@ (define_insn "*aarch64_rdffr_z_ptest"
 )
 
 ;; Same for unpredicated RDFFR when tested with a known PTRUE.
-(define_insn "*aarch64_rdffr_ptest"
+(define_insn "*aarch64_rdffr_ptest<mode>"
   [(set (reg:CC_NZC CC_REGNUM)
 	(unspec:CC_NZC
 	  [(match_operand:VNx16BI 1 "register_operand")
-	   (match_dup 1)
+	   (match_operand:VNx16BI_ONLY 2)
 	   (const_int SVE_KNOWN_PTRUE)
 	   (reg:VNx16BI FFRT_REGNUM)]
 	  UNSPEC_PTEST))
-   (clobber (match_scratch:VNx16BI 0))]
+   (clobber (match_scratch:VNx16BI_ONLY 0))]
+  "TARGET_SVE && TARGET_NON_STREAMING"
+  {@ [ cons: =0, 1   ; attrs: pred_clobber ]
+     [ &Upa    , Upa ; yes                 ] rdffrs\t%0.b, %1/z
+     [ ?Upa    , 0Upa; yes                 ] ^
+     [ Upa     , Upa ; no                  ] ^
+  }
+  [(set_attr "sve_type" "sve_ffr")]
+)
+(define_insn "*aarch64_rdffr_ptest<mode>"
+  [(set (reg:CC_NZC CC_REGNUM)
+	(unspec:CC_NZC
+	  [(match_operand:VNx16BI 1 "register_operand")
+	   (match_operand:PRED_HSD 2)
+	   (const_int SVE_KNOWN_PTRUE)
+	   (subreg:PRED_HSD (reg:VNx16BI FFRT_REGNUM) 0)]
+	  UNSPEC_PTEST))
+   (clobber (match_scratch:PRED_HSD 0))]
   "TARGET_SVE && TARGET_NON_STREAMING"
   {@ [ cons: =0, 1   ; attrs: pred_clobber ]
      [ &Upa    , Upa ; yes                 ] rdffrs\t%0.b, %1/z
@@ -1240,18 +1333,18 @@ (define_insn "*aarch64_rdffr_ptest"
 )
 
 ;; Read the FFR with zero predication and test the result.
-(define_insn "*aarch64_rdffr_z_cc"
+(define_insn "*aarch64_rdffr_z_cc<mode>"
   [(set (reg:CC_NZC CC_REGNUM)
 	(unspec:CC_NZC
 	  [(match_operand:VNx16BI 1 "register_operand")
 	   (match_dup 1)
 	   (match_operand:SI 2 "aarch64_sve_ptrue_flag")
-	   (and:VNx16BI
+	   (and:VNx16BI_ONLY
 	     (reg:VNx16BI FFRT_REGNUM)
 	     (match_dup 1))]
 	  UNSPEC_PTEST))
-   (set (match_operand:VNx16BI 0 "register_operand")
-	(and:VNx16BI
+   (set (match_operand:VNx16BI_ONLY 0 "register_operand")
+	(and:VNx16BI_ONLY
 	  (reg:VNx16BI FFRT_REGNUM)
 	  (match_dup 1)))]
   "TARGET_SVE && TARGET_NON_STREAMING"
@@ -1262,9 +1355,31 @@ (define_insn "*aarch64_rdffr_z_cc"
   }
   [(set_attr "sve_type" "sve_ffr")]
 )
+(define_insn "*aarch64_rdffr_z_cc<mode>"
+  [(set (reg:CC_NZC CC_REGNUM)
+	(unspec:CC_NZC
+	  [(match_operand:VNx16BI 1 "register_operand")
+	   (match_operand:PRED_HSD 3)
+	   (match_operand:SI 2 "aarch64_sve_ptrue_flag")
+	   (and:PRED_HSD
+	     (subreg:PRED_HSD (reg:VNx16BI FFRT_REGNUM) 0)
+	     (match_dup 3))]
+	  UNSPEC_PTEST))
+   (set (match_operand:PRED_HSD 0 "register_operand")
+	(and:PRED_HSD
+	  (subreg:PRED_HSD (reg:VNx16BI FFRT_REGNUM) 0)
+	  (match_dup 3)))]
+  "TARGET_SVE && TARGET_NON_STREAMING"
+  {@ [ cons: =0, 1   ; attrs: pred_clobber ]
+     [ &Upa    , Upa ; yes                 ] rdffrs\t%0.b, %1/z
+     [ ?Upa    , 0Upa; yes                 ] ^
+     [ Upa     , Upa ; no                  ] ^
+  }
+  [(set_attr "sve_type" "sve_ffr")]
+)
 
 ;; Same for unpredicated RDFFR when tested with a known PTRUE.
-(define_insn "*aarch64_rdffr_cc"
+(define_insn "*aarch64_rdffr_cc<mode>"
   [(set (reg:CC_NZC CC_REGNUM)
 	(unspec:CC_NZC
 	  [(match_operand:VNx16BI 1 "register_operand")
@@ -1272,7 +1387,7 @@ (define_insn "*aarch64_rdffr_cc"
 	   (const_int SVE_KNOWN_PTRUE)
 	   (reg:VNx16BI FFRT_REGNUM)]
 	  UNSPEC_PTEST))
-   (set (match_operand:VNx16BI 0 "register_operand")
+   (set (match_operand:VNx16BI_ONLY 0 "register_operand")
 	(reg:VNx16BI FFRT_REGNUM))]
   "TARGET_SVE && TARGET_NON_STREAMING"
   {@ [ cons: =0, 1   ; attrs: pred_clobber ]
@@ -1282,6 +1397,24 @@ (define_insn "*aarch64_rdffr_cc"
   }
   [(set_attr "sve_type" "sve_ffr")]
 )
+(define_insn "*aarch64_rdffr_cc<mode>"
+  [(set (reg:CC_NZC CC_REGNUM)
+	(unspec:CC_NZC
+	  [(match_operand:VNx16BI 1 "register_operand")
+	   (match_operand:PRED_HSD 2)
+	   (const_int SVE_KNOWN_PTRUE)
+	   (subreg:PRED_HSD (reg:VNx16BI FFRT_REGNUM) 0)]
+	  UNSPEC_PTEST))
+   (set (match_operand:PRED_HSD 0 "register_operand")
+	(subreg:PRED_HSD (reg:VNx16BI FFRT_REGNUM) 0))]
+  "TARGET_SVE && TARGET_NON_STREAMING"
+  {@ [ cons: =0, 1   ; attrs: pred_clobber ]
+     [ &Upa    , Upa ; yes                 ] rdffrs\t%0.b, %1/z
+     [ ?Upa    , 0Upa; yes                 ] ^
+     [ Upa     , Upa ; no                  ] ^
+  }
+  [(set_attr "sve_type" "sve_ffr")]
+)
 
 ;; [R3 in the block comment above about FFR handling]
 ;;
@@ -1295,6 +1428,27 @@ (define_insn "aarch64_update_ffrt"
   [(set_attr "type" "no_insn")]
 )
 
+/* Test if the last element of the predicate is not true.  */
+(define_expand "detect_hssr_fault<mode>"
+  [(match_operand:QI       0 "register_operand")
+   (match_operand:PRED_ALL 1 "register_operand")]
+  "TARGET_SVE && TARGET_NON_STREAMING"
+  {
+    rtx ptrue = force_reg (VNx16BImode, aarch64_ptrue_all (<data_bytes>));
+    rtx cast_ptrue = gen_lowpart (<MODE>mode, ptrue);
+    rtx ptrue_flag = gen_int_mode (SVE_KNOWN_PTRUE, SImode);
+    emit_insn (gen_aarch64_ptestvnx16bi (ptrue, cast_ptrue, ptrue_flag,
+					 operands[1]));
+
+    rtx cc_reg = gen_rtx_REG (CC_NZCmode, CC_REGNUM);
+    rtx cmp = gen_rtx_fmt_ee (GEU, QImode, cc_reg, const0_rtx);
+
+    emit_insn (gen_aarch64_cstoreqi (operands[0], cmp, cc_reg));
+
+    DONE;
+  }
+)
+
 ;; =========================================================================
 ;; == Loads
 ;; =========================================================================
@@ -1448,18 +1602,37 @@ (define_insn_and_rewrite "*aarch64_load_<ANY_EXTEND:optab>_mov<SVE_HSDI:mode><SV
 ;; -------------------------------------------------------------------------
 
 ;; Contiguous non-extending first-faulting or non-faulting loads.
+;; Also supports sparse vectors via extending loads
 (define_insn "@aarch64_ld<fn>f1<mode>"
-  [(set (match_operand:SVE_FULL 0 "register_operand" "=w")
-	(unspec:SVE_FULL
+  [(set (match_operand:SVE_ALL 0 "register_operand" "=w")
+	(unspec:SVE_ALL
 	  [(match_operand:<VPRED> 2 "register_operand" "Upl")
-	   (match_operand:SVE_FULL 1 "aarch64_sve_ld<fn>f1_operand" "Ut<fn>")
+	   (match_operand:SVE_ALL 1 "aarch64_sve_ld<fn>f1_operand" "Ut<fn>")
+	   (match_operand:SVE_ALL 3 "aarch64_maskload_else_operand")
 	   (reg:VNx16BI FFRT_REGNUM)]
 	  SVE_LDFF1_LDNF1))]
   "TARGET_SVE && TARGET_NON_STREAMING"
-  "ld<fn>f1<Vesize>\t%0.<Vetype>, %2/z, %1"
+  "ld<fn>f1<Vesize>\t%0.<Vctype>, %2/z, %1"
   [(set_attr "sve_type" "sve_load_1reg")]
 )
 
+;; Expose {non,first}-faulting loads to the middle-end.
+(define_expand "mask_ff_hssr_load<mode><vpred>"
+  [(set (match_operand:SVE_ALL 0 "register_operand")
+	(unspec:SVE_ALL
+	  [(match_operand:<VPRED> 2 "register_operand")
+	   (match_operand:SVE_ALL 1 "aarch64_sve_ldff1_operand")
+	   (match_operand:SVE_ALL 3 "aarch64_maskload_else_operand")
+	   (reg:VNx16BI FFRT_REGNUM)]
+	  UNSPEC_LDFF1))]
+  "TARGET_SVE && TARGET_NON_STREAMING"
+  {
+    /* Emit the L2 insn; L1 follows from the template above.
+       See the earlier block comment on FFR handling.  */
+    emit_insn (gen_aarch64_update_ffr_for_load ());
+  }
+)
+
 ;; -------------------------------------------------------------------------
 ;; ---- First-faulting extending contiguous loads
 ;; -------------------------------------------------------------------------
diff --git a/gcc/internal-fn.cc b/gcc/internal-fn.cc
index 0138c6f7ef0..5f5040cb0b9 100644
--- a/gcc/internal-fn.cc
+++ b/gcc/internal-fn.cc
@@ -196,6 +196,9 @@ init_internal_fns ()
 #define crc_direct { 1, -1, true }
 #define reduc_sbool_direct { 0, 0, true }
 #define select_vl_direct { 2, 0, false }
+#define read_hssr_state_direct { -1, -1, false }
+#define set_hssr_state_direct { 0, 0, false }
+#define detect_hssr_fault_direct { 0, 0, false }
 
 const direct_internal_fn_info direct_internal_fn_array[IFN_LAST + 1] = {
 #define DEF_INTERNAL_FN(CODE, FLAGS, FNSPEC) not_direct,
@@ -4315,6 +4318,21 @@ expand_reduc_sbool_optab_fn (internal_fn fn, gcall *stmt, direct_optab optab)
 #define expand_select_vl_optab_fn(FN, STMT, OPTAB) \
   expand_convert_optab_fn (FN, STMT, OPTAB, 3)
 
+#define expand_read_hssr_state_optab_fn(FN, STMT, OPTAB) \
+  expand_direct_optab_fn (FN, STMT, OPTAB, 0)
+
+#define expand_detect_hssr_fault_optab_fn(FN, STMT, OPTAB) \
+  expand_direct_optab_fn (FN, STMT, OPTAB, 1)
+
+static void
+expand_set_hssr_state_optab_fn (internal_fn fn, gcall *stmt,
+				      direct_optab optab)
+{
+  tree_pair types = direct_internal_fn_types (fn, stmt);
+  insn_code icode = direct_optab_handler (optab, TYPE_MODE (types.first));
+  expand_fn_using_insn (stmt, icode, 0, 1);
+}
+
 /* Expanders for optabs that can use expand_convert_optab_fn.  */
 
 #define expand_unary_convert_optab_fn(FN, STMT, OPTAB) \
@@ -4432,6 +4450,9 @@ multi_vector_optab_supported_p (convert_optab optab, tree_pair types,
 #define direct_vec_extract_optab_supported_p convert_optab_supported_p
 #define direct_reduc_sbool_optab_supported_p direct_optab_supported_p
 #define direct_select_vl_optab_supported_p convert_optab_supported_p
+#define direct_read_hssr_state_optab_supported_p direct_optab_supported_p
+#define direct_set_hssr_state_optab_supported_p direct_optab_supported_p
+#define direct_detect_hssr_fault_optab_supported_p direct_optab_supported_p
 
 /* Return the optab used by internal function FN.  */
 
@@ -5081,6 +5102,7 @@ internal_load_fn_p (internal_fn fn)
     case IFN_MASK_LEN_GATHER_LOAD:
     case IFN_LEN_LOAD:
     case IFN_MASK_LEN_LOAD:
+    case IFN_MASK_FF_HSSR_LOAD:
       return true;
 
     default:
@@ -5253,6 +5275,7 @@ internal_fn_else_index (internal_fn fn)
     case IFN_MASK_LEN_LOAD:
     case IFN_MASK_LOAD_LANES:
     case IFN_MASK_LEN_LOAD_LANES:
+    case IFN_MASK_FF_HSSR_LOAD:
       return 3;
 
     case IFN_COND_FMA:
@@ -5293,6 +5316,7 @@ internal_fn_mask_index (internal_fn fn)
     case IFN_MASK_LEN_STORE_LANES:
     case IFN_MASK_LEN_LOAD:
     case IFN_MASK_LEN_STORE:
+    case IFN_MASK_FF_HSSR_LOAD:
       return 2;
 
     case IFN_MASK_LEN_STRIDED_LOAD:
@@ -5361,6 +5385,7 @@ internal_fn_alias_ptr_index (internal_fn fn)
     case IFN_SCATTER_STORE:
     case IFN_MASK_SCATTER_STORE:
     case IFN_MASK_LEN_SCATTER_STORE:
+    case IFN_MASK_FF_HSSR_LOAD:
       return 1;
 
     default:
diff --git a/gcc/internal-fn.def b/gcc/internal-fn.def
index 28ad5b7809d..68397c7bced 100644
--- a/gcc/internal-fn.def
+++ b/gcc/internal-fn.def
@@ -651,6 +651,24 @@ DEF_INTERNAL_FN (BITINTTOFLOAT, ECF_PURE | ECF_LEAF, ". R . ")
 DEF_INTERNAL_OPTAB_FN (BIT_ANDN, ECF_CONST, andn, binary)
 DEF_INTERNAL_OPTAB_FN (BIT_IORN, ECF_CONST, iorn, binary)
 
+/* Hardware Safe Speculative Reads internal functions.  */
+DEF_INTERNAL_OPTAB_FN (MASK_FF_HSSR_LOAD,
+		       0,
+		       mask_ff_hssr_load,
+		       mask_load)
+DEF_INTERNAL_OPTAB_FN (SET_HSSR_STATE,
+		       ECF_NOTHROW,
+		       set_hssr_state,
+		       set_hssr_state)
+DEF_INTERNAL_OPTAB_FN (READ_HSSR_STATE,
+		       ECF_NOTHROW,
+		       read_hssr_state,
+		       read_hssr_state)
+DEF_INTERNAL_OPTAB_FN (DETECT_HSSR_FAULT,
+		       ECF_CONST,
+		       detect_hssr_fault,
+		       detect_hssr_fault)
+
 #undef DEF_INTERNAL_WIDENING_OPTAB_FN
 #undef DEF_INTERNAL_SIGNED_COND_FN
 #undef DEF_INTERNAL_COND_FN
diff --git a/gcc/optabs.def b/gcc/optabs.def
index da58147a513..9b5ca3434be 100644
--- a/gcc/optabs.def
+++ b/gcc/optabs.def
@@ -118,6 +118,9 @@ OPTAB_CD (usdot_prod_optab, "usdot_prod$I$a$b")
 OPTAB_CD (while_ult_optab, "while_ult$a$b")
 OPTAB_CD (select_vl_optab, "select_vl$a$b")
 
+/* Hardware Safe Speculative Reads optabs.  */
+OPTAB_CD (mask_ff_hssr_load_optab, "mask_ff_hssr_load$a$b")
+
 OPTAB_NL(add_optab, "add$P$a3", PLUS, "add", '3', gen_int_fp_fixed_libfunc)
 OPTAB_NX(add_optab, "add$F$a3")
 OPTAB_NX(add_optab, "add$Q$a3")
@@ -238,6 +241,11 @@ OPTAB_D (push_optab, "push$a1")
 OPTAB_D (reload_in_optab, "reload_in$a")
 OPTAB_D (reload_out_optab, "reload_out$a")
 
+/* Hardware Safe Speculative Reads optabs.  */
+OPTAB_D (read_hssr_state_optab, "read_hssr_state$a")
+OPTAB_D (set_hssr_state_optab, "set_hssr_state$a")
+OPTAB_D (detect_hssr_fault_optab, "detect_hssr_fault$a")
+
 OPTAB_DC(cbranch_optab, "cbranch$a4", COMPARE)
 OPTAB_D (tbranch_eq_optab, "tbranch_eq$a3")
 OPTAB_D (tbranch_ne_optab, "tbranch_ne$a3")
diff --git a/gcc/testsuite/gcc.target/aarch64/sve/pfalse-load.c b/gcc/testsuite/gcc.target/aarch64/sve/pfalse-load.c
index a32b636b278..faf433dfb61 100644
--- a/gcc/testsuite/gcc.target/aarch64/sve/pfalse-load.c
+++ b/gcc/testsuite/gcc.target/aarch64/sve/pfalse-load.c
@@ -24,9 +24,8 @@
   T (F##_u32, uint32_t)                                 \
   T (F##_u64, uint64_t)                                 \
 
-ALL_DATA (ldff1)
-ALL_DATA (ldnf1)
 ALL_DATA (ldnt1)
+ALL_DATA (ldnf1)
 
-/* { dg-final { scan-assembler-times {\t.cfi_startproc\n\tmovi?\t[vdz]([0-9]+)\.?(?:[0-9]*[bhsd])?, #?0\n\tret\n} 36 } } */
-/* { dg-final { scan-assembler-times {\t.cfi_startproc\n} 36 } } */
+/* { dg-final { scan-assembler-times {\t.cfi_startproc\n\tmovi?\t[vdz]([0-9]+)\.?(?:[0-9]*[bhsd])?, #?0\n\tret\n} 24 } } */
+/* { dg-final { scan-assembler-times {\t.cfi_startproc\n} 24 } } */
diff --git a/gcc/tree-data-ref.cc b/gcc/tree-data-ref.cc
index 8e9d7871348..e4f04a44e0c 100644
--- a/gcc/tree-data-ref.cc
+++ b/gcc/tree-data-ref.cc
@@ -5893,6 +5893,7 @@ get_references_in_stmt (gimple *stmt, vec<data_ref_loc, va_heap> *references)
 	      break;
 	    }
 	  case IFN_MASK_LOAD:
+	  case IFN_MASK_FF_HSSR_LOAD:
 	  case IFN_MASK_STORE:
 	  break;
 	  case IFN_MASK_CALL:
@@ -5950,6 +5951,7 @@ get_references_in_stmt (gimple *stmt, vec<data_ref_loc, va_heap> *references)
 	switch (gimple_call_internal_fn (stmt))
 	  {
 	  case IFN_MASK_LOAD:
+	  case IFN_MASK_FF_HSSR_LOAD:
 	    if (gimple_call_lhs (stmt) == NULL_TREE)
 	      break;
 	    ref.is_read = true;
diff --git a/gcc/tree-ssa-alias.cc b/gcc/tree-ssa-alias.cc
index 1d10f6dbd97..be47194ef3b 100644
--- a/gcc/tree-ssa-alias.cc
+++ b/gcc/tree-ssa-alias.cc
@@ -2888,6 +2888,7 @@ ref_maybe_used_by_call_p_1 (gcall *call, ao_ref *ref, bool tbaa_p)
       case IFN_MASK_LEN_STORE_LANES:
 	goto process_args;
       case IFN_MASK_LOAD:
+      case IFN_MASK_FF_HSSR_LOAD:
       case IFN_LEN_LOAD:
       case IFN_MASK_LEN_LOAD:
       case IFN_MASK_LOAD_LANES:
diff --git a/gcc/tree-ssa-loop-ivopts.cc b/gcc/tree-ssa-loop-ivopts.cc
index 3c0ce1794c2..881055212d7 100644
--- a/gcc/tree-ssa-loop-ivopts.cc
+++ b/gcc/tree-ssa-loop-ivopts.cc
@@ -2370,6 +2370,7 @@ get_mem_type_for_internal_fn (gcall *call, tree *op_p)
   switch (gimple_call_internal_fn (call))
     {
     case IFN_MASK_LOAD:
+    case IFN_MASK_FF_HSSR_LOAD:
     case IFN_MASK_LOAD_LANES:
     case IFN_MASK_LEN_LOAD_LANES:
     case IFN_LEN_LOAD:
@@ -7567,6 +7568,7 @@ get_alias_ptr_type_for_ptr_address (iv_use *use)
   switch (gimple_call_internal_fn (call))
     {
     case IFN_MASK_LOAD:
+    case IFN_MASK_FF_HSSR_LOAD:
     case IFN_MASK_STORE:
     case IFN_MASK_LOAD_LANES:
     case IFN_MASK_STORE_LANES:
diff --git a/gcc/tree-vect-slp.cc b/gcc/tree-vect-slp.cc
index b4eb9c204cb..3d767a5822f 100644
--- a/gcc/tree-vect-slp.cc
+++ b/gcc/tree-vect-slp.cc
@@ -581,6 +581,9 @@ vect_get_operand_map (const gimple *stmt, bool gather_scatter_p,
 	  case IFN_MASK_LOAD:
 	    return gather_scatter_p ? off_arg2_arg3_map : arg2_arg3_map;
 
+	  case IFN_MASK_FF_HSSR_LOAD:
+	    return arg2_arg3_map;
+
 	  case IFN_GATHER_LOAD:
 	    return arg2_map;
 
-- 
2.43.0
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.