Source file src/cmd/compile/internal/ssagen/intrinsics.go

     1  // Copyright 2024 The Go Authors. All rights reserved.
     2  // Use of this source code is governed by a BSD-style
     3  // license that can be found in the LICENSE file.
     4  
     5  package ssagen
     6  
     7  import (
     8  	"fmt"
     9  	"internal/abi"
    10  	"internal/buildcfg"
    11  
    12  	"cmd/compile/internal/base"
    13  	"cmd/compile/internal/ir"
    14  	"cmd/compile/internal/ssa"
    15  	"cmd/compile/internal/ssa/block"
    16  	"cmd/compile/internal/ssa/ssaconfig"
    17  	"cmd/compile/internal/ssa/ssaop"
    18  	"cmd/compile/internal/typecheck"
    19  	"cmd/compile/internal/types"
    20  	"cmd/internal/sys"
    21  )
    22  
    23  var intrinsics intrinsicBuilders
    24  
    25  // An intrinsicBuilder converts a call node n into an ssa value that
    26  // implements that call as an intrinsic. args is a list of arguments to the func.
    27  type intrinsicBuilder func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value
    28  
    29  type intrinsicKey struct {
    30  	arch *sys.Arch
    31  	pkg  string
    32  	fn   string
    33  }
    34  
    35  // intrinsicBuildConfig specifies the config to use for intrinsic building.
    36  type intrinsicBuildConfig struct {
    37  	instrumenting bool
    38  
    39  	go386     string
    40  	goamd64   int
    41  	goarm     buildcfg.GoarmFeatures
    42  	goarm64   buildcfg.Goarm64Features
    43  	gomips    string
    44  	gomips64  string
    45  	goppc64   int
    46  	goriscv64 int
    47  }
    48  
    49  type intrinsicBuilders map[intrinsicKey]intrinsicBuilder
    50  
    51  // add adds the intrinsic builder b for pkg.fn for the given architecture.
    52  func (ib intrinsicBuilders) add(arch *sys.Arch, pkg, fn string, b intrinsicBuilder) {
    53  	if _, found := ib[intrinsicKey{arch, pkg, fn}]; found {
    54  		panic(fmt.Sprintf("intrinsic already exists for %v.%v on %v", pkg, fn, arch.Name))
    55  	}
    56  	ib[intrinsicKey{arch, pkg, fn}] = b
    57  }
    58  
    59  // addForArchs adds the intrinsic builder b for pkg.fn for the given architectures.
    60  func (ib intrinsicBuilders) addForArchs(pkg, fn string, b intrinsicBuilder, archs ...*sys.Arch) {
    61  	for _, arch := range archs {
    62  		ib.add(arch, pkg, fn, b)
    63  	}
    64  }
    65  
    66  // addForFamilies does the same as addForArchs but operates on architecture families.
    67  func (ib intrinsicBuilders) addForFamilies(pkg, fn string, b intrinsicBuilder, archFamilies ...sys.ArchFamily) {
    68  	for _, arch := range sys.Archs {
    69  		if arch.InFamily(archFamilies...) {
    70  			intrinsics.add(arch, pkg, fn, b)
    71  		}
    72  	}
    73  }
    74  
    75  // alias aliases pkg.fn to targetPkg.targetFn for all architectures in archs
    76  // for which targetPkg.targetFn already exists.
    77  func (ib intrinsicBuilders) alias(pkg, fn, targetPkg, targetFn string, archs ...*sys.Arch) {
    78  	// TODO(jsing): Consider making this work even if the alias is added
    79  	// before the intrinsic.
    80  	aliased := false
    81  	for _, arch := range archs {
    82  		if b := intrinsics.lookup(arch, targetPkg, targetFn); b != nil {
    83  			intrinsics.add(arch, pkg, fn, b)
    84  			aliased = true
    85  		}
    86  	}
    87  	if !aliased {
    88  		panic(fmt.Sprintf("attempted to alias undefined intrinsic: %s.%s", pkg, fn))
    89  	}
    90  }
    91  
    92  // lookup looks up the intrinsic for a pkg.fn on the specified architecture.
    93  func (ib intrinsicBuilders) lookup(arch *sys.Arch, pkg, fn string) intrinsicBuilder {
    94  	return intrinsics[intrinsicKey{arch, pkg, fn}]
    95  }
    96  
    97  func initIntrinsics(cfg *intrinsicBuildConfig) {
    98  	if cfg == nil {
    99  		cfg = &intrinsicBuildConfig{
   100  			instrumenting: base.Flag.Cfg.Instrumenting,
   101  			go386:         buildcfg.GO386,
   102  			goamd64:       buildcfg.GOAMD64,
   103  			goarm:         buildcfg.GOARM,
   104  			goarm64:       buildcfg.GOARM64,
   105  			gomips:        buildcfg.GOMIPS,
   106  			gomips64:      buildcfg.GOMIPS64,
   107  			goppc64:       buildcfg.GOPPC64,
   108  			goriscv64:     buildcfg.GORISCV64,
   109  		}
   110  	}
   111  	intrinsics = intrinsicBuilders{}
   112  
   113  	var p4 []*sys.Arch
   114  	var p8 []*sys.Arch
   115  	var lwatomics []*sys.Arch
   116  	for _, a := range sys.Archs {
   117  		if a.PtrSize == 4 {
   118  			p4 = append(p4, a)
   119  		} else {
   120  			p8 = append(p8, a)
   121  		}
   122  		if a.Family != sys.PPC64 {
   123  			lwatomics = append(lwatomics, a)
   124  		}
   125  	}
   126  	all := sys.Archs[:]
   127  
   128  	add := func(pkg, fn string, b intrinsicBuilder, archs ...*sys.Arch) {
   129  		intrinsics.addForArchs(pkg, fn, b, archs...)
   130  	}
   131  	addF := func(pkg, fn string, b intrinsicBuilder, archFamilies ...sys.ArchFamily) {
   132  		intrinsics.addForFamilies(pkg, fn, b, archFamilies...)
   133  	}
   134  	alias := func(pkg, fn, pkg2, fn2 string, archs ...*sys.Arch) {
   135  		intrinsics.alias(pkg, fn, pkg2, fn2, archs...)
   136  	}
   137  
   138  	/******** runtime ********/
   139  	if !cfg.instrumenting {
   140  		add("runtime", "slicebytetostringtmp",
   141  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   142  				// Compiler frontend optimizations emit OBYTES2STRTMP nodes
   143  				// for the backend instead of slicebytetostringtmp calls
   144  				// when not instrumenting.
   145  				return s.newValue2(ssaop.OpStringMake, n.Type(), args[0], args[1])
   146  			},
   147  			all...)
   148  	}
   149  	addF("internal/runtime/math", "MulUintptr",
   150  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   151  			if s.config.PtrSize == 4 {
   152  				return s.newValue2(ssaop.OpMul32uover, types.NewTuple(types.Types[types.TUINT], types.Types[types.TUINT]), args[0], args[1])
   153  			}
   154  			return s.newValue2(ssaop.OpMul64uover, types.NewTuple(types.Types[types.TUINT], types.Types[types.TUINT]), args[0], args[1])
   155  		},
   156  		sys.AMD64, sys.I386, sys.Loong64, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.ARM64)
   157  	add("runtime", "KeepAlive",
   158  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   159  			data := s.newValue1(ssaop.OpIData, s.f.Config.Types.BytePtr, args[0])
   160  			s.vars[memVar] = s.newValue2(ssaop.OpKeepAlive, types.TypeMem, data, s.mem())
   161  			return nil
   162  		},
   163  		all...)
   164  
   165  	addF("runtime", "publicationBarrier",
   166  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   167  			s.vars[memVar] = s.newValue1(ssaop.OpPubBarrier, types.TypeMem, s.mem())
   168  			return nil
   169  		},
   170  		sys.ARM64, sys.Loong64, sys.MIPS, sys.MIPS64, sys.PPC64, sys.RISCV64)
   171  
   172  	/******** internal/runtime/sys ********/
   173  	add("internal/runtime/sys", "GetCallerPC",
   174  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   175  			return s.newValue0(ssaop.OpGetCallerPC, s.f.Config.Types.Uintptr)
   176  		},
   177  		all...)
   178  
   179  	add("internal/runtime/sys", "GetCallerSP",
   180  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   181  			return s.newValue1(ssaop.OpGetCallerSP, s.f.Config.Types.Uintptr, s.mem())
   182  		},
   183  		all...)
   184  
   185  	add("internal/runtime/sys", "GetClosurePtr",
   186  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   187  			return s.newValue0(ssaop.OpGetClosurePtr, s.f.Config.Types.Uintptr)
   188  		},
   189  		all...)
   190  
   191  	addF("internal/runtime/sys", "Bswap32",
   192  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   193  			return s.newValue1(ssaop.OpBswap32, types.Types[types.TUINT32], args[0])
   194  		},
   195  		sys.AMD64, sys.I386, sys.ARM64, sys.ARM, sys.Loong64, sys.S390X)
   196  	addF("internal/runtime/sys", "Bswap64",
   197  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   198  			return s.newValue1(ssaop.OpBswap64, types.Types[types.TUINT64], args[0])
   199  		},
   200  		sys.AMD64, sys.I386, sys.ARM64, sys.ARM, sys.Loong64, sys.S390X)
   201  
   202  	addF("runtime", "memequal",
   203  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   204  			return s.newValue4(ssaop.OpMemEq, s.f.Config.Types.Bool, args[0], args[1], args[2], s.mem())
   205  		},
   206  		sys.ARM64)
   207  
   208  	if cfg.goppc64 >= 10 {
   209  		// Use only on Power10 as the new byte reverse instructions that Power10 provide
   210  		// make it worthwhile as an intrinsic
   211  		addF("internal/runtime/sys", "Bswap32",
   212  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   213  				return s.newValue1(ssaop.OpBswap32, types.Types[types.TUINT32], args[0])
   214  			},
   215  			sys.PPC64)
   216  		addF("internal/runtime/sys", "Bswap64",
   217  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   218  				return s.newValue1(ssaop.OpBswap64, types.Types[types.TUINT64], args[0])
   219  			},
   220  			sys.PPC64)
   221  	}
   222  
   223  	if cfg.goriscv64 >= 22 {
   224  		addF("internal/runtime/sys", "Bswap32",
   225  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   226  				return s.newValue1(ssaop.OpBswap32, types.Types[types.TUINT32], args[0])
   227  			},
   228  			sys.RISCV64)
   229  		addF("internal/runtime/sys", "Bswap64",
   230  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   231  				return s.newValue1(ssaop.OpBswap64, types.Types[types.TUINT64], args[0])
   232  			},
   233  			sys.RISCV64)
   234  	}
   235  
   236  	/****** Prefetch ******/
   237  	makePrefetchFunc := func(op ssaop.Op) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   238  		return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   239  			s.vars[memVar] = s.newValue2(op, types.TypeMem, args[0], s.mem())
   240  			return nil
   241  		}
   242  	}
   243  
   244  	// Make Prefetch intrinsics for supported platforms
   245  	// On the unsupported platforms stub function will be eliminated
   246  	addF("internal/runtime/sys", "Prefetch", makePrefetchFunc(ssaop.OpPrefetchCache),
   247  		sys.AMD64, sys.ARM64, sys.Loong64, sys.PPC64)
   248  	addF("internal/runtime/sys", "PrefetchStreamed", makePrefetchFunc(ssaop.OpPrefetchCacheStreamed),
   249  		sys.AMD64, sys.ARM64, sys.Loong64, sys.PPC64)
   250  
   251  	/******** internal/runtime/atomic ********/
   252  	type atomicOpEmitter func(s *state, n *ir.CallExpr, args []*ssa.Value, op ssaop.Op, typ types.Kind, needReturn bool)
   253  
   254  	addF("internal/runtime/atomic", "Load",
   255  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   256  			v := s.newValue2(ssaop.OpAtomicLoad32, types.NewTuple(types.Types[types.TUINT32], types.TypeMem), args[0], s.mem())
   257  			s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   258  			return s.newValue1(ssaop.OpSelect0, types.Types[types.TUINT32], v)
   259  		},
   260  		sys.AMD64, sys.ARM64, sys.Loong64, sys.MIPS, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   261  	addF("internal/runtime/atomic", "Load8",
   262  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   263  			v := s.newValue2(ssaop.OpAtomicLoad8, types.NewTuple(types.Types[types.TUINT8], types.TypeMem), args[0], s.mem())
   264  			s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   265  			return s.newValue1(ssaop.OpSelect0, types.Types[types.TUINT8], v)
   266  		},
   267  		sys.AMD64, sys.ARM64, sys.Loong64, sys.MIPS, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   268  	addF("internal/runtime/atomic", "Load64",
   269  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   270  			v := s.newValue2(ssaop.OpAtomicLoad64, types.NewTuple(types.Types[types.TUINT64], types.TypeMem), args[0], s.mem())
   271  			s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   272  			return s.newValue1(ssaop.OpSelect0, types.Types[types.TUINT64], v)
   273  		},
   274  		sys.AMD64, sys.ARM64, sys.Loong64, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   275  	addF("internal/runtime/atomic", "LoadAcq",
   276  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   277  			v := s.newValue2(ssaop.OpAtomicLoadAcq32, types.NewTuple(types.Types[types.TUINT32], types.TypeMem), args[0], s.mem())
   278  			s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   279  			return s.newValue1(ssaop.OpSelect0, types.Types[types.TUINT32], v)
   280  		},
   281  		sys.PPC64)
   282  	addF("internal/runtime/atomic", "LoadAcq64",
   283  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   284  			v := s.newValue2(ssaop.OpAtomicLoadAcq64, types.NewTuple(types.Types[types.TUINT64], types.TypeMem), args[0], s.mem())
   285  			s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   286  			return s.newValue1(ssaop.OpSelect0, types.Types[types.TUINT64], v)
   287  		},
   288  		sys.PPC64)
   289  	addF("internal/runtime/atomic", "Loadp",
   290  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   291  			v := s.newValue2(ssaop.OpAtomicLoadPtr, types.NewTuple(s.f.Config.Types.BytePtr, types.TypeMem), args[0], s.mem())
   292  			s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   293  			return s.newValue1(ssaop.OpSelect0, s.f.Config.Types.BytePtr, v)
   294  		},
   295  		sys.AMD64, sys.ARM64, sys.Loong64, sys.MIPS, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   296  
   297  	addF("internal/runtime/atomic", "Store",
   298  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   299  			s.vars[memVar] = s.newValue3(ssaop.OpAtomicStore32, types.TypeMem, args[0], args[1], s.mem())
   300  			return nil
   301  		},
   302  		sys.AMD64, sys.ARM64, sys.MIPS, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   303  	addF("internal/runtime/atomic", "Store8",
   304  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   305  			s.vars[memVar] = s.newValue3(ssaop.OpAtomicStore8, types.TypeMem, args[0], args[1], s.mem())
   306  			return nil
   307  		},
   308  		sys.AMD64, sys.ARM64, sys.Loong64, sys.MIPS, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   309  	addF("internal/runtime/atomic", "Store64",
   310  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   311  			s.vars[memVar] = s.newValue3(ssaop.OpAtomicStore64, types.TypeMem, args[0], args[1], s.mem())
   312  			return nil
   313  		},
   314  		sys.AMD64, sys.ARM64, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   315  	addF("internal/runtime/atomic", "StorepNoWB",
   316  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   317  			s.vars[memVar] = s.newValue3(ssaop.OpAtomicStorePtrNoWB, types.TypeMem, args[0], args[1], s.mem())
   318  			return nil
   319  		},
   320  		sys.AMD64, sys.ARM64, sys.Loong64, sys.MIPS, sys.MIPS64, sys.RISCV64, sys.S390X)
   321  	addF("internal/runtime/atomic", "StoreRel",
   322  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   323  			s.vars[memVar] = s.newValue3(ssaop.OpAtomicStoreRel32, types.TypeMem, args[0], args[1], s.mem())
   324  			return nil
   325  		},
   326  		sys.PPC64)
   327  	addF("internal/runtime/atomic", "StoreRel64",
   328  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   329  			s.vars[memVar] = s.newValue3(ssaop.OpAtomicStoreRel64, types.TypeMem, args[0], args[1], s.mem())
   330  			return nil
   331  		},
   332  		sys.PPC64)
   333  
   334  	makeAtomicStoreGuardedIntrinsicLoong64 := func(op0, op1 ssaop.Op, typ types.Kind, emit atomicOpEmitter) intrinsicBuilder {
   335  		return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   336  			// Target Atomic feature is identified by dynamic detection
   337  			addr := s.entryNewValue1A(ssaop.OpAddr, types.Types[types.TBOOL].PtrTo(), ir.Syms.Loong64HasDBAR_HINTS, s.sb)
   338  			v := s.load(types.Types[types.TBOOL], addr)
   339  			b := s.endBlock()
   340  			b.Kind = block.BlockIf
   341  			b.SetControl(v)
   342  			bTrue := s.f.NewBlock(block.BlockPlain)
   343  			bFalse := s.f.NewBlock(block.BlockPlain)
   344  			bEnd := s.f.NewBlock(block.BlockPlain)
   345  			b.AddEdgeTo(bTrue)
   346  			b.AddEdgeTo(bFalse)
   347  			b.Likely = ssa.BranchLikely
   348  
   349  			// most loong64 machines support the finer-grained DBAR hints
   350  			s.startBlock(bTrue)
   351  			emit(s, n, args, op0, typ, false)
   352  			s.endBlock().AddEdgeTo(bEnd)
   353  
   354  			// Use original instruction sequence.
   355  			s.startBlock(bFalse)
   356  			emit(s, n, args, op1, typ, false)
   357  			s.endBlock().AddEdgeTo(bEnd)
   358  
   359  			// Merge results.
   360  			s.startBlock(bEnd)
   361  
   362  			return nil
   363  		}
   364  	}
   365  
   366  	atomicStoreEmitterLoong64 := func(s *state, n *ir.CallExpr, args []*ssa.Value, op ssaop.Op, typ types.Kind, needReturn bool) {
   367  		v := s.newValue3(op, types.NewTuple(types.Types[typ], types.TypeMem), args[0], args[1], s.mem())
   368  		s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   369  		if needReturn {
   370  			s.vars[n] = s.newValue1(ssaop.OpSelect0, types.Types[typ], v)
   371  		}
   372  	}
   373  
   374  	addF("internal/runtime/atomic", "Store",
   375  		makeAtomicStoreGuardedIntrinsicLoong64(ssaop.OpAtomicStore32, ssaop.OpAtomicStore32Variant, types.TUINT8, atomicStoreEmitterLoong64),
   376  		sys.Loong64)
   377  	addF("internal/runtime/atomic", "Store64",
   378  		makeAtomicStoreGuardedIntrinsicLoong64(ssaop.OpAtomicStore64, ssaop.OpAtomicStore64Variant, types.TUINT8, atomicStoreEmitterLoong64),
   379  		sys.Loong64)
   380  
   381  	addF("internal/runtime/atomic", "Xchg8",
   382  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   383  			v := s.newValue3(ssaop.OpAtomicExchange8, types.NewTuple(types.Types[types.TUINT8], types.TypeMem), args[0], args[1], s.mem())
   384  			s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   385  			return s.newValue1(ssaop.OpSelect0, types.Types[types.TUINT8], v)
   386  		},
   387  		sys.AMD64, sys.PPC64)
   388  	addF("internal/runtime/atomic", "Xchg",
   389  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   390  			v := s.newValue3(ssaop.OpAtomicExchange32, types.NewTuple(types.Types[types.TUINT32], types.TypeMem), args[0], args[1], s.mem())
   391  			s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   392  			return s.newValue1(ssaop.OpSelect0, types.Types[types.TUINT32], v)
   393  		},
   394  		sys.AMD64, sys.Loong64, sys.MIPS, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   395  	addF("internal/runtime/atomic", "Xchg64",
   396  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   397  			v := s.newValue3(ssaop.OpAtomicExchange64, types.NewTuple(types.Types[types.TUINT64], types.TypeMem), args[0], args[1], s.mem())
   398  			s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   399  			return s.newValue1(ssaop.OpSelect0, types.Types[types.TUINT64], v)
   400  		},
   401  		sys.AMD64, sys.Loong64, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   402  
   403  	makeAtomicGuardedIntrinsicARM64common := func(op0, op1 ssaop.Op, typ types.Kind, emit atomicOpEmitter, needReturn bool) intrinsicBuilder {
   404  
   405  		return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   406  			if cfg.goarm64.LSE {
   407  				emit(s, n, args, op1, typ, needReturn)
   408  			} else {
   409  				// Target Atomic feature is identified by dynamic detection
   410  				addr := s.entryNewValue1A(ssaop.OpAddr, types.Types[types.TBOOL].PtrTo(), ir.Syms.ARM64HasATOMICS, s.sb)
   411  				v := s.load(types.Types[types.TBOOL], addr)
   412  				b := s.endBlock()
   413  				b.Kind = block.BlockIf
   414  				b.SetControl(v)
   415  				bTrue := s.f.NewBlock(block.BlockPlain)
   416  				bFalse := s.f.NewBlock(block.BlockPlain)
   417  				bEnd := s.f.NewBlock(block.BlockPlain)
   418  				b.AddEdgeTo(bTrue)
   419  				b.AddEdgeTo(bFalse)
   420  				b.Likely = ssa.BranchLikely
   421  
   422  				// We have atomic instructions - use it directly.
   423  				s.startBlock(bTrue)
   424  				emit(s, n, args, op1, typ, needReturn)
   425  				s.endBlock().AddEdgeTo(bEnd)
   426  
   427  				// Use original instruction sequence.
   428  				s.startBlock(bFalse)
   429  				emit(s, n, args, op0, typ, needReturn)
   430  				s.endBlock().AddEdgeTo(bEnd)
   431  
   432  				// Merge results.
   433  				s.startBlock(bEnd)
   434  			}
   435  			if needReturn {
   436  				return s.variable(n, types.Types[typ])
   437  			} else {
   438  				return nil
   439  			}
   440  		}
   441  	}
   442  	makeAtomicGuardedIntrinsicARM64 := func(op0, op1 ssaop.Op, typ types.Kind, emit atomicOpEmitter) intrinsicBuilder {
   443  		return makeAtomicGuardedIntrinsicARM64common(op0, op1, typ, emit, true)
   444  	}
   445  	makeAtomicGuardedIntrinsicARM64old := func(op0, op1 ssaop.Op, typ types.Kind, emit atomicOpEmitter) intrinsicBuilder {
   446  		return makeAtomicGuardedIntrinsicARM64common(op0, op1, typ, emit, false)
   447  	}
   448  
   449  	atomicEmitterARM64 := func(s *state, n *ir.CallExpr, args []*ssa.Value, op ssaop.Op, typ types.Kind, needReturn bool) {
   450  		v := s.newValue3(op, types.NewTuple(types.Types[typ], types.TypeMem), args[0], args[1], s.mem())
   451  		s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   452  		if needReturn {
   453  			s.vars[n] = s.newValue1(ssaop.OpSelect0, types.Types[typ], v)
   454  		}
   455  	}
   456  	addF("internal/runtime/atomic", "Xchg8",
   457  		makeAtomicGuardedIntrinsicARM64(ssaop.OpAtomicExchange8, ssaop.OpAtomicExchange8Variant, types.TUINT8, atomicEmitterARM64),
   458  		sys.ARM64)
   459  	addF("internal/runtime/atomic", "Xchg",
   460  		makeAtomicGuardedIntrinsicARM64(ssaop.OpAtomicExchange32, ssaop.OpAtomicExchange32Variant, types.TUINT32, atomicEmitterARM64),
   461  		sys.ARM64)
   462  	addF("internal/runtime/atomic", "Xchg64",
   463  		makeAtomicGuardedIntrinsicARM64(ssaop.OpAtomicExchange64, ssaop.OpAtomicExchange64Variant, types.TUINT64, atomicEmitterARM64),
   464  		sys.ARM64)
   465  
   466  	makeAtomicXchg8GuardedIntrinsicLoong64 := func(op ssaop.Op) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   467  		return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   468  			addr := s.entryNewValue1A(ssaop.OpAddr, types.Types[types.TBOOL].PtrTo(), ir.Syms.Loong64HasLAM_BH, s.sb)
   469  			v := s.load(types.Types[types.TBOOL], addr)
   470  			b := s.endBlock()
   471  			b.Kind = block.BlockIf
   472  			b.SetControl(v)
   473  			bTrue := s.f.NewBlock(block.BlockPlain)
   474  			bFalse := s.f.NewBlock(block.BlockPlain)
   475  			bEnd := s.f.NewBlock(block.BlockPlain)
   476  			b.AddEdgeTo(bTrue)
   477  			b.AddEdgeTo(bFalse)
   478  			b.Likely = ssa.BranchLikely // most loong64 machines support the amswapdb.b
   479  
   480  			// We have the intrinsic - use it directly.
   481  			s.startBlock(bTrue)
   482  			s.vars[n] = s.newValue3(op, types.NewTuple(types.Types[types.TUINT8], types.TypeMem), args[0], args[1], s.mem())
   483  			s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, s.vars[n])
   484  			s.vars[n] = s.newValue1(ssaop.OpSelect0, types.Types[types.TUINT8], s.vars[n])
   485  			s.endBlock().AddEdgeTo(bEnd)
   486  
   487  			// Call the pure Go version.
   488  			s.startBlock(bFalse)
   489  			s.vars[n] = s.callResult(n, callNormal) // types.Types[TUINT8]
   490  			s.endBlock().AddEdgeTo(bEnd)
   491  
   492  			// Merge results.
   493  			s.startBlock(bEnd)
   494  			return s.variable(n, types.Types[types.TUINT8])
   495  		}
   496  	}
   497  	addF("internal/runtime/atomic", "Xchg8",
   498  		makeAtomicXchg8GuardedIntrinsicLoong64(ssaop.OpAtomicExchange8Variant),
   499  		sys.Loong64)
   500  
   501  	addF("internal/runtime/atomic", "Xadd",
   502  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   503  			v := s.newValue3(ssaop.OpAtomicAdd32, types.NewTuple(types.Types[types.TUINT32], types.TypeMem), args[0], args[1], s.mem())
   504  			s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   505  			return s.newValue1(ssaop.OpSelect0, types.Types[types.TUINT32], v)
   506  		},
   507  		sys.AMD64, sys.Loong64, sys.MIPS, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   508  	addF("internal/runtime/atomic", "Xadd64",
   509  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   510  			v := s.newValue3(ssaop.OpAtomicAdd64, types.NewTuple(types.Types[types.TUINT64], types.TypeMem), args[0], args[1], s.mem())
   511  			s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   512  			return s.newValue1(ssaop.OpSelect0, types.Types[types.TUINT64], v)
   513  		},
   514  		sys.AMD64, sys.Loong64, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   515  
   516  	addF("internal/runtime/atomic", "Xadd",
   517  		makeAtomicGuardedIntrinsicARM64(ssaop.OpAtomicAdd32, ssaop.OpAtomicAdd32Variant, types.TUINT32, atomicEmitterARM64),
   518  		sys.ARM64)
   519  	addF("internal/runtime/atomic", "Xadd64",
   520  		makeAtomicGuardedIntrinsicARM64(ssaop.OpAtomicAdd64, ssaop.OpAtomicAdd64Variant, types.TUINT64, atomicEmitterARM64),
   521  		sys.ARM64)
   522  
   523  	addF("internal/runtime/atomic", "Cas",
   524  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   525  			v := s.newValue4(ssaop.OpAtomicCompareAndSwap32, types.NewTuple(types.Types[types.TBOOL], types.TypeMem), args[0], args[1], args[2], s.mem())
   526  			s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   527  			return s.newValue1(ssaop.OpSelect0, types.Types[types.TBOOL], v)
   528  		},
   529  		sys.AMD64, sys.MIPS, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   530  	addF("internal/runtime/atomic", "Cas64",
   531  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   532  			v := s.newValue4(ssaop.OpAtomicCompareAndSwap64, types.NewTuple(types.Types[types.TBOOL], types.TypeMem), args[0], args[1], args[2], s.mem())
   533  			s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   534  			return s.newValue1(ssaop.OpSelect0, types.Types[types.TBOOL], v)
   535  		},
   536  		sys.AMD64, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   537  	addF("internal/runtime/atomic", "CasRel",
   538  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   539  			v := s.newValue4(ssaop.OpAtomicCompareAndSwap32, types.NewTuple(types.Types[types.TBOOL], types.TypeMem), args[0], args[1], args[2], s.mem())
   540  			s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   541  			return s.newValue1(ssaop.OpSelect0, types.Types[types.TBOOL], v)
   542  		},
   543  		sys.PPC64)
   544  
   545  	atomicCasEmitterARM64 := func(s *state, n *ir.CallExpr, args []*ssa.Value, op ssaop.Op, typ types.Kind, needReturn bool) {
   546  		v := s.newValue4(op, types.NewTuple(types.Types[types.TBOOL], types.TypeMem), args[0], args[1], args[2], s.mem())
   547  		s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   548  		if needReturn {
   549  			s.vars[n] = s.newValue1(ssaop.OpSelect0, types.Types[typ], v)
   550  		}
   551  	}
   552  
   553  	addF("internal/runtime/atomic", "Cas",
   554  		makeAtomicGuardedIntrinsicARM64(ssaop.OpAtomicCompareAndSwap32, ssaop.OpAtomicCompareAndSwap32Variant, types.TBOOL, atomicCasEmitterARM64),
   555  		sys.ARM64)
   556  	addF("internal/runtime/atomic", "Cas64",
   557  		makeAtomicGuardedIntrinsicARM64(ssaop.OpAtomicCompareAndSwap64, ssaop.OpAtomicCompareAndSwap64Variant, types.TBOOL, atomicCasEmitterARM64),
   558  		sys.ARM64)
   559  
   560  	atomicCasEmitterLoong64 := func(s *state, n *ir.CallExpr, args []*ssa.Value, op ssaop.Op, typ types.Kind, needReturn bool) {
   561  		v := s.newValue4(op, types.NewTuple(types.Types[types.TBOOL], types.TypeMem), args[0], args[1], args[2], s.mem())
   562  		s.vars[memVar] = s.newValue1(ssaop.OpSelect1, types.TypeMem, v)
   563  		if needReturn {
   564  			s.vars[n] = s.newValue1(ssaop.OpSelect0, types.Types[typ], v)
   565  		}
   566  	}
   567  
   568  	makeAtomicCasGuardedIntrinsicLoong64 := func(op0, op1 ssaop.Op, emit atomicOpEmitter) intrinsicBuilder {
   569  		return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   570  			// Target Atomic feature is identified by dynamic detection
   571  			addr := s.entryNewValue1A(ssaop.OpAddr, types.Types[types.TBOOL].PtrTo(), ir.Syms.Loong64HasLAMCAS, s.sb)
   572  			v := s.load(types.Types[types.TBOOL], addr)
   573  			b := s.endBlock()
   574  			b.Kind = block.BlockIf
   575  			b.SetControl(v)
   576  			bTrue := s.f.NewBlock(block.BlockPlain)
   577  			bFalse := s.f.NewBlock(block.BlockPlain)
   578  			bEnd := s.f.NewBlock(block.BlockPlain)
   579  			b.AddEdgeTo(bTrue)
   580  			b.AddEdgeTo(bFalse)
   581  			b.Likely = ssa.BranchLikely
   582  
   583  			// We have atomic instructions - use it directly.
   584  			s.startBlock(bTrue)
   585  			emit(s, n, args, op1, types.TBOOL, true)
   586  			s.endBlock().AddEdgeTo(bEnd)
   587  
   588  			// Use original instruction sequence.
   589  			s.startBlock(bFalse)
   590  			emit(s, n, args, op0, types.TBOOL, true)
   591  			s.endBlock().AddEdgeTo(bEnd)
   592  
   593  			// Merge results.
   594  			s.startBlock(bEnd)
   595  
   596  			return s.variable(n, types.Types[types.TBOOL])
   597  		}
   598  	}
   599  
   600  	addF("internal/runtime/atomic", "Cas",
   601  		makeAtomicCasGuardedIntrinsicLoong64(ssaop.OpAtomicCompareAndSwap32, ssaop.OpAtomicCompareAndSwap32Variant, atomicCasEmitterLoong64),
   602  		sys.Loong64)
   603  	addF("internal/runtime/atomic", "Cas64",
   604  		makeAtomicCasGuardedIntrinsicLoong64(ssaop.OpAtomicCompareAndSwap64, ssaop.OpAtomicCompareAndSwap64Variant, atomicCasEmitterLoong64),
   605  		sys.Loong64)
   606  
   607  	// Old-style atomic logical operation API (all supported archs except arm64).
   608  	addF("internal/runtime/atomic", "And8",
   609  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   610  			s.vars[memVar] = s.newValue3(ssaop.OpAtomicAnd8, types.TypeMem, args[0], args[1], s.mem())
   611  			return nil
   612  		},
   613  		sys.AMD64, sys.Loong64, sys.MIPS, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   614  	addF("internal/runtime/atomic", "And",
   615  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   616  			s.vars[memVar] = s.newValue3(ssaop.OpAtomicAnd32, types.TypeMem, args[0], args[1], s.mem())
   617  			return nil
   618  		},
   619  		sys.AMD64, sys.Loong64, sys.MIPS, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   620  	addF("internal/runtime/atomic", "Or8",
   621  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   622  			s.vars[memVar] = s.newValue3(ssaop.OpAtomicOr8, types.TypeMem, args[0], args[1], s.mem())
   623  			return nil
   624  		},
   625  		sys.AMD64, sys.Loong64, sys.MIPS, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   626  	addF("internal/runtime/atomic", "Or",
   627  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   628  			s.vars[memVar] = s.newValue3(ssaop.OpAtomicOr32, types.TypeMem, args[0], args[1], s.mem())
   629  			return nil
   630  		},
   631  		sys.AMD64, sys.Loong64, sys.MIPS, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X)
   632  
   633  	// arm64 always uses the new-style atomic logical operations, for both the
   634  	// old and new style API.
   635  	addF("internal/runtime/atomic", "And8",
   636  		makeAtomicGuardedIntrinsicARM64old(ssaop.OpAtomicAnd8value, ssaop.OpAtomicAnd8valueVariant, types.TUINT8, atomicEmitterARM64),
   637  		sys.ARM64)
   638  	addF("internal/runtime/atomic", "Or8",
   639  		makeAtomicGuardedIntrinsicARM64old(ssaop.OpAtomicOr8value, ssaop.OpAtomicOr8valueVariant, types.TUINT8, atomicEmitterARM64),
   640  		sys.ARM64)
   641  	addF("internal/runtime/atomic", "And64",
   642  		makeAtomicGuardedIntrinsicARM64(ssaop.OpAtomicAnd64value, ssaop.OpAtomicAnd64valueVariant, types.TUINT64, atomicEmitterARM64),
   643  		sys.ARM64)
   644  	addF("internal/runtime/atomic", "And32",
   645  		makeAtomicGuardedIntrinsicARM64(ssaop.OpAtomicAnd32value, ssaop.OpAtomicAnd32valueVariant, types.TUINT32, atomicEmitterARM64),
   646  		sys.ARM64)
   647  	addF("internal/runtime/atomic", "And",
   648  		makeAtomicGuardedIntrinsicARM64old(ssaop.OpAtomicAnd32value, ssaop.OpAtomicAnd32valueVariant, types.TUINT32, atomicEmitterARM64),
   649  		sys.ARM64)
   650  	addF("internal/runtime/atomic", "Or64",
   651  		makeAtomicGuardedIntrinsicARM64(ssaop.OpAtomicOr64value, ssaop.OpAtomicOr64valueVariant, types.TUINT64, atomicEmitterARM64),
   652  		sys.ARM64)
   653  	addF("internal/runtime/atomic", "Or32",
   654  		makeAtomicGuardedIntrinsicARM64(ssaop.OpAtomicOr32value, ssaop.OpAtomicOr32valueVariant, types.TUINT32, atomicEmitterARM64),
   655  		sys.ARM64)
   656  	addF("internal/runtime/atomic", "Or",
   657  		makeAtomicGuardedIntrinsicARM64old(ssaop.OpAtomicOr32value, ssaop.OpAtomicOr32valueVariant, types.TUINT32, atomicEmitterARM64),
   658  		sys.ARM64)
   659  
   660  	// New-style atomic logical operations, which return the old memory value.
   661  	addF("internal/runtime/atomic", "And64",
   662  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   663  			v := s.newValue3(ssaop.OpAtomicAnd64value, types.NewTuple(types.Types[types.TUINT64], types.TypeMem), args[0], args[1], s.mem())
   664  			p0, p1 := s.split(v)
   665  			s.vars[memVar] = p1
   666  			return p0
   667  		},
   668  		sys.AMD64, sys.Loong64, sys.RISCV64)
   669  	addF("internal/runtime/atomic", "And32",
   670  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   671  			v := s.newValue3(ssaop.OpAtomicAnd32value, types.NewTuple(types.Types[types.TUINT32], types.TypeMem), args[0], args[1], s.mem())
   672  			p0, p1 := s.split(v)
   673  			s.vars[memVar] = p1
   674  			return p0
   675  		},
   676  		sys.AMD64, sys.Loong64, sys.RISCV64)
   677  	addF("internal/runtime/atomic", "Or64",
   678  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   679  			v := s.newValue3(ssaop.OpAtomicOr64value, types.NewTuple(types.Types[types.TUINT64], types.TypeMem), args[0], args[1], s.mem())
   680  			p0, p1 := s.split(v)
   681  			s.vars[memVar] = p1
   682  			return p0
   683  		},
   684  		sys.AMD64, sys.Loong64, sys.RISCV64)
   685  	addF("internal/runtime/atomic", "Or32",
   686  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   687  			v := s.newValue3(ssaop.OpAtomicOr32value, types.NewTuple(types.Types[types.TUINT32], types.TypeMem), args[0], args[1], s.mem())
   688  			p0, p1 := s.split(v)
   689  			s.vars[memVar] = p1
   690  			return p0
   691  		},
   692  		sys.AMD64, sys.Loong64, sys.RISCV64)
   693  
   694  	// Aliases for atomic load operations
   695  	alias("internal/runtime/atomic", "Loadint32", "internal/runtime/atomic", "Load", all...)
   696  	alias("internal/runtime/atomic", "Loadint64", "internal/runtime/atomic", "Load64", all...)
   697  	alias("internal/runtime/atomic", "Loaduintptr", "internal/runtime/atomic", "Load", p4...)
   698  	alias("internal/runtime/atomic", "Loaduintptr", "internal/runtime/atomic", "Load64", p8...)
   699  	alias("internal/runtime/atomic", "Loaduint", "internal/runtime/atomic", "Load", p4...)
   700  	alias("internal/runtime/atomic", "Loaduint", "internal/runtime/atomic", "Load64", p8...)
   701  	alias("internal/runtime/atomic", "LoadAcq", "internal/runtime/atomic", "Load", lwatomics...)
   702  	alias("internal/runtime/atomic", "LoadAcq64", "internal/runtime/atomic", "Load64", lwatomics...)
   703  	alias("internal/runtime/atomic", "LoadAcquintptr", "internal/runtime/atomic", "LoadAcq", p4...)
   704  	alias("internal/runtime/atomic", "LoadAcquintptr", "internal/runtime/atomic", "LoadAcq64", p8...)
   705  
   706  	// Aliases for atomic store operations
   707  	alias("internal/runtime/atomic", "Storeint32", "internal/runtime/atomic", "Store", all...)
   708  	alias("internal/runtime/atomic", "Storeint64", "internal/runtime/atomic", "Store64", all...)
   709  	alias("internal/runtime/atomic", "Storeuintptr", "internal/runtime/atomic", "Store", p4...)
   710  	alias("internal/runtime/atomic", "Storeuintptr", "internal/runtime/atomic", "Store64", p8...)
   711  	alias("internal/runtime/atomic", "StoreRel", "internal/runtime/atomic", "Store", lwatomics...)
   712  	alias("internal/runtime/atomic", "StoreRel64", "internal/runtime/atomic", "Store64", lwatomics...)
   713  	alias("internal/runtime/atomic", "StoreReluintptr", "internal/runtime/atomic", "StoreRel", p4...)
   714  	alias("internal/runtime/atomic", "StoreReluintptr", "internal/runtime/atomic", "StoreRel64", p8...)
   715  
   716  	// Aliases for atomic swap operations
   717  	alias("internal/runtime/atomic", "Xchgint32", "internal/runtime/atomic", "Xchg", all...)
   718  	alias("internal/runtime/atomic", "Xchgint64", "internal/runtime/atomic", "Xchg64", all...)
   719  	alias("internal/runtime/atomic", "Xchguintptr", "internal/runtime/atomic", "Xchg", p4...)
   720  	alias("internal/runtime/atomic", "Xchguintptr", "internal/runtime/atomic", "Xchg64", p8...)
   721  
   722  	// Aliases for atomic add operations
   723  	alias("internal/runtime/atomic", "Xaddint32", "internal/runtime/atomic", "Xadd", all...)
   724  	alias("internal/runtime/atomic", "Xaddint64", "internal/runtime/atomic", "Xadd64", all...)
   725  	alias("internal/runtime/atomic", "Xadduintptr", "internal/runtime/atomic", "Xadd", p4...)
   726  	alias("internal/runtime/atomic", "Xadduintptr", "internal/runtime/atomic", "Xadd64", p8...)
   727  
   728  	// Aliases for atomic CAS operations
   729  	alias("internal/runtime/atomic", "Casint32", "internal/runtime/atomic", "Cas", all...)
   730  	alias("internal/runtime/atomic", "Casint64", "internal/runtime/atomic", "Cas64", all...)
   731  	alias("internal/runtime/atomic", "Casuintptr", "internal/runtime/atomic", "Cas", p4...)
   732  	alias("internal/runtime/atomic", "Casuintptr", "internal/runtime/atomic", "Cas64", p8...)
   733  	alias("internal/runtime/atomic", "Casp1", "internal/runtime/atomic", "Cas", p4...)
   734  	alias("internal/runtime/atomic", "Casp1", "internal/runtime/atomic", "Cas64", p8...)
   735  	alias("internal/runtime/atomic", "CasRel", "internal/runtime/atomic", "Cas", lwatomics...)
   736  
   737  	// Aliases for atomic And/Or operations
   738  	alias("internal/runtime/atomic", "Anduintptr", "internal/runtime/atomic", "And64", sys.ArchARM64, sys.ArchLoong64)
   739  	alias("internal/runtime/atomic", "Oruintptr", "internal/runtime/atomic", "Or64", sys.ArchARM64, sys.ArchLoong64)
   740  
   741  	/******** math ********/
   742  	addF("math", "sqrt",
   743  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   744  			return s.newValue1(ssaop.OpSqrt, types.Types[types.TFLOAT64], args[0])
   745  		},
   746  		sys.I386, sys.AMD64, sys.ARM, sys.ARM64, sys.Loong64, sys.MIPS, sys.MIPS64, sys.PPC64, sys.RISCV64, sys.S390X, sys.Wasm)
   747  	addF("math", "Trunc",
   748  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   749  			return s.newValue1(ssaop.OpTrunc, types.Types[types.TFLOAT64], args[0])
   750  		},
   751  		sys.ARM64, sys.PPC64, sys.S390X, sys.Wasm)
   752  	addF("math", "Ceil",
   753  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   754  			return s.newValue1(ssaop.OpCeil, types.Types[types.TFLOAT64], args[0])
   755  		},
   756  		sys.ARM64, sys.PPC64, sys.S390X, sys.Wasm)
   757  	addF("math", "Floor",
   758  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   759  			return s.newValue1(ssaop.OpFloor, types.Types[types.TFLOAT64], args[0])
   760  		},
   761  		sys.ARM64, sys.PPC64, sys.S390X, sys.Wasm)
   762  	addF("math", "Round",
   763  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   764  			return s.newValue1(ssaop.OpRound, types.Types[types.TFLOAT64], args[0])
   765  		},
   766  		sys.ARM64, sys.PPC64, sys.S390X)
   767  	addF("math", "RoundToEven",
   768  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   769  			return s.newValue1(ssaop.OpRoundToEven, types.Types[types.TFLOAT64], args[0])
   770  		},
   771  		sys.ARM64, sys.S390X, sys.Wasm)
   772  	addF("math", "Abs",
   773  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   774  			return s.newValue1(ssaop.OpAbs, types.Types[types.TFLOAT64], args[0])
   775  		},
   776  		sys.ARM64, sys.ARM, sys.Loong64, sys.PPC64, sys.RISCV64, sys.Wasm, sys.MIPS, sys.MIPS64)
   777  	addF("math", "Copysign",
   778  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   779  			return s.newValue2(ssaop.OpCopysign, types.Types[types.TFLOAT64], args[0], args[1])
   780  		},
   781  		sys.Loong64, sys.PPC64, sys.RISCV64, sys.Wasm)
   782  	addF("math", "FMA",
   783  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   784  			return s.newValue3(ssaop.OpFMA, types.Types[types.TFLOAT64], args[0], args[1], args[2])
   785  		},
   786  		sys.ARM64, sys.Loong64, sys.PPC64, sys.RISCV64, sys.S390X)
   787  	addF("math", "FMA",
   788  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   789  			if cfg.goamd64 >= 3 {
   790  				return s.newValue3(ssaop.OpFMA, types.Types[types.TFLOAT64], args[0], args[1], args[2])
   791  			}
   792  
   793  			v := s.entryNewValue0A(ssaop.OpHasCPUFeature, types.Types[types.TBOOL], ir.Syms.X86HasFMA)
   794  			b := s.endBlock()
   795  			b.Kind = block.BlockIf
   796  			b.SetControl(v)
   797  			bTrue := s.f.NewBlock(block.BlockPlain)
   798  			bFalse := s.f.NewBlock(block.BlockPlain)
   799  			bEnd := s.f.NewBlock(block.BlockPlain)
   800  			b.AddEdgeTo(bTrue)
   801  			b.AddEdgeTo(bFalse)
   802  			b.Likely = ssa.BranchLikely // >= haswell cpus are common
   803  
   804  			// We have the intrinsic - use it directly.
   805  			s.startBlock(bTrue)
   806  			s.vars[n] = s.newValue3(ssaop.OpFMA, types.Types[types.TFLOAT64], args[0], args[1], args[2])
   807  			s.endBlock().AddEdgeTo(bEnd)
   808  
   809  			// Call the pure Go version.
   810  			s.startBlock(bFalse)
   811  			s.vars[n] = s.callResult(n, callNormal) // types.Types[TFLOAT64]
   812  			s.endBlock().AddEdgeTo(bEnd)
   813  
   814  			// Merge results.
   815  			s.startBlock(bEnd)
   816  			return s.variable(n, types.Types[types.TFLOAT64])
   817  		},
   818  		sys.AMD64)
   819  	addF("math", "FMA",
   820  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   821  			addr := s.entryNewValue1A(ssaop.OpAddr, types.Types[types.TBOOL].PtrTo(), ir.Syms.ARMHasVFPv4, s.sb)
   822  			v := s.load(types.Types[types.TBOOL], addr)
   823  			b := s.endBlock()
   824  			b.Kind = block.BlockIf
   825  			b.SetControl(v)
   826  			bTrue := s.f.NewBlock(block.BlockPlain)
   827  			bFalse := s.f.NewBlock(block.BlockPlain)
   828  			bEnd := s.f.NewBlock(block.BlockPlain)
   829  			b.AddEdgeTo(bTrue)
   830  			b.AddEdgeTo(bFalse)
   831  			b.Likely = ssa.BranchLikely
   832  
   833  			// We have the intrinsic - use it directly.
   834  			s.startBlock(bTrue)
   835  			s.vars[n] = s.newValue3(ssaop.OpFMA, types.Types[types.TFLOAT64], args[0], args[1], args[2])
   836  			s.endBlock().AddEdgeTo(bEnd)
   837  
   838  			// Call the pure Go version.
   839  			s.startBlock(bFalse)
   840  			s.vars[n] = s.callResult(n, callNormal) // types.Types[TFLOAT64]
   841  			s.endBlock().AddEdgeTo(bEnd)
   842  
   843  			// Merge results.
   844  			s.startBlock(bEnd)
   845  			return s.variable(n, types.Types[types.TFLOAT64])
   846  		},
   847  		sys.ARM)
   848  
   849  	makeRoundAMD64 := func(op ssaop.Op) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   850  		return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   851  			if cfg.goamd64 >= 2 {
   852  				return s.newValue1(op, types.Types[types.TFLOAT64], args[0])
   853  			}
   854  
   855  			v := s.entryNewValue0A(ssaop.OpHasCPUFeature, types.Types[types.TBOOL], ir.Syms.X86HasSSE41)
   856  			b := s.endBlock()
   857  			b.Kind = block.BlockIf
   858  			b.SetControl(v)
   859  			bTrue := s.f.NewBlock(block.BlockPlain)
   860  			bFalse := s.f.NewBlock(block.BlockPlain)
   861  			bEnd := s.f.NewBlock(block.BlockPlain)
   862  			b.AddEdgeTo(bTrue)
   863  			b.AddEdgeTo(bFalse)
   864  			b.Likely = ssa.BranchLikely // most machines have sse4.1 nowadays
   865  
   866  			// We have the intrinsic - use it directly.
   867  			s.startBlock(bTrue)
   868  			s.vars[n] = s.newValue1(op, types.Types[types.TFLOAT64], args[0])
   869  			s.endBlock().AddEdgeTo(bEnd)
   870  
   871  			// Call the pure Go version.
   872  			s.startBlock(bFalse)
   873  			s.vars[n] = s.callResult(n, callNormal) // types.Types[TFLOAT64]
   874  			s.endBlock().AddEdgeTo(bEnd)
   875  
   876  			// Merge results.
   877  			s.startBlock(bEnd)
   878  			return s.variable(n, types.Types[types.TFLOAT64])
   879  		}
   880  	}
   881  	addF("math", "RoundToEven",
   882  		makeRoundAMD64(ssaop.OpRoundToEven),
   883  		sys.AMD64)
   884  	addF("math", "Floor",
   885  		makeRoundAMD64(ssaop.OpFloor),
   886  		sys.AMD64)
   887  	addF("math", "Ceil",
   888  		makeRoundAMD64(ssaop.OpCeil),
   889  		sys.AMD64)
   890  	addF("math", "Trunc",
   891  		makeRoundAMD64(ssaop.OpTrunc),
   892  		sys.AMD64)
   893  
   894  	makeRoundLoong64 := func(op ssaop.Op) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   895  		return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   896  			addr := s.entryNewValue1A(ssaop.OpAddr, types.Types[types.TBOOL].PtrTo(), ir.Syms.Loong64HasLSX, s.sb)
   897  			v := s.load(types.Types[types.TBOOL], addr)
   898  			b := s.endBlock()
   899  			b.Kind = block.BlockIf
   900  			b.SetControl(v)
   901  			bTrue := s.f.NewBlock(block.BlockPlain)
   902  			bFalse := s.f.NewBlock(block.BlockPlain)
   903  			bEnd := s.f.NewBlock(block.BlockPlain)
   904  			b.AddEdgeTo(bTrue)
   905  			b.AddEdgeTo(bFalse)
   906  			b.Likely = ssa.BranchLikely // most loong64 machines support the LSX
   907  
   908  			// We have the intrinsic - use it directly.
   909  			s.startBlock(bTrue)
   910  			s.vars[n] = s.newValue1(op, types.Types[types.TFLOAT64], args[0])
   911  			s.endBlock().AddEdgeTo(bEnd)
   912  
   913  			// Call the pure Go version.
   914  			s.startBlock(bFalse)
   915  			s.vars[n] = s.callResult(n, callNormal) // types.Types[TFLOAT64]
   916  			s.endBlock().AddEdgeTo(bEnd)
   917  
   918  			// Merge results.
   919  			s.startBlock(bEnd)
   920  			return s.variable(n, types.Types[types.TFLOAT64])
   921  		}
   922  	}
   923  	addF("math", "RoundToEven",
   924  		makeRoundLoong64(ssaop.OpRoundToEven),
   925  		sys.Loong64)
   926  	addF("math", "Floor",
   927  		makeRoundLoong64(ssaop.OpFloor),
   928  		sys.Loong64)
   929  	addF("math", "Ceil",
   930  		makeRoundLoong64(ssaop.OpCeil),
   931  		sys.Loong64)
   932  	addF("math", "Trunc",
   933  		makeRoundLoong64(ssaop.OpTrunc),
   934  		sys.Loong64)
   935  
   936  	/******** math/bits ********/
   937  	addF("math/bits", "TrailingZeros64",
   938  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   939  			return s.newValue1(ssaop.OpCtz64, types.Types[types.TINT], args[0])
   940  		},
   941  		sys.AMD64, sys.ARM64, sys.ARM, sys.Loong64, sys.S390X, sys.MIPS, sys.PPC64, sys.Wasm)
   942  	addF("math/bits", "TrailingZeros64",
   943  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   944  			lo := s.newValue1(ssaop.OpInt64Lo, types.Types[types.TUINT32], args[0])
   945  			hi := s.newValue1(ssaop.OpInt64Hi, types.Types[types.TUINT32], args[0])
   946  			return s.newValue2(ssaop.OpCtz64On32, types.Types[types.TINT], lo, hi)
   947  		},
   948  		sys.I386)
   949  	addF("math/bits", "TrailingZeros32",
   950  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   951  			return s.newValue1(ssaop.OpCtz32, types.Types[types.TINT], args[0])
   952  		},
   953  		sys.AMD64, sys.I386, sys.ARM64, sys.ARM, sys.Loong64, sys.S390X, sys.MIPS, sys.PPC64, sys.Wasm)
   954  	addF("math/bits", "TrailingZeros16",
   955  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   956  			return s.newValue1(ssaop.OpCtz16, types.Types[types.TINT], args[0])
   957  		},
   958  		sys.AMD64, sys.ARM, sys.ARM64, sys.I386, sys.MIPS, sys.Loong64, sys.PPC64, sys.S390X, sys.Wasm)
   959  	addF("math/bits", "TrailingZeros8",
   960  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   961  			return s.newValue1(ssaop.OpCtz8, types.Types[types.TINT], args[0])
   962  		},
   963  		sys.AMD64, sys.ARM, sys.ARM64, sys.I386, sys.MIPS, sys.Loong64, sys.PPC64, sys.S390X, sys.Wasm)
   964  
   965  	if cfg.goriscv64 >= 22 {
   966  		addF("math/bits", "TrailingZeros64",
   967  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   968  				return s.newValue1(ssaop.OpCtz64, types.Types[types.TINT], args[0])
   969  			},
   970  			sys.RISCV64)
   971  		addF("math/bits", "TrailingZeros32",
   972  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   973  				return s.newValue1(ssaop.OpCtz32, types.Types[types.TINT], args[0])
   974  			},
   975  			sys.RISCV64)
   976  		addF("math/bits", "TrailingZeros16",
   977  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   978  				return s.newValue1(ssaop.OpCtz16, types.Types[types.TINT], args[0])
   979  			},
   980  			sys.RISCV64)
   981  		addF("math/bits", "TrailingZeros8",
   982  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   983  				return s.newValue1(ssaop.OpCtz8, types.Types[types.TINT], args[0])
   984  			},
   985  			sys.RISCV64)
   986  	}
   987  
   988  	// ReverseBytes inlines correctly, no need to intrinsify it.
   989  	alias("math/bits", "ReverseBytes64", "internal/runtime/sys", "Bswap64", all...)
   990  	alias("math/bits", "ReverseBytes32", "internal/runtime/sys", "Bswap32", all...)
   991  	// Nothing special is needed for targets where ReverseBytes16 lowers to a rotate
   992  	addF("math/bits", "ReverseBytes16",
   993  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
   994  			return s.newValue1(ssaop.OpBswap16, types.Types[types.TUINT16], args[0])
   995  		},
   996  		sys.Loong64)
   997  	if cfg.goppc64 >= 10 {
   998  		// On Power10, 16-bit rotate is not available so use BRH instruction
   999  		addF("math/bits", "ReverseBytes16",
  1000  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1001  				return s.newValue1(ssaop.OpBswap16, types.Types[types.TUINT], args[0])
  1002  			},
  1003  			sys.PPC64)
  1004  	}
  1005  	if cfg.goriscv64 >= 22 {
  1006  		addF("math/bits", "ReverseBytes16",
  1007  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1008  				return s.newValue1(ssaop.OpBswap16, types.Types[types.TUINT16], args[0])
  1009  			},
  1010  			sys.RISCV64)
  1011  	}
  1012  
  1013  	addF("math/bits", "Len64",
  1014  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1015  			return s.newValue1(ssaop.OpBitLen64, types.Types[types.TINT], args[0])
  1016  		},
  1017  		sys.AMD64, sys.ARM, sys.ARM64, sys.Loong64, sys.MIPS, sys.PPC64, sys.S390X, sys.Wasm)
  1018  	addF("math/bits", "Len32",
  1019  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1020  			return s.newValue1(ssaop.OpBitLen32, types.Types[types.TINT], args[0])
  1021  		},
  1022  		sys.AMD64, sys.ARM, sys.ARM64, sys.Loong64, sys.MIPS, sys.PPC64, sys.S390X, sys.Wasm)
  1023  	addF("math/bits", "Len16",
  1024  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1025  			return s.newValue1(ssaop.OpBitLen16, types.Types[types.TINT], args[0])
  1026  		},
  1027  		sys.AMD64, sys.ARM, sys.ARM64, sys.Loong64, sys.MIPS, sys.PPC64, sys.S390X, sys.Wasm)
  1028  	addF("math/bits", "Len8",
  1029  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1030  			return s.newValue1(ssaop.OpBitLen8, types.Types[types.TINT], args[0])
  1031  		},
  1032  		sys.AMD64, sys.ARM, sys.ARM64, sys.Loong64, sys.MIPS, sys.PPC64, sys.S390X, sys.Wasm)
  1033  
  1034  	if cfg.goriscv64 >= 22 {
  1035  		addF("math/bits", "Len64",
  1036  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1037  				return s.newValue1(ssaop.OpBitLen64, types.Types[types.TINT], args[0])
  1038  			},
  1039  			sys.RISCV64)
  1040  		addF("math/bits", "Len32",
  1041  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1042  				return s.newValue1(ssaop.OpBitLen32, types.Types[types.TINT], args[0])
  1043  			},
  1044  			sys.RISCV64)
  1045  		addF("math/bits", "Len16",
  1046  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1047  				return s.newValue1(ssaop.OpBitLen16, types.Types[types.TINT], args[0])
  1048  			},
  1049  			sys.RISCV64)
  1050  		addF("math/bits", "Len8",
  1051  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1052  				return s.newValue1(ssaop.OpBitLen8, types.Types[types.TINT], args[0])
  1053  			},
  1054  			sys.RISCV64)
  1055  	}
  1056  
  1057  	alias("math/bits", "Len", "math/bits", "Len64", p8...)
  1058  	alias("math/bits", "Len", "math/bits", "Len32", p4...)
  1059  
  1060  	// LeadingZeros is handled because it trivially calls Len.
  1061  	addF("math/bits", "Reverse64",
  1062  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1063  			return s.newValue1(ssaop.OpBitRev64, types.Types[types.TUINT64], args[0])
  1064  		},
  1065  		sys.ARM64, sys.Loong64)
  1066  	addF("math/bits", "Reverse32",
  1067  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1068  			return s.newValue1(ssaop.OpBitRev32, types.Types[types.TUINT32], args[0])
  1069  		},
  1070  		sys.ARM64, sys.Loong64)
  1071  	addF("math/bits", "Reverse16",
  1072  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1073  			return s.newValue1(ssaop.OpBitRev16, types.Types[types.TUINT16], args[0])
  1074  		},
  1075  		sys.ARM64, sys.Loong64)
  1076  	addF("math/bits", "Reverse8",
  1077  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1078  			return s.newValue1(ssaop.OpBitRev8, types.Types[types.TUINT8], args[0])
  1079  		},
  1080  		sys.ARM64, sys.Loong64)
  1081  	addF("math/bits", "Reverse",
  1082  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1083  			return s.newValue1(ssaop.OpBitRev64, types.Types[types.TUINT], args[0])
  1084  		},
  1085  		sys.ARM64, sys.Loong64)
  1086  	addF("math/bits", "RotateLeft8",
  1087  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1088  			return s.newValue2(ssaop.OpRotateLeft8, types.Types[types.TUINT8], args[0], args[1])
  1089  		},
  1090  		sys.AMD64, sys.RISCV64)
  1091  	addF("math/bits", "RotateLeft16",
  1092  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1093  			return s.newValue2(ssaop.OpRotateLeft16, types.Types[types.TUINT16], args[0], args[1])
  1094  		},
  1095  		sys.AMD64, sys.RISCV64)
  1096  	addF("math/bits", "RotateLeft32",
  1097  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1098  			return s.newValue2(ssaop.OpRotateLeft32, types.Types[types.TUINT32], args[0], args[1])
  1099  		},
  1100  		sys.AMD64, sys.ARM, sys.ARM64, sys.Loong64, sys.PPC64, sys.RISCV64, sys.S390X, sys.Wasm)
  1101  	addF("math/bits", "RotateLeft64",
  1102  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1103  			return s.newValue2(ssaop.OpRotateLeft64, types.Types[types.TUINT64], args[0], args[1])
  1104  		},
  1105  		sys.AMD64, sys.ARM64, sys.Loong64, sys.PPC64, sys.RISCV64, sys.S390X, sys.Wasm)
  1106  	alias("math/bits", "RotateLeft", "math/bits", "RotateLeft64", p8...)
  1107  
  1108  	makeOnesCountAMD64 := func(op ssaop.Op) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1109  		return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1110  			if cfg.goamd64 >= 2 {
  1111  				return s.newValue1(op, types.Types[types.TINT], args[0])
  1112  			}
  1113  
  1114  			v := s.entryNewValue0A(ssaop.OpHasCPUFeature, types.Types[types.TBOOL], ir.Syms.X86HasPOPCNT)
  1115  			b := s.endBlock()
  1116  			b.Kind = block.BlockIf
  1117  			b.SetControl(v)
  1118  			bTrue := s.f.NewBlock(block.BlockPlain)
  1119  			bFalse := s.f.NewBlock(block.BlockPlain)
  1120  			bEnd := s.f.NewBlock(block.BlockPlain)
  1121  			b.AddEdgeTo(bTrue)
  1122  			b.AddEdgeTo(bFalse)
  1123  			b.Likely = ssa.BranchLikely // most machines have popcnt nowadays
  1124  
  1125  			// We have the intrinsic - use it directly.
  1126  			s.startBlock(bTrue)
  1127  			s.vars[n] = s.newValue1(op, types.Types[types.TINT], args[0])
  1128  			s.endBlock().AddEdgeTo(bEnd)
  1129  
  1130  			// Call the pure Go version.
  1131  			s.startBlock(bFalse)
  1132  			s.vars[n] = s.callResult(n, callNormal) // types.Types[TINT]
  1133  			s.endBlock().AddEdgeTo(bEnd)
  1134  
  1135  			// Merge results.
  1136  			s.startBlock(bEnd)
  1137  			return s.variable(n, types.Types[types.TINT])
  1138  		}
  1139  	}
  1140  
  1141  	makeOnesCountLoong64 := func(op ssaop.Op) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1142  		return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1143  			addr := s.entryNewValue1A(ssaop.OpAddr, types.Types[types.TBOOL].PtrTo(), ir.Syms.Loong64HasLSX, s.sb)
  1144  			v := s.load(types.Types[types.TBOOL], addr)
  1145  			b := s.endBlock()
  1146  			b.Kind = block.BlockIf
  1147  			b.SetControl(v)
  1148  			bTrue := s.f.NewBlock(block.BlockPlain)
  1149  			bFalse := s.f.NewBlock(block.BlockPlain)
  1150  			bEnd := s.f.NewBlock(block.BlockPlain)
  1151  			b.AddEdgeTo(bTrue)
  1152  			b.AddEdgeTo(bFalse)
  1153  			b.Likely = ssa.BranchLikely // most loong64 machines support the LSX
  1154  
  1155  			// We have the intrinsic - use it directly.
  1156  			s.startBlock(bTrue)
  1157  			s.vars[n] = s.newValue1(op, types.Types[types.TINT], args[0])
  1158  			s.endBlock().AddEdgeTo(bEnd)
  1159  
  1160  			// Call the pure Go version.
  1161  			s.startBlock(bFalse)
  1162  			s.vars[n] = s.callResult(n, callNormal) // types.Types[TINT]
  1163  			s.endBlock().AddEdgeTo(bEnd)
  1164  
  1165  			// Merge results.
  1166  			s.startBlock(bEnd)
  1167  			return s.variable(n, types.Types[types.TINT])
  1168  		}
  1169  	}
  1170  
  1171  	makeOnesCountRISCV64 := func(op ssaop.Op) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1172  		return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1173  			if cfg.goriscv64 >= 22 {
  1174  				return s.newValue1(op, types.Types[types.TINT], args[0])
  1175  			}
  1176  
  1177  			addr := s.entryNewValue1A(ssaop.OpAddr, types.Types[types.TBOOL].PtrTo(), ir.Syms.RISCV64HasZbb, s.sb)
  1178  			v := s.load(types.Types[types.TBOOL], addr)
  1179  			b := s.endBlock()
  1180  			b.Kind = block.BlockIf
  1181  			b.SetControl(v)
  1182  			bTrue := s.f.NewBlock(block.BlockPlain)
  1183  			bFalse := s.f.NewBlock(block.BlockPlain)
  1184  			bEnd := s.f.NewBlock(block.BlockPlain)
  1185  			b.AddEdgeTo(bTrue)
  1186  			b.AddEdgeTo(bFalse)
  1187  			b.Likely = ssa.BranchLikely // Majority of RISC-V support Zbb.
  1188  
  1189  			// We have the intrinsic - use it directly.
  1190  			s.startBlock(bTrue)
  1191  			s.vars[n] = s.newValue1(op, types.Types[types.TINT], args[0])
  1192  			s.endBlock().AddEdgeTo(bEnd)
  1193  
  1194  			// Call the pure Go version.
  1195  			s.startBlock(bFalse)
  1196  			s.vars[n] = s.callResult(n, callNormal) // types.Types[TINT]
  1197  			s.endBlock().AddEdgeTo(bEnd)
  1198  
  1199  			// Merge results.
  1200  			s.startBlock(bEnd)
  1201  			return s.variable(n, types.Types[types.TINT])
  1202  		}
  1203  	}
  1204  
  1205  	addF("math/bits", "OnesCount64",
  1206  		makeOnesCountAMD64(ssaop.OpPopCount64),
  1207  		sys.AMD64)
  1208  	addF("math/bits", "OnesCount64",
  1209  		makeOnesCountLoong64(ssaop.OpPopCount64),
  1210  		sys.Loong64)
  1211  	addF("math/bits", "OnesCount64",
  1212  		makeOnesCountRISCV64(ssaop.OpPopCount64),
  1213  		sys.RISCV64)
  1214  	addF("math/bits", "OnesCount64",
  1215  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1216  			return s.newValue1(ssaop.OpPopCount64, types.Types[types.TINT], args[0])
  1217  		},
  1218  		sys.PPC64, sys.ARM64, sys.S390X, sys.Wasm)
  1219  	addF("math/bits", "OnesCount32",
  1220  		makeOnesCountAMD64(ssaop.OpPopCount32),
  1221  		sys.AMD64)
  1222  	addF("math/bits", "OnesCount32",
  1223  		makeOnesCountLoong64(ssaop.OpPopCount32),
  1224  		sys.Loong64)
  1225  	addF("math/bits", "OnesCount32",
  1226  		makeOnesCountRISCV64(ssaop.OpPopCount32),
  1227  		sys.RISCV64)
  1228  	addF("math/bits", "OnesCount32",
  1229  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1230  			return s.newValue1(ssaop.OpPopCount32, types.Types[types.TINT], args[0])
  1231  		},
  1232  		sys.PPC64, sys.ARM64, sys.S390X, sys.Wasm)
  1233  	addF("math/bits", "OnesCount16",
  1234  		makeOnesCountAMD64(ssaop.OpPopCount16),
  1235  		sys.AMD64)
  1236  	addF("math/bits", "OnesCount16",
  1237  		makeOnesCountLoong64(ssaop.OpPopCount16),
  1238  		sys.Loong64)
  1239  	addF("math/bits", "OnesCount16",
  1240  		makeOnesCountRISCV64(ssaop.OpPopCount16),
  1241  		sys.RISCV64)
  1242  	addF("math/bits", "OnesCount16",
  1243  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1244  			return s.newValue1(ssaop.OpPopCount16, types.Types[types.TINT], args[0])
  1245  		},
  1246  		sys.ARM64, sys.S390X, sys.PPC64, sys.Wasm)
  1247  	addF("math/bits", "OnesCount8",
  1248  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1249  			return s.newValue1(ssaop.OpPopCount8, types.Types[types.TINT], args[0])
  1250  		},
  1251  		sys.S390X, sys.PPC64, sys.Wasm)
  1252  
  1253  	if cfg.goriscv64 >= 22 {
  1254  		addF("math/bits", "OnesCount8",
  1255  			makeOnesCountRISCV64(ssaop.OpPopCount8),
  1256  			sys.RISCV64)
  1257  	}
  1258  
  1259  	alias("math/bits", "OnesCount", "math/bits", "OnesCount64", p8...)
  1260  
  1261  	add("math/bits", "Mul64",
  1262  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1263  			return s.newValue2(ssaop.OpMul64uhilo, types.NewTuple(types.Types[types.TUINT64], types.Types[types.TUINT64]), args[0], args[1])
  1264  		},
  1265  		all...)
  1266  	alias("math/bits", "Mul", "math/bits", "Mul64", p8...)
  1267  	addF("math/bits", "Add64",
  1268  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1269  			return s.newValue3(ssaop.OpAdd64carry, types.NewTuple(types.Types[types.TUINT64], types.Types[types.TUINT64]), args[0], args[1], args[2])
  1270  		},
  1271  		sys.AMD64, sys.ARM64, sys.PPC64, sys.S390X, sys.RISCV64, sys.Loong64, sys.MIPS64)
  1272  	alias("math/bits", "Add", "math/bits", "Add64", p8...)
  1273  	alias("internal/runtime/math", "Add64", "math/bits", "Add64", all...)
  1274  	addF("math/bits", "Sub64",
  1275  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1276  			return s.newValue3(ssaop.OpSub64borrow, types.NewTuple(types.Types[types.TUINT64], types.Types[types.TUINT64]), args[0], args[1], args[2])
  1277  		},
  1278  		sys.AMD64, sys.ARM64, sys.PPC64, sys.S390X, sys.RISCV64, sys.Loong64, sys.MIPS64)
  1279  	alias("math/bits", "Sub", "math/bits", "Sub64", p8...)
  1280  	addF("math/bits", "Div64",
  1281  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1282  			// check for divide-by-zero/overflow and panic with appropriate message
  1283  			cmpZero := s.newValue2(s.ssaOp(ir.ONE, types.Types[types.TUINT64]), types.Types[types.TBOOL], args[2], s.zeroVal(types.Types[types.TUINT64]))
  1284  			s.check(cmpZero, ir.Syms.Panicdivide)
  1285  			cmpOverflow := s.newValue2(s.ssaOp(ir.OLT, types.Types[types.TUINT64]), types.Types[types.TBOOL], args[0], args[2])
  1286  			s.check(cmpOverflow, ir.Syms.Panicoverflow)
  1287  			return s.newValue3(ssaop.OpDiv128u, types.NewTuple(types.Types[types.TUINT64], types.Types[types.TUINT64]), args[0], args[1], args[2])
  1288  		},
  1289  		sys.AMD64)
  1290  	alias("math/bits", "Div", "math/bits", "Div64", sys.ArchAMD64)
  1291  
  1292  	alias("internal/runtime/sys", "TrailingZeros8", "math/bits", "TrailingZeros8", all...)
  1293  	alias("internal/runtime/sys", "TrailingZeros32", "math/bits", "TrailingZeros32", all...)
  1294  	alias("internal/runtime/sys", "TrailingZeros64", "math/bits", "TrailingZeros64", all...)
  1295  	alias("internal/runtime/sys", "Len8", "math/bits", "Len8", all...)
  1296  	alias("internal/runtime/sys", "Len64", "math/bits", "Len64", all...)
  1297  	alias("internal/runtime/sys", "OnesCount64", "math/bits", "OnesCount64", all...)
  1298  
  1299  	/******** sync/atomic ********/
  1300  
  1301  	// Note: these are disabled by flag_race in findIntrinsic below.
  1302  	alias("sync/atomic", "LoadInt32", "internal/runtime/atomic", "Load", all...)
  1303  	alias("sync/atomic", "LoadInt64", "internal/runtime/atomic", "Load64", all...)
  1304  	alias("sync/atomic", "LoadPointer", "internal/runtime/atomic", "Loadp", all...)
  1305  	alias("sync/atomic", "LoadUint32", "internal/runtime/atomic", "Load", all...)
  1306  	alias("sync/atomic", "LoadUint64", "internal/runtime/atomic", "Load64", all...)
  1307  	alias("sync/atomic", "LoadUintptr", "internal/runtime/atomic", "Load", p4...)
  1308  	alias("sync/atomic", "LoadUintptr", "internal/runtime/atomic", "Load64", p8...)
  1309  
  1310  	alias("sync/atomic", "StoreInt32", "internal/runtime/atomic", "Store", all...)
  1311  	alias("sync/atomic", "StoreInt64", "internal/runtime/atomic", "Store64", all...)
  1312  	// Note: not StorePointer, that needs a write barrier.  Same below for {CompareAnd}Swap.
  1313  	alias("sync/atomic", "StoreUint32", "internal/runtime/atomic", "Store", all...)
  1314  	alias("sync/atomic", "StoreUint64", "internal/runtime/atomic", "Store64", all...)
  1315  	alias("sync/atomic", "StoreUintptr", "internal/runtime/atomic", "Store", p4...)
  1316  	alias("sync/atomic", "StoreUintptr", "internal/runtime/atomic", "Store64", p8...)
  1317  
  1318  	alias("sync/atomic", "SwapInt32", "internal/runtime/atomic", "Xchg", all...)
  1319  	alias("sync/atomic", "SwapInt64", "internal/runtime/atomic", "Xchg64", all...)
  1320  	alias("sync/atomic", "SwapUint32", "internal/runtime/atomic", "Xchg", all...)
  1321  	alias("sync/atomic", "SwapUint64", "internal/runtime/atomic", "Xchg64", all...)
  1322  	alias("sync/atomic", "SwapUintptr", "internal/runtime/atomic", "Xchg", p4...)
  1323  	alias("sync/atomic", "SwapUintptr", "internal/runtime/atomic", "Xchg64", p8...)
  1324  
  1325  	alias("sync/atomic", "CompareAndSwapInt32", "internal/runtime/atomic", "Cas", all...)
  1326  	alias("sync/atomic", "CompareAndSwapInt64", "internal/runtime/atomic", "Cas64", all...)
  1327  	alias("sync/atomic", "CompareAndSwapUint32", "internal/runtime/atomic", "Cas", all...)
  1328  	alias("sync/atomic", "CompareAndSwapUint64", "internal/runtime/atomic", "Cas64", all...)
  1329  	alias("sync/atomic", "CompareAndSwapUintptr", "internal/runtime/atomic", "Cas", p4...)
  1330  	alias("sync/atomic", "CompareAndSwapUintptr", "internal/runtime/atomic", "Cas64", p8...)
  1331  
  1332  	alias("sync/atomic", "AddInt32", "internal/runtime/atomic", "Xadd", all...)
  1333  	alias("sync/atomic", "AddInt64", "internal/runtime/atomic", "Xadd64", all...)
  1334  	alias("sync/atomic", "AddUint32", "internal/runtime/atomic", "Xadd", all...)
  1335  	alias("sync/atomic", "AddUint64", "internal/runtime/atomic", "Xadd64", all...)
  1336  	alias("sync/atomic", "AddUintptr", "internal/runtime/atomic", "Xadd", p4...)
  1337  	alias("sync/atomic", "AddUintptr", "internal/runtime/atomic", "Xadd64", p8...)
  1338  
  1339  	alias("sync/atomic", "AndInt32", "internal/runtime/atomic", "And32", sys.ArchARM64, sys.ArchAMD64, sys.ArchLoong64, sys.ArchRISCV64)
  1340  	alias("sync/atomic", "AndUint32", "internal/runtime/atomic", "And32", sys.ArchARM64, sys.ArchAMD64, sys.ArchLoong64, sys.ArchRISCV64)
  1341  	alias("sync/atomic", "AndInt64", "internal/runtime/atomic", "And64", sys.ArchARM64, sys.ArchAMD64, sys.ArchLoong64, sys.ArchRISCV64)
  1342  	alias("sync/atomic", "AndUint64", "internal/runtime/atomic", "And64", sys.ArchARM64, sys.ArchAMD64, sys.ArchLoong64, sys.ArchRISCV64)
  1343  	alias("sync/atomic", "AndUintptr", "internal/runtime/atomic", "And64", sys.ArchARM64, sys.ArchAMD64, sys.ArchLoong64, sys.ArchRISCV64)
  1344  	alias("sync/atomic", "OrInt32", "internal/runtime/atomic", "Or32", sys.ArchARM64, sys.ArchAMD64, sys.ArchLoong64, sys.ArchRISCV64)
  1345  	alias("sync/atomic", "OrUint32", "internal/runtime/atomic", "Or32", sys.ArchARM64, sys.ArchAMD64, sys.ArchLoong64, sys.ArchRISCV64)
  1346  	alias("sync/atomic", "OrInt64", "internal/runtime/atomic", "Or64", sys.ArchARM64, sys.ArchAMD64, sys.ArchLoong64, sys.ArchRISCV64)
  1347  	alias("sync/atomic", "OrUint64", "internal/runtime/atomic", "Or64", sys.ArchARM64, sys.ArchAMD64, sys.ArchLoong64, sys.ArchRISCV64)
  1348  	alias("sync/atomic", "OrUintptr", "internal/runtime/atomic", "Or64", sys.ArchARM64, sys.ArchAMD64, sys.ArchLoong64, sys.ArchRISCV64)
  1349  
  1350  	/******** math/big ********/
  1351  	alias("math/big", "mulWW", "math/bits", "Mul64", p8...)
  1352  
  1353  	/******** internal/runtime/maps ********/
  1354  
  1355  	// Important: The intrinsic implementations below return a packed
  1356  	// bitset, while the portable Go implementation uses an unpacked
  1357  	// representation (one bit set in each byte).
  1358  	//
  1359  	// Thus we must replace most bitset methods with implementations that
  1360  	// work with the packed representation.
  1361  	//
  1362  	// TODO(prattmic): The bitset implementations don't use SIMD, so they
  1363  	// could be handled with build tags (though that would break
  1364  	// -d=ssa/intrinsics/off=1).
  1365  
  1366  	// With a packed representation we no longer need to shift the result
  1367  	// of TrailingZeros64.
  1368  	alias("internal/runtime/maps", "bitsetFirst", "internal/runtime/sys", "TrailingZeros64", sys.ArchAMD64)
  1369  
  1370  	addF("internal/runtime/maps", "bitsetRemoveBelow",
  1371  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1372  			b := args[0]
  1373  			i := args[1]
  1374  
  1375  			// Clear the lower i bits in b.
  1376  			//
  1377  			// out = b &^ ((1 << i) - 1)
  1378  
  1379  			one := s.constInt64(types.Types[types.TUINT64], 1)
  1380  
  1381  			mask := s.newValue2(ssaop.OpLsh8x8, types.Types[types.TUINT64], one, i)
  1382  			mask = s.newValue2(ssaop.OpSub64, types.Types[types.TUINT64], mask, one)
  1383  			mask = s.newValue1(ssaop.OpCom64, types.Types[types.TUINT64], mask)
  1384  
  1385  			return s.newValue2(ssaop.OpAnd64, types.Types[types.TUINT64], b, mask)
  1386  		},
  1387  		sys.AMD64)
  1388  
  1389  	addF("internal/runtime/maps", "bitsetLowestSet",
  1390  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1391  			b := args[0]
  1392  
  1393  			// Test the lowest bit in b.
  1394  			//
  1395  			// out = (b & 1) == 1
  1396  
  1397  			one := s.constInt64(types.Types[types.TUINT64], 1)
  1398  			and := s.newValue2(ssaop.OpAnd64, types.Types[types.TUINT64], b, one)
  1399  			return s.newValue2(ssaop.OpEq64, types.Types[types.TBOOL], and, one)
  1400  		},
  1401  		sys.AMD64)
  1402  
  1403  	addF("internal/runtime/maps", "bitsetShiftOutLowest",
  1404  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1405  			b := args[0]
  1406  
  1407  			// Right shift out the lowest bit in b.
  1408  			//
  1409  			// out = b >> 1
  1410  
  1411  			one := s.constInt64(types.Types[types.TUINT64], 1)
  1412  			return s.newValue2(ssaop.OpRsh64Ux64, types.Types[types.TUINT64], b, one)
  1413  		},
  1414  		sys.AMD64)
  1415  
  1416  	addF("internal/runtime/maps", "ctrlGroupMatchH2",
  1417  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1418  			g := args[0]
  1419  			h := args[1]
  1420  
  1421  			// Explicit copies to fp registers. See
  1422  			// https://go.dev/issue/70451.
  1423  			gfp := s.newValue1(ssaop.OpAMD64MOVQi2f, types.TypeInt128, g)
  1424  			hfp := s.newValue1(ssaop.OpAMD64MOVQi2f, types.TypeInt128, h)
  1425  
  1426  			// Broadcast h2 into each byte of a word.
  1427  			var broadcast *ssa.Value
  1428  			if buildcfg.GOAMD64 >= 4 {
  1429  				// VPBROADCASTB saves 1 instruction vs PSHUFB
  1430  				// because the input can come from a GP
  1431  				// register, while PSHUFB requires moving into
  1432  				// an FP register first.
  1433  				//
  1434  				// Nominally PSHUFB would require a second
  1435  				// additional instruction to load the control
  1436  				// mask into a FP register. But broadcast uses
  1437  				// a control mask of 0, and the register ABI
  1438  				// already defines X15 as a zero register.
  1439  				broadcast = s.newValue1(ssaop.OpAMD64VPBROADCASTB, types.TypeInt128, h) // use gp copy of h
  1440  			} else if buildcfg.GOAMD64 >= 2 {
  1441  				// PSHUFB performs a byte broadcast when given
  1442  				// a control input of 0.
  1443  				broadcast = s.newValue1(ssaop.OpAMD64PSHUFBbroadcast, types.TypeInt128, hfp)
  1444  			} else {
  1445  				// No direct byte broadcast. First we must
  1446  				// duplicate the lower byte and then do a
  1447  				// 16-bit broadcast.
  1448  
  1449  				// "Unpack" h2 with itself. This duplicates the
  1450  				// input, resulting in h2 in the lower two
  1451  				// bytes.
  1452  				unpack := s.newValue2(ssaop.OpAMD64PUNPCKLBW, types.TypeInt128, hfp, hfp)
  1453  
  1454  				// Copy the lower 16-bits of unpack into every
  1455  				// 16-bit slot in the lower 64-bits of the
  1456  				// output register. Note that immediate 0
  1457  				// selects the low word as the source for every
  1458  				// destination slot.
  1459  				broadcast = s.newValue1I(ssaop.OpAMD64PSHUFLW, types.TypeInt128, 0, unpack)
  1460  
  1461  				// No need to broadcast into the upper 64-bits,
  1462  				// as we don't use those.
  1463  			}
  1464  
  1465  			// Compare each byte of the control word with h2. Each
  1466  			// matching byte has every bit set.
  1467  			eq := s.newValue2(ssaop.OpAMD64PCMPEQB, types.TypeInt128, broadcast, gfp)
  1468  
  1469  			// Construct a "byte mask": each output bit is equal to
  1470  			// the sign bit each input byte.
  1471  			//
  1472  			// This results in a packed output (bit N set means
  1473  			// byte N matched).
  1474  			//
  1475  			// NOTE: See comment above on bitsetFirst.
  1476  			out := s.newValue1(ssaop.OpAMD64PMOVMSKB, types.Types[types.TUINT8], eq)
  1477  
  1478  			// g is only 64-bits so the upper 64-bits of the
  1479  			// 128-bit register will be zero. If h2 is also zero,
  1480  			// then we'll get matches on those bytes. Truncate the
  1481  			// upper bits to ignore such matches.
  1482  			ret := s.newValue1(ssaop.OpZeroExt8to64, types.Types[types.TUINT64], out)
  1483  
  1484  			return ret
  1485  		},
  1486  		sys.AMD64)
  1487  
  1488  	addF("internal/runtime/maps", "ctrlGroupMatchEmpty",
  1489  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1490  			// An empty slot is   1000 0000
  1491  			// A deleted slot is  1111 1110
  1492  			// A full slot is     0??? ????
  1493  
  1494  			g := args[0]
  1495  
  1496  			// Explicit copy to fp register. See
  1497  			// https://go.dev/issue/70451.
  1498  			gfp := s.newValue1(ssaop.OpAMD64MOVQi2f, types.TypeInt128, g)
  1499  
  1500  			if buildcfg.GOAMD64 >= 2 {
  1501  				// "PSIGNB negates each data element of the
  1502  				// destination operand (the first operand) if
  1503  				// the signed integer value of the
  1504  				// corresponding data element in the source
  1505  				// operand (the second operand) is less than
  1506  				// zero. If the signed integer value of a data
  1507  				// element in the source operand is positive,
  1508  				// the corresponding data element in the
  1509  				// destination operand is unchanged. If a data
  1510  				// element in the source operand is zero, the
  1511  				// corresponding data element in the
  1512  				// destination operand is set to zero" - Intel SDM
  1513  				//
  1514  				// If we pass the group control word as both
  1515  				// arguments:
  1516  				// - Full slots are unchanged.
  1517  				// - Deleted slots are negated, becoming
  1518  				//   0000 0010.
  1519  				// - Empty slots are negated, becoming
  1520  				//   1000 0000 (unchanged!).
  1521  				//
  1522  				// The result is that only empty slots have the
  1523  				// sign bit set. We then use PMOVMSKB to
  1524  				// extract the sign bits.
  1525  				sign := s.newValue2(ssaop.OpAMD64PSIGNB, types.TypeInt128, gfp, gfp)
  1526  
  1527  				// Construct a "byte mask": each output bit is
  1528  				// equal to the sign bit each input byte. The
  1529  				// sign bit is only set for empty or deleted
  1530  				// slots.
  1531  				//
  1532  				// This results in a packed output (bit N set
  1533  				// means byte N matched).
  1534  				//
  1535  				// NOTE: See comment above on bitsetFirst.
  1536  				ret := s.newValue1(ssaop.OpAMD64PMOVMSKB, types.Types[types.TUINT64], sign)
  1537  
  1538  				// g is only 64-bits so the upper 64-bits of
  1539  				// the 128-bit register will be zero. PSIGNB
  1540  				// will keep all of these bytes zero, so no
  1541  				// need to truncate.
  1542  
  1543  				return ret
  1544  			}
  1545  
  1546  			// No PSIGNB, simply do byte equality with ctrlEmpty.
  1547  
  1548  			// Load ctrlEmpty into each byte of a control word.
  1549  			var ctrlsEmpty uint64 = abi.MapCtrlEmpty
  1550  			e := s.constInt64(types.Types[types.TUINT64], int64(ctrlsEmpty))
  1551  			// Explicit copy to fp register. See
  1552  			// https://go.dev/issue/70451.
  1553  			efp := s.newValue1(ssaop.OpAMD64MOVQi2f, types.TypeInt128, e)
  1554  
  1555  			// Compare each byte of the control word with ctrlEmpty. Each
  1556  			// matching byte has every bit set.
  1557  			eq := s.newValue2(ssaop.OpAMD64PCMPEQB, types.TypeInt128, efp, gfp)
  1558  
  1559  			// Construct a "byte mask": each output bit is equal to
  1560  			// the sign bit each input byte.
  1561  			//
  1562  			// This results in a packed output (bit N set means
  1563  			// byte N matched).
  1564  			//
  1565  			// NOTE: See comment above on bitsetFirst.
  1566  			out := s.newValue1(ssaop.OpAMD64PMOVMSKB, types.Types[types.TUINT8], eq)
  1567  
  1568  			// g is only 64-bits so the upper 64-bits of the
  1569  			// 128-bit register will be zero. The upper 64-bits of
  1570  			// efp are also zero, so we'll get matches on those
  1571  			// bytes. Truncate the upper bits to ignore such
  1572  			// matches.
  1573  			return s.newValue1(ssaop.OpZeroExt8to64, types.Types[types.TUINT64], out)
  1574  		},
  1575  		sys.AMD64)
  1576  
  1577  	addF("internal/runtime/maps", "ctrlGroupMatchEmptyOrDeleted",
  1578  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1579  			// An empty slot is   1000 0000
  1580  			// A deleted slot is  1111 1110
  1581  			// A full slot is     0??? ????
  1582  			//
  1583  			// A slot is empty or deleted iff bit 7 (sign bit) is
  1584  			// set.
  1585  
  1586  			g := args[0]
  1587  
  1588  			// Explicit copy to fp register. See
  1589  			// https://go.dev/issue/70451.
  1590  			gfp := s.newValue1(ssaop.OpAMD64MOVQi2f, types.TypeInt128, g)
  1591  
  1592  			// Construct a "byte mask": each output bit is equal to
  1593  			// the sign bit each input byte. The sign bit is only
  1594  			// set for empty or deleted slots.
  1595  			//
  1596  			// This results in a packed output (bit N set means
  1597  			// byte N matched).
  1598  			//
  1599  			// NOTE: See comment above on bitsetFirst.
  1600  			ret := s.newValue1(ssaop.OpAMD64PMOVMSKB, types.Types[types.TUINT64], gfp)
  1601  
  1602  			// g is only 64-bits so the upper 64-bits of the
  1603  			// 128-bit register will be zero. Zero will never match
  1604  			// ctrlEmpty or ctrlDeleted, so no need to truncate.
  1605  
  1606  			return ret
  1607  		},
  1608  		sys.AMD64)
  1609  
  1610  	addF("internal/runtime/maps", "ctrlGroupMatchFull",
  1611  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1612  			// An empty slot is   1000 0000
  1613  			// A deleted slot is  1111 1110
  1614  			// A full slot is     0??? ????
  1615  			//
  1616  			// A slot is full iff bit 7 (sign bit) is unset.
  1617  
  1618  			g := args[0]
  1619  
  1620  			// Explicit copy to fp register. See
  1621  			// https://go.dev/issue/70451.
  1622  			gfp := s.newValue1(ssaop.OpAMD64MOVQi2f, types.TypeInt128, g)
  1623  
  1624  			// Construct a "byte mask": each output bit is equal to
  1625  			// the sign bit each input byte. The sign bit is only
  1626  			// set for empty or deleted slots.
  1627  			//
  1628  			// This results in a packed output (bit N set means
  1629  			// byte N matched).
  1630  			//
  1631  			// NOTE: See comment above on bitsetFirst.
  1632  			mask := s.newValue1(ssaop.OpAMD64PMOVMSKB, types.Types[types.TUINT8], gfp)
  1633  
  1634  			// Invert the mask to set the bits for the full slots.
  1635  			out := s.newValue1(ssaop.OpCom8, types.Types[types.TUINT8], mask)
  1636  
  1637  			// g is only 64-bits so the upper 64-bits of the
  1638  			// 128-bit register will be zero, with bit 7 unset.
  1639  			// Truncate the upper bits to ignore these.
  1640  			return s.newValue1(ssaop.OpZeroExt8to64, types.Types[types.TUINT64], out)
  1641  		},
  1642  		sys.AMD64)
  1643  
  1644  	/******** crypto/internal/constanttime ********/
  1645  	// We implement a superset of the Select promise:
  1646  	// Select returns x if v != 0 and y if v == 0.
  1647  	hasCMOV := []*sys.Arch{sys.ArchAMD64, sys.ArchARM64, sys.ArchLoong64, sys.ArchPPC64, sys.ArchPPC64LE, sys.ArchWasm}
  1648  	if cfg.goriscv64 >= 23 {
  1649  		hasCMOV = append(hasCMOV, sys.ArchRISCV64)
  1650  	}
  1651  	add("crypto/internal/constanttime", "Select",
  1652  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1653  			v, x, y := args[0], args[1], args[2]
  1654  
  1655  			var checkOp ssaop.Op
  1656  			var zero *ssa.Value
  1657  			switch s.config.PtrSize {
  1658  			case 8:
  1659  				checkOp = ssaop.OpNeq64
  1660  				zero = s.constInt64(types.Types[types.TINT], 0)
  1661  			case 4:
  1662  				checkOp = ssaop.OpNeq32
  1663  				zero = s.constInt32(types.Types[types.TINT], 0)
  1664  			default:
  1665  				panic("unreachable")
  1666  			}
  1667  			check := s.newValue2(checkOp, types.Types[types.TBOOL], zero, v)
  1668  
  1669  			return s.newValue3(ssaop.OpCondSelect, types.Types[types.TINT], x, y, check)
  1670  		}, hasCMOV...) // all with CMOV support.
  1671  	add("crypto/internal/constanttime", "boolToUint8",
  1672  		func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1673  			return s.newValue1(ssaop.OpCvtBoolToUint8, types.Types[types.TUINT8], args[0])
  1674  		},
  1675  		all...)
  1676  
  1677  	if buildcfg.Experiment.SIMD {
  1678  		// Only enable intrinsics, if SIMD experiment.
  1679  		simdAMD64Intrinsics(addF)
  1680  		simdARM64Intrinsics(addF)
  1681  		initWasmSIMD()
  1682  		simdARM64SVEIntrinsics(addF)
  1683  		// Hand-written SVE infra intrinsic (not generated from a type op): vl
  1684  		// reads the runtime vector length so the package can bound-check it.
  1685  		addF(simdPackage, "vl", opLen0(ssaop.OpScalableVectorLen, types.Types[types.TINT]), sys.ARM64)
  1686  		// TODO: generate these once simdgen supports predicates (mask CL).
  1687  		for _, t := range []struct {
  1688  			name  string
  1689  			bytes int64
  1690  		}{
  1691  			{"Int8s", 1}, {"Uint8s", 1}, {"Int16s", 2}, {"Uint16s", 2},
  1692  			{"Int32s", 4}, {"Uint32s", 4}, {"Float32s", 4},
  1693  			{"Int64s", 8}, {"Uint64s", 8}, {"Float64s", 8},
  1694  		} {
  1695  			// The exported LoadT/StoreT and LoadTPart/StorePart are generated Go
  1696  			// wrappers (see types_sve.go); only the raw whole-register and predicated
  1697  			// load/store are intrinsics.
  1698  			addF(simdPackage, "load"+t.name, sveLoadWhole(), sys.ARM64)
  1699  			addF(simdPackage, t.name+".store", sveStoreWhole(), sys.ARM64)
  1700  			addF(simdPackage, "load"+t.name+"Part", sveLoadPart(t.bytes), sys.ARM64)
  1701  			addF(simdPackage, t.name+".storePart", sveStorePart(t.bytes), sys.ARM64)
  1702  		}
  1703  		// IfElse backs both the IfElse and Masked methods on every scalable vector.
  1704  		for _, t := range []struct {
  1705  			name string
  1706  			op   ssaop.Op
  1707  		}{
  1708  			{"Int8s", ssaop.OpIfElseInt8s},
  1709  			{"Uint8s", ssaop.OpIfElseUint8s},
  1710  			{"Int16s", ssaop.OpIfElseInt16s},
  1711  			{"Uint16s", ssaop.OpIfElseUint16s},
  1712  			{"Int32s", ssaop.OpIfElseInt32s},
  1713  			{"Uint32s", ssaop.OpIfElseUint32s},
  1714  			{"Float32s", ssaop.OpIfElseFloat32s},
  1715  			{"Int64s", ssaop.OpIfElseInt64s},
  1716  			{"Uint64s", ssaop.OpIfElseUint64s},
  1717  			{"Float64s", ssaop.OpIfElseFloat64s},
  1718  		} {
  1719  			addF(simdPackage, t.name+".IfElse", opLen3(t.op, types.TypeVec256), sys.ARM64)
  1720  		}
  1721  
  1722  		addF(simdPackage, "ClearAVXUpperBits",
  1723  			func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1724  				s.vars[memVar] = s.newValue1(ssaop.OpAMD64VZEROUPPER, types.TypeMem, s.mem())
  1725  				return nil
  1726  			},
  1727  			sys.AMD64)
  1728  
  1729  		addF(simdPackage, "Int8x16.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1730  		addF(simdPackage, "Int16x8.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1731  		addF(simdPackage, "Int32x4.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1732  		addF(simdPackage, "Int64x2.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1733  		addF(simdPackage, "Uint8x16.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1734  		addF(simdPackage, "Uint16x8.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1735  		addF(simdPackage, "Uint32x4.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1736  		addF(simdPackage, "Uint64x2.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1737  		addF(simdPackage, "Int8x32.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1738  		addF(simdPackage, "Int16x16.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1739  		addF(simdPackage, "Int32x8.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1740  		addF(simdPackage, "Int64x4.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1741  		addF(simdPackage, "Uint8x32.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1742  		addF(simdPackage, "Uint16x16.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1743  		addF(simdPackage, "Uint32x8.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1744  		addF(simdPackage, "Uint64x4.IsZero", opLen1(ssaop.OpIsZeroVec, types.Types[types.TBOOL]), sys.AMD64)
  1745  		addF(simdPackage, "Float32x4.IsNaN", opLen1(ssaop.OpIsNaNFloat32x4, types.TypeVec128), sys.AMD64)
  1746  		addF(simdPackage, "Float32x8.IsNaN", opLen1(ssaop.OpIsNaNFloat32x8, types.TypeVec256), sys.AMD64)
  1747  		addF(simdPackage, "Float32x16.IsNaN", opLen1(ssaop.OpIsNaNFloat32x16, types.TypeVec512), sys.AMD64)
  1748  		addF(simdPackage, "Float64x2.IsNaN", opLen1(ssaop.OpIsNaNFloat64x2, types.TypeVec128), sys.AMD64)
  1749  		addF(simdPackage, "Float64x4.IsNaN", opLen1(ssaop.OpIsNaNFloat64x4, types.TypeVec256), sys.AMD64)
  1750  		addF(simdPackage, "Float64x8.IsNaN", opLen1(ssaop.OpIsNaNFloat64x8, types.TypeVec512), sys.AMD64)
  1751  
  1752  		// sfp4 is intrinsic-if-constant, but otherwise it's complicated enough to just implement in Go.
  1753  		sfp4 := func(method string, hwop ssaop.Op, vectype *types.Type) {
  1754  			addF(simdPackage, method,
  1755  				func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1756  					x, a, b, c, d, y := args[0], args[1], args[2], args[3], args[4], args[5]
  1757  					if a.Op == ssaop.OpConst8 && b.Op == ssaop.OpConst8 && c.Op == ssaop.OpConst8 && d.Op == ssaop.OpConst8 {
  1758  						z := select4FromPair(x, a, b, c, d, y, s, hwop, vectype)
  1759  						if z != nil {
  1760  							return z
  1761  						}
  1762  					}
  1763  					return s.callResult(n, callNormal)
  1764  				},
  1765  				sys.AMD64)
  1766  		}
  1767  
  1768  		sfp4("Int32x4.ConcatPermuteScalars", ssaop.OpconcatSelectedConstantInt32x4, types.TypeVec128)
  1769  		sfp4("Uint32x4.ConcatPermuteScalars", ssaop.OpconcatSelectedConstantUint32x4, types.TypeVec128)
  1770  		sfp4("Float32x4.ConcatPermuteScalars", ssaop.OpconcatSelectedConstantFloat32x4, types.TypeVec128)
  1771  
  1772  		sfp4("Int32x8.ConcatPermuteScalarsGrouped", ssaop.OpconcatSelectedConstantGroupedInt32x8, types.TypeVec256)
  1773  		sfp4("Uint32x8.ConcatPermuteScalarsGrouped", ssaop.OpconcatSelectedConstantGroupedUint32x8, types.TypeVec256)
  1774  		sfp4("Float32x8.ConcatPermuteScalarsGrouped", ssaop.OpconcatSelectedConstantGroupedFloat32x8, types.TypeVec256)
  1775  
  1776  		sfp4("Int32x16.ConcatPermuteScalarsGrouped", ssaop.OpconcatSelectedConstantGroupedInt32x16, types.TypeVec512)
  1777  		sfp4("Uint32x16.ConcatPermuteScalarsGrouped", ssaop.OpconcatSelectedConstantGroupedUint32x16, types.TypeVec512)
  1778  		sfp4("Float32x16.ConcatPermuteScalarsGrouped", ssaop.OpconcatSelectedConstantGroupedFloat32x16, types.TypeVec512)
  1779  
  1780  		// sfp2 is intrinsic-if-constant, but otherwise it's complicated enough to just implement in Go.
  1781  		sfp2 := func(method string, hwop ssaop.Op, vectype *types.Type, cscimm func(i, j uint8) int64) {
  1782  			addF(simdPackage, method,
  1783  				func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1784  					x, a, b, y := args[0], args[1], args[2], args[3]
  1785  					if a.Op == ssaop.OpConst8 && b.Op == ssaop.OpConst8 {
  1786  						z := select2FromPair(x, a, b, y, s, hwop, vectype, cscimm)
  1787  						if z != nil {
  1788  							return z
  1789  						}
  1790  					}
  1791  					return s.callResult(n, callNormal)
  1792  				},
  1793  				sys.AMD64)
  1794  		}
  1795  
  1796  		sfp2("Uint64x2.ConcatPermuteScalars", ssaop.OpconcatSelectedConstantUint64x2, types.TypeVec128, cscimm2)
  1797  		sfp2("Int64x2.ConcatPermuteScalars", ssaop.OpconcatSelectedConstantInt64x2, types.TypeVec128, cscimm2)
  1798  		sfp2("Float64x2.ConcatPermuteScalars", ssaop.OpconcatSelectedConstantFloat64x2, types.TypeVec128, cscimm2)
  1799  
  1800  		sfp2("Uint64x4.ConcatPermuteScalarsGrouped", ssaop.OpconcatSelectedConstantGroupedUint64x4, types.TypeVec256, cscimm2g2)
  1801  		sfp2("Int64x4.ConcatPermuteScalarsGrouped", ssaop.OpconcatSelectedConstantGroupedInt64x4, types.TypeVec256, cscimm2g2)
  1802  		sfp2("Float64x4.ConcatPermuteScalarsGrouped", ssaop.OpconcatSelectedConstantGroupedFloat64x4, types.TypeVec256, cscimm2g2)
  1803  
  1804  		sfp2("Uint64x8.ConcatPermuteScalarsGrouped", ssaop.OpconcatSelectedConstantGroupedUint64x8, types.TypeVec512, cscimm2g4)
  1805  		sfp2("Int64x8.ConcatPermuteScalarsGrouped", ssaop.OpconcatSelectedConstantGroupedInt64x8, types.TypeVec512, cscimm2g4)
  1806  		sfp2("Float64x8.ConcatPermuteScalarsGrouped", ssaop.OpconcatSelectedConstantGroupedFloat64x8, types.TypeVec512, cscimm2g4)
  1807  
  1808  	}
  1809  }
  1810  
  1811  const simdPackage = "simd/archsimd"
  1812  
  1813  func cscimm4(a, b, c, d uint8) int64 {
  1814  	return se(a + b<<2 + c<<4 + d<<6)
  1815  }
  1816  
  1817  func cscimm2(a, b uint8) int64 {
  1818  	return se(a + b<<1)
  1819  }
  1820  
  1821  func cscimm2g2(a, b uint8) int64 {
  1822  	g := cscimm2(a, b)
  1823  	return int64(int8(g + g<<2))
  1824  }
  1825  
  1826  func cscimm2g4(a, b uint8) int64 {
  1827  	g := cscimm2g2(a, b)
  1828  	return int64(int8(g + g<<4))
  1829  }
  1830  
  1831  const (
  1832  	_LLLL = iota
  1833  	_HLLL
  1834  	_LHLL
  1835  	_HHLL
  1836  	_LLHL
  1837  	_HLHL
  1838  	_LHHL
  1839  	_HHHL
  1840  	_LLLH
  1841  	_HLLH
  1842  	_LHLH
  1843  	_HHLH
  1844  	_LLHH
  1845  	_HLHH
  1846  	_LHHH
  1847  	_HHHH
  1848  )
  1849  
  1850  const (
  1851  	_LL = iota
  1852  	_HL
  1853  	_LH
  1854  	_HH
  1855  )
  1856  
  1857  func select2FromPair(x, _a, _b, y *ssa.Value, s *state, op ssaop.Op, t *types.Type, csc func(a, b uint8) int64) *ssa.Value {
  1858  	a, b := uint8(_a.AuxInt8()), uint8(_b.AuxInt8())
  1859  	if a > 3 || b > 3 {
  1860  		return nil
  1861  	}
  1862  	pattern := (a&2)>>1 + (b & 2)
  1863  	a, b = a&1, b&1
  1864  
  1865  	switch pattern {
  1866  	case _LL:
  1867  		return s.newValue2I(op, t, csc(a, b), x, x)
  1868  	case _HH:
  1869  		return s.newValue2I(op, t, csc(a, b), y, y)
  1870  	case _LH:
  1871  		return s.newValue2I(op, t, csc(a, b), x, y)
  1872  	case _HL:
  1873  		return s.newValue2I(op, t, csc(a, b), y, x)
  1874  	}
  1875  	panic("The preceding switch should have been exhaustive")
  1876  }
  1877  
  1878  func select4FromPair(x, _a, _b, _c, _d, y *ssa.Value, s *state, op ssaop.Op, t *types.Type) *ssa.Value {
  1879  	a, b, c, d := uint8(_a.AuxInt8()), uint8(_b.AuxInt8()), uint8(_c.AuxInt8()), uint8(_d.AuxInt8())
  1880  	if a > 7 || b > 7 || c > 7 || d > 7 {
  1881  		return nil
  1882  	}
  1883  	pattern := a>>2 + (b&4)>>1 + (c & 4) + (d&4)<<1
  1884  
  1885  	a, b, c, d = a&3, b&3, c&3, d&3
  1886  
  1887  	switch pattern {
  1888  	case _LLLL:
  1889  		// TODO DETECT 0,1,2,3, 0,0,0,0
  1890  		return s.newValue2I(op, t, cscimm4(a, b, c, d), x, x)
  1891  	case _HHHH:
  1892  		// TODO DETECT 0,1,2,3, 0,0,0,0
  1893  		return s.newValue2I(op, t, cscimm4(a, b, c, d), y, y)
  1894  	case _LLHH:
  1895  		return s.newValue2I(op, t, cscimm4(a, b, c, d), x, y)
  1896  	case _HHLL:
  1897  		return s.newValue2I(op, t, cscimm4(a, b, c, d), y, x)
  1898  
  1899  	case _HLLL:
  1900  		z := s.newValue2I(op, t, cscimm4(a, a, b, b), y, x)
  1901  		return s.newValue2I(op, t, cscimm4(0, 2, c, d), z, x)
  1902  	case _LHLL:
  1903  		z := s.newValue2I(op, t, cscimm4(a, a, b, b), x, y)
  1904  		return s.newValue2I(op, t, cscimm4(0, 2, c, d), z, x)
  1905  	case _HLHH:
  1906  		z := s.newValue2I(op, t, cscimm4(a, a, b, b), y, x)
  1907  		return s.newValue2I(op, t, cscimm4(0, 2, c, d), z, y)
  1908  	case _LHHH:
  1909  		z := s.newValue2I(op, t, cscimm4(a, a, b, b), x, y)
  1910  		return s.newValue2I(op, t, cscimm4(0, 2, c, d), z, y)
  1911  
  1912  	case _LLLH:
  1913  		z := s.newValue2I(op, t, cscimm4(c, c, d, d), x, y)
  1914  		return s.newValue2I(op, t, cscimm4(a, b, 0, 2), x, z)
  1915  	case _LLHL:
  1916  		z := s.newValue2I(op, t, cscimm4(c, c, d, d), y, x)
  1917  		return s.newValue2I(op, t, cscimm4(a, b, 0, 2), x, z)
  1918  
  1919  	case _HHLH:
  1920  		z := s.newValue2I(op, t, cscimm4(c, c, d, d), x, y)
  1921  		return s.newValue2I(op, t, cscimm4(a, b, 0, 2), y, z)
  1922  
  1923  	case _HHHL:
  1924  		z := s.newValue2I(op, t, cscimm4(c, c, d, d), y, x)
  1925  		return s.newValue2I(op, t, cscimm4(a, b, 0, 2), y, z)
  1926  
  1927  	case _LHLH:
  1928  		z := s.newValue2I(op, t, cscimm4(a, c, b, d), x, y)
  1929  		return s.newValue2I(op, t, se(0b11_01_10_00), z, z)
  1930  	case _HLHL:
  1931  		z := s.newValue2I(op, t, cscimm4(b, d, a, c), x, y)
  1932  		return s.newValue2I(op, t, se(0b01_11_00_10), z, z)
  1933  	case _HLLH:
  1934  		z := s.newValue2I(op, t, cscimm4(b, c, a, d), x, y)
  1935  		return s.newValue2I(op, t, se(0b11_01_00_10), z, z)
  1936  	case _LHHL:
  1937  		z := s.newValue2I(op, t, cscimm4(a, d, b, c), x, y)
  1938  		return s.newValue2I(op, t, se(0b01_11_10_00), z, z)
  1939  	}
  1940  	panic("The preceding switch should have been exhaustive")
  1941  }
  1942  
  1943  // se smears the not-really-a-sign bit of a uint8 to conform to the conventions
  1944  // for representing AuxInt in ssa.
  1945  func se(x uint8) int64 {
  1946  	return int64(int8(x))
  1947  }
  1948  
  1949  func opLen0(op ssaop.Op, t *types.Type) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1950  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1951  		return s.newValue0(op, t)
  1952  	}
  1953  }
  1954  
  1955  func opLen1(op ssaop.Op, t *types.Type) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1956  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1957  		return s.newValue1(op, t, args[0])
  1958  	}
  1959  }
  1960  
  1961  func opLen2(op ssaop.Op, t *types.Type) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1962  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1963  		return s.newValue2(op, t, args[0], args[1])
  1964  	}
  1965  }
  1966  
  1967  func opLen2_21(op ssaop.Op, t *types.Type) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1968  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1969  		return s.newValue2(op, t, args[1], args[0])
  1970  	}
  1971  }
  1972  
  1973  func opLen3(op ssaop.Op, t *types.Type) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1974  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1975  		return s.newValue3(op, t, args[0], args[1], args[2])
  1976  	}
  1977  }
  1978  
  1979  var ssaVecBySize = map[int64]*types.Type{
  1980  	16: types.TypeVec128,
  1981  	32: types.TypeVec256,
  1982  	64: types.TypeVec512,
  1983  }
  1984  
  1985  func opLen3_31Zero3(op ssaop.Op, t *types.Type) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1986  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1987  		if t, ok := ssaVecBySize[args[1].Type.Size()]; !ok {
  1988  			panic("unknown simd vector size")
  1989  		} else {
  1990  			return s.newValue3(op, t, s.newValue0(ssaop.OpZeroSIMD, t), args[1], args[0])
  1991  		}
  1992  	}
  1993  }
  1994  
  1995  func opLen3_21(op ssaop.Op, t *types.Type) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1996  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  1997  		return s.newValue3(op, t, args[1], args[0], args[2])
  1998  	}
  1999  }
  2000  
  2001  func opLen3_231(op ssaop.Op, t *types.Type) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2002  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2003  		return s.newValue3(op, t, args[2], args[0], args[1])
  2004  	}
  2005  }
  2006  
  2007  func opLen4(op ssaop.Op, t *types.Type) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2008  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2009  		return s.newValue4(op, t, args[0], args[1], args[2], args[3])
  2010  	}
  2011  }
  2012  
  2013  func opLen4_231(op ssaop.Op, t *types.Type) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2014  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2015  		return s.newValue4(op, t, args[2], args[0], args[1], args[3])
  2016  	}
  2017  }
  2018  
  2019  func opLen4_31(op ssaop.Op, t *types.Type) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2020  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2021  		return s.newValue4(op, t, args[2], args[1], args[0], args[3])
  2022  	}
  2023  }
  2024  
  2025  func immJumpTable(s *state, idx *ssa.Value, intrinsicCall *ir.CallExpr, genOp func(*state, int)) *ssa.Value {
  2026  	if !idx.Type.IsKind(types.TUINT8) && !idx.Type.IsKind(types.TUINT64) {
  2027  		panic("immJumpTable expects uint8 or uint64 value")
  2028  	}
  2029  	if idx.Type.IsKind(types.TUINT64) {
  2030  		// Match the constant path and keep the jump table index in range.
  2031  		idx = s.conv(nil, idx, idx.Type, types.Types[types.TUINT8])
  2032  	}
  2033  
  2034  	if base.Ctxt.Retpoline {
  2035  		// Note spectre=all implies retpoline which requires binary search instead of table switch.
  2036  		return branchTableImm8(s, idx, intrinsicCall, genOp)
  2037  	}
  2038  
  2039  	// Make blocks we'll need.
  2040  	bEnd := s.f.NewBlock(block.BlockPlain)
  2041  
  2042  	// We will exhaust 0-255, so no need to check the bounds.
  2043  	t := types.Types[types.TUINTPTR]
  2044  	idx = s.conv(nil, idx, idx.Type, t)
  2045  
  2046  	b := s.curBlock
  2047  	b.Kind = block.BlockJumpTable
  2048  	b.Pos = intrinsicCall.Pos()
  2049  
  2050  	b.SetControl(idx)
  2051  	targets := [256]*ssa.Block{}
  2052  	for i := range 256 {
  2053  		t := s.f.NewBlock(block.BlockPlain)
  2054  		targets[i] = t
  2055  		b.AddEdgeTo(t)
  2056  	}
  2057  	s.endBlock()
  2058  
  2059  	for i, t := range targets {
  2060  		s.startBlock(t)
  2061  		genOp(s, i)
  2062  		if t.Kind != block.BlockExit {
  2063  			t.AddEdgeTo(bEnd)
  2064  		}
  2065  		s.endBlock()
  2066  	}
  2067  
  2068  	s.startBlock(bEnd)
  2069  	ret := s.variable(intrinsicCall, intrinsicCall.Type())
  2070  	return ret
  2071  }
  2072  
  2073  func branchTableImm8(s *state, idx *ssa.Value, intrinsicCall *ir.CallExpr, genOp func(*state, int)) *ssa.Value {
  2074  	return branchTableN(s, idx, intrinsicCall, genOp, 256, true)
  2075  }
  2076  
  2077  func branchTableN(s *state, idx *ssa.Value, intrinsicCall *ir.CallExpr, genOp func(*state, int), immLimit uint64, preChecked bool) *ssa.Value {
  2078  	// Make blocks we'll need.
  2079  	bEnd := s.f.NewBlock(block.BlockPlain)
  2080  	bPanic := s.f.NewBlock(block.BlockPlain)
  2081  
  2082  	jt := s.f.NewBlock(block.BlockPlain)
  2083  
  2084  	t := types.Types[types.TUINTPTR]
  2085  	idx = s.conv(nil, idx, idx.Type, t)
  2086  
  2087  	if !preChecked {
  2088  		// Begin with a bounds check
  2089  		width := s.uintptrConstant(immLimit)
  2090  		cmp := s.newValue2(s.ssaOp(ir.OLT, t), types.Types[types.TBOOL], idx, width)
  2091  		bb := s.endBlock()
  2092  		bb.Kind = block.BlockIf
  2093  		bb.SetControl(cmp)
  2094  		bb.AddEdgeTo(jt)             // in range - use jump table
  2095  		bb.AddEdgeTo(bPanic)         // out of range - panic
  2096  		bb.Likely = ssa.BranchLikely // panic is unlikely
  2097  
  2098  		s.startBlock(bPanic)
  2099  		s.rtcall(ir.Syms.PanicSimdImm, false, nil)
  2100  	}
  2101  	if s.curBlock != nil {
  2102  		bb := s.endBlock()
  2103  		bb.AddEdgeTo(jt)
  2104  	}
  2105  
  2106  	s.startBlock(jt)
  2107  	jt.Kind = block.BlockPlain
  2108  	jt.Pos = intrinsicCall.Pos()
  2109  
  2110  	branchTableNInner(s, idx, 0, immLimit, genOp, bEnd)
  2111  
  2112  	s.startBlock(bEnd)
  2113  	ret := s.variable(intrinsicCall, intrinsicCall.Type())
  2114  	return ret
  2115  }
  2116  
  2117  func branchTableNInner(s *state, idx *ssa.Value, lowInclusive, len uint64, genOp func(*state, int), bEnd *ssa.Block) {
  2118  	t := types.Types[types.TUINTPTR]
  2119  	if len == 0 {
  2120  		panic("empty branch table")
  2121  	}
  2122  	if len == 1 {
  2123  		genOp(s, int(lowInclusive+len-1))
  2124  		if s.curBlock != nil { // if genOp was "panic" then curBlock is already ended and nil
  2125  			if s.curBlock.Kind != block.BlockExit {
  2126  				s.curBlock.AddEdgeTo(bEnd)
  2127  			}
  2128  			s.endBlock()
  2129  		}
  2130  		return
  2131  	}
  2132  
  2133  	s.curBlock.Kind = block.BlockIf
  2134  	cmp := s.newValue2(s.ssaOp(ir.OLT, t), types.Types[types.TBOOL], idx, s.uintptrConstant(lowInclusive+len/2))
  2135  	bb := s.endBlock()
  2136  	bb.Kind = block.BlockIf
  2137  	bb.SetControl(cmp)
  2138  	bMatch := s.f.NewBlock(block.BlockPlain)
  2139  	bNext := s.f.NewBlock(block.BlockPlain)
  2140  	bb.AddEdgeTo(bMatch)
  2141  	bb.AddEdgeTo(bNext)
  2142  	s.startBlock(bMatch)
  2143  	branchTableNInner(s, idx, lowInclusive, len/2, genOp, bEnd)
  2144  	s.startBlock(bNext)
  2145  	branchTableNInner(s, idx, lowInclusive+len/2, len-len/2, genOp, bEnd)
  2146  }
  2147  
  2148  // immJumpTableN emits a jump table to one of a number of indexed cases, from zero to n-1.
  2149  // an index of n or larger will panic
  2150  func immJumpTableN(s *state, idx *ssa.Value, intrinsicCall *ir.CallExpr, immLimit uint64, genOp func(*state, int)) *ssa.Value {
  2151  
  2152  	if !idx.Type.IsKind(types.TUINT8) && !idx.Type.IsKind(types.TUINT64) {
  2153  		s.Fatalf("immJumpTable expects uint8 or uint64 value, saw %v instead, val=%s", idx.Type.String(), idx.LongString())
  2154  	}
  2155  
  2156  	if base.Flag.N != 0 || !Arch.LinkArch.CanJumpTable || base.Ctxt.Retpoline {
  2157  		return branchTableN(s, idx, intrinsicCall, genOp, immLimit, false)
  2158  	}
  2159  
  2160  	// Make blocks we'll need.
  2161  	bEnd := s.f.NewBlock(block.BlockPlain)
  2162  	bPanic := s.f.NewBlock(block.BlockPlain)
  2163  
  2164  	jt := s.f.NewBlock(block.BlockJumpTable)
  2165  
  2166  	t := types.Types[types.TUINTPTR]
  2167  	idx = s.conv(nil, idx, idx.Type, t)
  2168  	width := s.uintptrConstant(immLimit)
  2169  
  2170  	// Begin with a bounds check
  2171  	cmp := s.newValue2(s.ssaOp(ir.OLT, t), types.Types[types.TBOOL], idx, width)
  2172  	bb := s.endBlock()
  2173  	bb.Kind = block.BlockIf
  2174  	bb.SetControl(cmp)
  2175  	bb.AddEdgeTo(jt)             // in range - use jump table
  2176  	bb.AddEdgeTo(bPanic)         // out of range - panic
  2177  	bb.Likely = ssa.BranchLikely // panic is unlikely
  2178  
  2179  	s.startBlock(bPanic)
  2180  	s.rtcall(ir.Syms.PanicSimdImm, false, nil)
  2181  	s.endBlock()
  2182  
  2183  	s.startBlock(jt)
  2184  	jt.Kind = block.BlockJumpTable
  2185  	jt.Pos = intrinsicCall.Pos()
  2186  	if base.Flag.Cfg.SpectreIndex {
  2187  		// Potential Spectre vulnerability hardening?
  2188  		idx = s.newValue2(ssaop.OpSpectreSliceIndex, t, idx, s.uintptrConstant(immLimit-1))
  2189  	}
  2190  	jt.SetControl(idx)
  2191  	targets := make([]*ssa.Block, immLimit, immLimit)
  2192  	for i := range immLimit {
  2193  		t := s.f.NewBlock(block.BlockPlain)
  2194  		targets[i] = t
  2195  		jt.AddEdgeTo(t)
  2196  	}
  2197  	s.endBlock()
  2198  
  2199  	for i, t := range targets {
  2200  		s.startBlock(t)
  2201  		genOp(s, i)
  2202  		if t.Kind != block.BlockExit {
  2203  			t.AddEdgeTo(bEnd)
  2204  		}
  2205  		s.endBlock()
  2206  	}
  2207  
  2208  	s.startBlock(bEnd)
  2209  	ret := s.variable(intrinsicCall, intrinsicCall.Type())
  2210  	return ret
  2211  }
  2212  
  2213  func opLen1Imm8(op ssaop.Op, t *types.Type, offset int) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2214  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2215  		if args[1].Op == ssaop.OpConst8 || args[1].Op == ssaop.OpConst64 {
  2216  			return s.newValue1I(op, t, int64(int8(args[1].AuxInt<<int64(offset))), args[0])
  2217  		}
  2218  		return immJumpTable(s, args[1], n, func(sNew *state, idx int) {
  2219  			// Encode as int8 due to requirement of AuxInt, check its comment for details.
  2220  			s.vars[n] = sNew.newValue1I(op, t, int64(int8(idx<<offset)), args[0])
  2221  		})
  2222  	}
  2223  }
  2224  
  2225  func opLen2Imm8(op ssaop.Op, t *types.Type, offset int) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2226  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2227  		if args[1].Op == ssaop.OpConst8 || args[1].Op == ssaop.OpConst64 {
  2228  			return s.newValue2I(op, t, int64(int8(args[1].AuxInt<<int64(offset))), args[0], args[2])
  2229  		}
  2230  		return immJumpTable(s, args[1], n, func(sNew *state, idx int) {
  2231  			// Encode as int8 due to requirement of AuxInt, check its comment for details.
  2232  			s.vars[n] = sNew.newValue2I(op, t, int64(int8(idx<<offset)), args[0], args[2])
  2233  		})
  2234  	}
  2235  }
  2236  
  2237  func opLen3Imm8(op ssaop.Op, t *types.Type, offset int) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2238  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2239  		if args[1].Op == ssaop.OpConst8 || args[1].Op == ssaop.OpConst64 {
  2240  			return s.newValue3I(op, t, int64(int8(args[1].AuxInt<<int64(offset))), args[0], args[2], args[3])
  2241  		}
  2242  		return immJumpTable(s, args[1], n, func(sNew *state, idx int) {
  2243  			// Encode as int8 due to requirement of AuxInt, check its comment for details.
  2244  			s.vars[n] = sNew.newValue3I(op, t, int64(int8(idx<<offset)), args[0], args[2], args[3])
  2245  		})
  2246  	}
  2247  }
  2248  
  2249  func opLen2Imm8_2I(op ssaop.Op, t *types.Type, offset int) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2250  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2251  		if args[2].Op == ssaop.OpConst8 || args[2].Op == ssaop.OpConst64 {
  2252  			return s.newValue2I(op, t, int64(int8(args[2].AuxInt<<int64(offset))), args[0], args[1])
  2253  		}
  2254  		return immJumpTable(s, args[2], n, func(sNew *state, idx int) {
  2255  			// Encode as int8 due to requirement of AuxInt, check its comment for details.
  2256  			s.vars[n] = sNew.newValue2I(op, t, int64(int8(idx<<offset)), args[0], args[1])
  2257  		})
  2258  	}
  2259  }
  2260  
  2261  // Two immediates instead of just 1.  Offset is ignored, so it is a _ parameter instead.
  2262  func opLen2Imm8_II(op ssaop.Op, t *types.Type, _ int) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2263  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2264  		if (args[1].Op == ssaop.OpConst8 || args[1].Op == ssaop.OpConst64) && (args[2].Op == ssaop.OpConst8 || args[2].Op == ssaop.OpConst64) && args[1].AuxInt & ^3 == 0 && args[2].AuxInt & ^3 == 0 {
  2265  			i1, i2 := args[1].AuxInt, args[2].AuxInt
  2266  			return s.newValue2I(op, t, int64(int8(i1+i2<<4)), args[0], args[3])
  2267  		}
  2268  		four := s.constInt64(types.Types[types.TUINT8], 4)
  2269  		shifted := s.newValue2(ssaop.OpLsh8x8, types.Types[types.TUINT8], args[2], four)
  2270  		combined := s.newValue2(ssaop.OpAdd8, types.Types[types.TUINT8], args[1], shifted)
  2271  		return immJumpTable(s, combined, n, func(sNew *state, idx int) {
  2272  			// Encode as int8 due to requirement of AuxInt, check its comment for details.
  2273  			// TODO for "zeroing" values, panic instead.
  2274  			if idx & ^(3+3<<4) == 0 {
  2275  				s.vars[n] = sNew.newValue2I(op, t, int64(int8(idx)), args[0], args[3])
  2276  			} else {
  2277  				sNew.rtcall(ir.Syms.PanicSimdImm, false, nil)
  2278  			}
  2279  		})
  2280  	}
  2281  }
  2282  
  2283  // The assembler requires the imm value of a SHA1RNDS4 instruction to be one of 0,1,2,3...
  2284  func opLen2Imm8_SHA1RNDS4(op ssaop.Op, t *types.Type, offset int) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2285  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2286  		if args[1].Op == ssaop.OpConst8 || args[1].Op == ssaop.OpConst64 {
  2287  			return s.newValue2I(op, t, int64(int8((args[1].AuxInt<<int64(offset))&0b11)), args[0], args[2])
  2288  		}
  2289  		return immJumpTable(s, args[1], n, func(sNew *state, idx int) {
  2290  			// Encode as int8 due to requirement of AuxInt, check its comment for details.
  2291  			s.vars[n] = sNew.newValue2I(op, t, int64(int8(idx<<offset))&0b11, args[0], args[2])
  2292  		})
  2293  	}
  2294  }
  2295  
  2296  func opLen1Imm(op ssaop.Op, t *types.Type, offset int, immMax uint64) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2297  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2298  		if (args[1].Op == ssaop.OpConst8 || args[1].Op == ssaop.OpConst64) && uint64(args[1].AuxInt) <= immMax {
  2299  			return s.newValue1I(op, t, int64(int8(args[1].AuxInt<<int64(offset))), args[0])
  2300  		}
  2301  		return immJumpTableN(s, args[1], n, immMax+1, func(sNew *state, idx int) {
  2302  			// Encode as int8 due to requirement of AuxInt, check its comment for details.
  2303  			s.vars[n] = sNew.newValue1I(op, t, int64(int8(idx<<offset)), args[0])
  2304  		})
  2305  	}
  2306  }
  2307  
  2308  func opLen2Imm(op ssaop.Op, t *types.Type, offset int, immMax uint64) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2309  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2310  		if (args[1].Op == ssaop.OpConst8 || args[1].Op == ssaop.OpConst64) && uint64(args[1].AuxInt) <= immMax {
  2311  			return s.newValue2I(op, t, int64(int8(args[1].AuxInt<<int64(offset))), args[0], args[2])
  2312  		}
  2313  		return immJumpTableN(s, args[1], n, immMax+1, func(sNew *state, idx int) {
  2314  			// Encode as int8 due to requirement of AuxInt, check its comment for details.
  2315  			s.vars[n] = sNew.newValue2I(op, t, int64(int8(idx<<offset)), args[0], args[2])
  2316  		})
  2317  	}
  2318  }
  2319  
  2320  func opLen3Imm(op ssaop.Op, t *types.Type, offset int, immMax uint64) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2321  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2322  		if (args[1].Op == ssaop.OpConst8 || args[1].Op == ssaop.OpConst64) && uint64(args[1].AuxInt) <= immMax {
  2323  			return s.newValue3I(op, t, int64(int8(args[1].AuxInt<<int64(offset))), args[0], args[2], args[3])
  2324  		}
  2325  		return immJumpTableN(s, args[1], n, immMax+1, func(sNew *state, idx int) {
  2326  			// Encode as int8 due to requirement of AuxInt, check its comment for details.
  2327  			s.vars[n] = sNew.newValue3I(op, t, int64(int8(idx<<offset)), args[0], args[2], args[3])
  2328  		})
  2329  	}
  2330  }
  2331  
  2332  func opLen2Imm_2I(op ssaop.Op, t *types.Type, offset int, immMax uint64) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2333  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2334  		if (args[2].Op == ssaop.OpConst8 || args[2].Op == ssaop.OpConst64) && uint64(args[2].AuxInt) <= immMax {
  2335  			return s.newValue2I(op, t, int64(int8(args[2].AuxInt<<int64(offset))), args[0], args[1])
  2336  		}
  2337  		return immJumpTableN(s, args[2], n, immMax+1, func(sNew *state, idx int) {
  2338  			// Encode as int8 due to requirement of AuxInt, check its comment for details.
  2339  			s.vars[n] = sNew.newValue2I(op, t, int64(int8(idx<<offset)), args[0], args[1])
  2340  		})
  2341  	}
  2342  }
  2343  
  2344  func opLen3Imm8_2I(op ssaop.Op, t *types.Type, offset int) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2345  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2346  		if args[2].Op == ssaop.OpConst8 || args[2].Op == ssaop.OpConst64 {
  2347  			return s.newValue3I(op, t, int64(int8(args[2].AuxInt<<int64(offset))), args[0], args[1], args[3])
  2348  		}
  2349  		return immJumpTable(s, args[2], n, func(sNew *state, idx int) {
  2350  			// Encode as int8 due to requirement of AuxInt, check its comment for details.
  2351  			s.vars[n] = sNew.newValue3I(op, t, int64(int8(idx<<offset)), args[0], args[1], args[3])
  2352  		})
  2353  	}
  2354  }
  2355  
  2356  func opLen4Imm8(op ssaop.Op, t *types.Type, offset int) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2357  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2358  		if args[1].Op == ssaop.OpConst8 || args[1].Op == ssaop.OpConst64 {
  2359  			return s.newValue4I(op, t, int64(int8(args[1].AuxInt<<int64(offset))), args[0], args[2], args[3], args[4])
  2360  		}
  2361  		return immJumpTable(s, args[1], n, func(sNew *state, idx int) {
  2362  			// Encode as int8 due to requirement of AuxInt, check its comment for details.
  2363  			s.vars[n] = sNew.newValue4I(op, t, int64(int8(idx<<offset)), args[0], args[2], args[3], args[4])
  2364  		})
  2365  	}
  2366  }
  2367  
  2368  func simdBroadcast(op ssaop.Op) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2369  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2370  		return s.newValue2(op, n.Type(), args[0], s.mem())
  2371  	}
  2372  }
  2373  
  2374  func simdLoad() func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2375  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2376  		ptr := s.nilCheck(args[0])
  2377  		return s.newValue2(ssaop.OpLoad, n.Type(), ptr, s.mem())
  2378  	}
  2379  }
  2380  
  2381  func simdStore() func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2382  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2383  		ptr := s.nilCheck(args[1])
  2384  		s.store(args[0].Type, ptr, args[0])
  2385  		return nil
  2386  	}
  2387  }
  2388  
  2389  var cvtVToMaskOpcodes = map[int]map[int]ssaop.Op{
  2390  	8:  {16: ssaop.OpCvt16toMask8x16, 32: ssaop.OpCvt32toMask8x32, 64: ssaop.OpCvt64toMask8x64},
  2391  	16: {8: ssaop.OpCvt8toMask16x8, 16: ssaop.OpCvt16toMask16x16, 32: ssaop.OpCvt32toMask16x32},
  2392  	32: {4: ssaop.OpCvt8toMask32x4, 8: ssaop.OpCvt8toMask32x8, 16: ssaop.OpCvt16toMask32x16},
  2393  	64: {2: ssaop.OpCvt8toMask64x2, 4: ssaop.OpCvt8toMask64x4, 8: ssaop.OpCvt8toMask64x8},
  2394  }
  2395  
  2396  var cvtMaskToVOpcodes = map[int]map[int]ssaop.Op{
  2397  	8:  {16: ssaop.OpCvtMask8x16to16, 32: ssaop.OpCvtMask8x32to32, 64: ssaop.OpCvtMask8x64to64},
  2398  	16: {8: ssaop.OpCvtMask16x8to8, 16: ssaop.OpCvtMask16x16to16, 32: ssaop.OpCvtMask16x32to32},
  2399  	32: {4: ssaop.OpCvtMask32x4to8, 8: ssaop.OpCvtMask32x8to8, 16: ssaop.OpCvtMask32x16to16},
  2400  	64: {2: ssaop.OpCvtMask64x2to8, 4: ssaop.OpCvtMask64x4to8, 8: ssaop.OpCvtMask64x8to8},
  2401  }
  2402  
  2403  func simdCvtVToMask(elemBits, lanes int) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2404  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2405  		op := cvtVToMaskOpcodes[elemBits][lanes]
  2406  		if op == 0 {
  2407  			panic(fmt.Sprintf("Unknown mask shape: Mask%dx%d", elemBits, lanes))
  2408  		}
  2409  		return s.newValue1(op, types.TypeMask, args[0])
  2410  	}
  2411  }
  2412  
  2413  func simdCvtMaskToV(elemBits, lanes int) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2414  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2415  		op := cvtMaskToVOpcodes[elemBits][lanes]
  2416  		if op == 0 {
  2417  			panic(fmt.Sprintf("Unknown mask shape: Mask%dx%d", elemBits, lanes))
  2418  		}
  2419  		return s.newValue1(op, n.Type(), args[0])
  2420  	}
  2421  }
  2422  
  2423  func simdMaskedLoad(op ssaop.Op) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2424  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2425  		return s.newValue3(op, n.Type(), args[0], args[1], s.mem())
  2426  	}
  2427  }
  2428  
  2429  func simdMaskedStore(op ssaop.Op) func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2430  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2431  		s.vars[memVar] = s.newValue4A(op, types.TypeMem, args[0].Type, args[1], args[2], args[0], s.mem())
  2432  		return nil
  2433  	}
  2434  }
  2435  
  2436  // sveByteCount returns len(s)*elemBytes as an SSA int, the number of active bytes
  2437  // for a byte-granular scalable load/store of an elemBytes-wide element type.
  2438  func sveByteCount(s *state, length *ssa.Value, elemBytes int64) *ssa.Value {
  2439  	if elemBytes == 1 {
  2440  		return length
  2441  	}
  2442  	return s.newValue2(ssaop.OpMul64, types.Types[types.TINT], length, s.constInt64(types.Types[types.TINT], elemBytes))
  2443  }
  2444  
  2445  // slicePtrLen extracts the data pointer and length of a slice SSA value.
  2446  func slicePtrLen(s *state, slice *ssa.Value) (ptr, length *ssa.Value) {
  2447  	ptr = s.newValue1(ssaop.OpSlicePtr, types.NewPtr(slice.Type.Elem()), slice)
  2448  	length = s.newValue1(ssaop.OpSliceLen, types.Types[types.TINT], slice)
  2449  	return
  2450  }
  2451  
  2452  // sveLoadWhole builds a raw whole-register load loadT(s) / loadMask*s(bits): a
  2453  // generic Load of the return type from the slice's data pointer, lowered to ZLDR
  2454  // (a 32-byte scalable vector) or PLDR (an 8-byte predicate). The exported wrapper
  2455  // (generated Go) bounds-checks the slice — and panics if it is too short — so
  2456  // this raw intrinsic never reads past it. args are (s).
  2457  func sveLoadWhole() intrinsicBuilder {
  2458  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2459  		ptr, _ := slicePtrLen(s, args[0])
  2460  		return s.newValue2(ssaop.OpLoad, n.Type(), ptr, s.mem())
  2461  	}
  2462  }
  2463  
  2464  // sveStoreWhole is the store counterpart of sveLoadWhole: x.store(s) /
  2465  // m.store(bits). args are (x, s).
  2466  func sveStoreWhole() intrinsicBuilder {
  2467  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2468  		ptr, _ := slicePtrLen(s, args[1])
  2469  		s.vars[memVar] = s.newValue3A(ssaop.OpStore, types.TypeMem, args[0].Type, ptr, args[0], s.mem())
  2470  		return nil
  2471  	}
  2472  }
  2473  
  2474  // sveLoadPart builds the raw predicated load loadTPart(s): a PWHILELT-governed
  2475  // ZLD1B that reads len(s) elements (inactive lanes zeroed). It is byte-granular
  2476  // for every element type, which is correct because arm64 is little-endian, so a
  2477  // contiguous byte copy preserves element layout. The exported LoadTPart wrapper
  2478  // (generated Go) passes s[:min(len(s), Len())] and handles the empty/nil case,
  2479  // so this intrinsic never sees a length past the slice or the vector.
  2480  func sveLoadPart(elemBytes int64) intrinsicBuilder {
  2481  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2482  		ptr, length := slicePtrLen(s, args[0])
  2483  		mask := s.newValue1(ssaop.OpCount8s, types.TypeMask, sveByteCount(s, length, elemBytes))
  2484  		return s.newValue3(ssaop.OpLoadMasked8, n.Type(), ptr, mask, s.mem())
  2485  	}
  2486  }
  2487  
  2488  // sveStorePart is the store counterpart of sveLoadPart: x.storePart(s). args are
  2489  // (x, s).
  2490  func sveStorePart(elemBytes int64) intrinsicBuilder {
  2491  	return func(s *state, n *ir.CallExpr, args []*ssa.Value) *ssa.Value {
  2492  		ptr, length := slicePtrLen(s, args[1])
  2493  		mask := s.newValue1(ssaop.OpCount8s, types.TypeMask, sveByteCount(s, length, elemBytes))
  2494  		s.vars[memVar] = s.newValue4A(ssaop.OpStoreMasked8, types.TypeMem, args[0].Type, ptr, mask, args[0], s.mem())
  2495  		return nil
  2496  	}
  2497  }
  2498  
  2499  // findIntrinsic returns a function which builds the SSA equivalent of the
  2500  // function identified by the symbol sym.  If sym is not an intrinsic call, returns nil.
  2501  func findIntrinsic(sym *types.Sym) intrinsicBuilder {
  2502  	if sym == nil || sym.Pkg == nil {
  2503  		return nil
  2504  	}
  2505  	pkg := sym.Pkg.Path
  2506  	if sym.Pkg == ir.Pkgs.Runtime {
  2507  		pkg = "runtime"
  2508  	}
  2509  	if base.Flag.Race && pkg == "sync/atomic" {
  2510  		// The race detector needs to be able to intercept these calls.
  2511  		// We can't intrinsify them.
  2512  		return nil
  2513  	}
  2514  	// Skip intrinsifying math functions (which may contain hard-float
  2515  	// instructions) when soft-float
  2516  	if Arch.SoftFloat && pkg == "math" {
  2517  		return nil
  2518  	}
  2519  
  2520  	fn := sym.Name
  2521  	if ssaconfig.IntrinsicsDisable {
  2522  		if pkg == "internal/runtime/sys" && (fn == "GetCallerPC" || fn == "GetCallerSP" || fn == "GetClosurePtr") ||
  2523  			pkg == simdPackage {
  2524  			// These runtime functions don't have definitions, must be intrinsics.
  2525  		} else {
  2526  			return nil
  2527  		}
  2528  	}
  2529  	return intrinsics.lookup(Arch.LinkArch.Arch, pkg, fn)
  2530  }
  2531  
  2532  func IsIntrinsicCall(n *ir.CallExpr) bool {
  2533  	if n == nil {
  2534  		return false
  2535  	}
  2536  	name, ok := n.Fun.(*ir.Name)
  2537  	if !ok {
  2538  		if n.Fun.Op() == ir.OMETHEXPR {
  2539  			if meth := ir.MethodExprName(n.Fun); meth != nil {
  2540  				if fn := meth.Func; fn != nil {
  2541  					return IsIntrinsicSym(fn.Sym())
  2542  				}
  2543  			}
  2544  		}
  2545  		return false
  2546  	}
  2547  	return IsIntrinsicSym(name.Sym())
  2548  }
  2549  
  2550  func IsIntrinsicSym(sym *types.Sym) bool {
  2551  	return findIntrinsic(sym) != nil
  2552  }
  2553  
  2554  // GenIntrinsicBody generates the function body for a bodyless intrinsic.
  2555  // This is used when the intrinsic is used in a non-call context, e.g.
  2556  // as a function pointer, or (for a method) being referenced from the type
  2557  // descriptor.
  2558  //
  2559  // The compiler already recognizes a call to fn as an intrinsic and can
  2560  // directly generate code for it. So we just fill in the body with a call
  2561  // to fn.
  2562  func GenIntrinsicBody(fn *ir.Func) {
  2563  	if ir.CurFunc != nil {
  2564  		base.FatalfAt(fn.Pos(), "enqueueFunc %v inside %v", fn, ir.CurFunc)
  2565  	}
  2566  
  2567  	if base.Flag.LowerR != 0 {
  2568  		fmt.Println("generate intrinsic for", ir.FuncName(fn))
  2569  	}
  2570  
  2571  	pos := fn.Pos()
  2572  	ft := fn.Type()
  2573  	var ret ir.Node
  2574  
  2575  	// For a method, it usually starts with an ODOTMETH (pre-typecheck) or
  2576  	// OMETHEXPR (post-typecheck) referencing the method symbol without the
  2577  	// receiver type, and Walk rewrites it to a call directly to the
  2578  	// type-qualified method symbol, moving the receiver to an argument.
  2579  	// Here fn has already the type-qualified method symbol, and it is hard
  2580  	// to get the unqualified symbol. So we just generate the post-Walk form
  2581  	// and mark it typechecked and Walked.
  2582  	call := ir.NewCallExpr(pos, ir.OCALLFUNC, fn.Nname, nil)
  2583  	call.Args = ir.RecvParamNames(ft)
  2584  	call.IsDDD = ft.IsVariadic()
  2585  	typecheck.Exprs(call.Args)
  2586  	call.SetTypecheck(1)
  2587  	call.SetWalked(true)
  2588  	ret = call
  2589  	if ft.NumResults() > 0 {
  2590  		if ft.NumResults() == 1 {
  2591  			call.SetType(ft.Result(0).Type)
  2592  		} else {
  2593  			call.SetType(ft.ResultsTuple())
  2594  		}
  2595  		n := ir.NewReturnStmt(base.Pos, nil)
  2596  		n.Results = []ir.Node{call}
  2597  		ret = n
  2598  	}
  2599  	fn.Body.Append(ret)
  2600  
  2601  	if base.Flag.LowerR != 0 {
  2602  		ir.DumpList("generate intrinsic body", fn.Body)
  2603  	}
  2604  
  2605  	ir.CurFunc = fn
  2606  	typecheck.Stmts(fn.Body)
  2607  	ir.CurFunc = nil // we know CurFunc is nil at entry
  2608  }
  2609  

View as plain text