# Example 1 - code generation

## In this learning path

- [Introduction](https://learn.arm.com/learning-paths/cross-platform/function-multiversioning/)
- [About function multiversioning](https://learn.arm.com/learning-paths/cross-platform/function-multiversioning/semantics/)
- [Example 1 - code generation](https://learn.arm.com/learning-paths/cross-platform/function-multiversioning/examples/)
- [Example 2 - runtime using ACLE intrinsics](https://learn.arm.com/learning-paths/cross-platform/function-multiversioning/examples2/)
- [Example 3 - inline assembly at runtime](https://learn.arm.com/learning-paths/cross-platform/function-multiversioning/examples3/)
- [Compatibility with streaming mode](https://learn.arm.com/learning-paths/cross-platform/function-multiversioning/streaming-mode/)
- [Further information on implementation](https://learn.arm.com/learning-paths/cross-platform/function-multiversioning/implementation-details/)
- [Changes from released compilers](https://learn.arm.com/learning-paths/cross-platform/function-multiversioning/changes-from-past-releases/)
- [Next Steps](https://learn.arm.com/learning-paths/cross-platform/function-multiversioning/_next-steps/)

This example specifies two versions of `sumPosEltsScaledByIndex` using the `target_clones` attribute. The order in which they are listed does not matter.

At certain optimization levels, compilers can decide to perform loop vectorization depending on the target’s vector capabilities.

The intention is to enable the compiler to use SVE instructions in the specialized case, while restricting it to use only Armv8 instructions in the default case.

Use a text editor to create a file named `loop.c` with the code below:

```c
__attribute__((target_clones("sve", "default")))
int sumPosEltsScaledByIndex(int *v, unsigned n) {
  int s = 0;
  for (unsigned i = 0; i < n; ++i)
    if (v[i] > 0)
      s += v[i] * i;
  return s;
}
```

You can use either Clang or GCC to compile the code example.

To compile with Clang, run:

```bash
clang --target=aarch64-linux-gnu -march=armv8-a -O3 --rtlib=compiler-rt -S -o - loop.c
```

To compile with GCC, use:

```bash
gcc -march=armv8-a -O3 -S -o - loop.c
```

> **Note**
> When using the `clang` compiler, specify the option `--rtlib=compiler-rt` on the command line. This allows the compiler to generate runtime checks for detecting the presence of hardware features.

Here is the generated compiler output for the SVE version of `sumPosEltsScaledByIndex` (using `clang`):

```assembly
__output__ .text
__output__   .globl	sumPosEltsScaledByIndex._Msve
__output__   .p2align	2
__output__   .type	sumPosEltsScaledByIndex._Msve,@function
__output__ sumPosEltsScaledByIndex._Msve:
__output__   cbz	w1, .LBB0_3
__output__   mov	w9, w1
__output__   cntw	x8
__output__   cmp	x8, x9
__output__   b.ls	.LBB0_4
__output__   mov	x10, xzr
__output__   mov	w8, wzr
__output__   b	.LBB0_7
__output__ .LBB0_3:
__output__   mov	w8, wzr
__output__   mov	w0, w8
__output__   ret
__output__ .LBB0_4:
__output__   ptrue	p0.s
__output__   mov	z0.s, #0
__output__   index	z2.s, #0, #1
__output__   cntw	x10
__output__   sub	x12, x8, #1
__output__   rdvl	x13, #1
__output__   mov	z1.s, w10
__output__   and	x12, x9, x12
__output__   mov	x11, xzr
__output__   mov	z3.d, z0.d
__output__   sub	x10, x9, x12
__output__   add	x13, x0, x13
__output__ .LBB0_5:
__output__   ld1w	{ z4.s }, p0/z, [x0, x11, lsl #2]
__output__   ld1w	{ z5.s }, p0/z, [x13, x11, lsl #2]
__output__   add	z6.s, z2.s, z1.s
__output__   add	x11, x11, x8
__output__   cmpgt	p1.s, p0/z, z4.s, #0
__output__   cmpgt	p2.s, p0/z, z5.s, #0
__output__   cmp	x10, x11
__output__   mla	z0.s, p1/m, z4.s, z2.s
__output__   mla	z3.s, p2/m, z5.s, z6.s
__output__   add	z2.s, z6.s, z1.s
__output__   b.ne	.LBB0_5
__output__   add	z0.s, z3.s, z0.s
__output__   uaddv	d0, p0, z0.s
__output__   fmov	x8, d0
__output__   cbz	x12, .LBB0_8
__output__ .LBB0_7:
__output__   ldr	w11, [x0, x10, lsl #2]
__output__   mul	w12, w11, w10
__output__   cmp	w11, #0
__output__   add	x10, x10, #1
__output__   csel	w11, w12, wzr, gt
__output__   cmp	x9, x10
__output__   add	w8, w11, w8
__output__   b.ne	.LBB0_7
__output__ .LBB0_8:
__output__   mov	w0, w8
__output__   ret
```

This is the default version of `sumPosEltsScaledByIndex`:

```assembly
__output__ .section	.rodata.cst16,"aM",@progbits,16
__output__   .p2align	4, 0x0
__output__ .LCPI2_0:
__output__   .word	0
__output__   .word	1
__output__   .word	2
__output__   .word	3
__output__ .text
__output__   .globl	sumPosEltsScaledByIndex.default
__output__   .p2align	2
__output__   .type	sumPosEltsScaledByIndex.default,@function
__output__ sumPosEltsScaledByIndex.default:
__output__   cbz	w1, .LBB2_3
__output__   cmp	w1, #8
__output__   mov	w9, w1
__output__   b.hs	.LBB2_4
__output__   mov	x10, xzr
__output__   mov	w8, wzr
__output__   b	.LBB2_7
__output__ .LBB2_3:
__output__   mov	w0, wzr
__output__   ret
__output__ .LBB2_4:
__output__   movi	v0.2d, #0000000000000000
__output__   movi	v1.4s, #4
__output__   adrp	x8, .LCPI2_0
__output__   movi	v2.4s, #8
__output__   movi	v3.2d, #0000000000000000
__output__   and	x10, x9, #0xfffffff8
__output__   ldr	q4, [x8, :lo12:.LCPI2_0]
__output__   add	x8, x0, #16
__output__   mov	x11, x10
__output__ .LBB2_5:
__output__   add	v5.4s, v4.4s, v1.4s
__output__   ldp	q6, q7, [x8, #-16]
__output__   subs	x11, x11, #8
__output__   add	x8, x8, #32
__output__   mul	v16.4s, v6.4s, v4.4s
__output__   cmgt	v6.4s, v6.4s, #0
__output__   add	v4.4s, v4.4s, v2.4s
__output__   mul	v5.4s, v7.4s, v5.4s
__output__   cmgt	v7.4s, v7.4s, #0
__output__   and	v6.16b, v16.16b, v6.16b
__output__   and	v5.16b, v5.16b, v7.16b
__output__   add	v0.4s, v6.4s, v0.4s
__output__   add	v3.4s, v5.4s, v3.4s
__output__   b.ne	.LBB2_5
__output__   add	v0.4s, v3.4s, v0.4s
__output__   cmp	x10, x9
__output__   addv	s0, v0.4s
__output__   fmov	w8, s0
__output__   b.eq	.LBB2_8
__output__ .LBB2_7:
__output__   ldr	w11, [x0, x10, lsl #2]
__output__   mul	w12, w11, w10
__output__   cmp	w11, #0
__output__   add	x10, x10, #1
__output__   csel	w11, w12, wzr, gt
__output__   cmp	x9, x10
__output__   add	w8, w11, w8
__output__   b.ne	.LBB2_7
__output__ .LBB2_8:
__output__   mov	w0, w8
__output__   ret
```

Any calls to `sumPosEltsScaledByIndex` are routed through `sumPosEltsScaledByIndex.resolver`. This is the function which contains the runtime checks for feature detection.

```assembly
__output__ .section	.text.sumPosEltsScaledByIndex.resolver,"axG",@progbits,sumPosEltsScaledByIndex.resolver,comdat
__output__   .weak	sumPosEltsScaledByIndex.resolver
__output__   .p2align	2
__output__   .type	sumPosEltsScaledByIndex.resolver,@function
__output__ sumPosEltsScaledByIndex.resolver:
__output__   str	x30, [sp, #-16]!
__output__   bl	__init_cpu_features_resolver
__output__   adrp	x8, __aarch64_cpu_features+3
__output__   adrp	x9, sumPosEltsScaledByIndex._Msve
__output__   add	x9, x9, :lo12:sumPosEltsScaledByIndex._Msve
__output__   ldrb	w8, [x8, :lo12:__aarch64_cpu_features+3]
__output__   tst	w8, #0x40
__output__   adrp	x8, sumPosEltsScaledByIndex.default
__output__   add	x8, x8, :lo12:sumPosEltsScaledByIndex.default
__output__   csel	x0, x8, x9, eq
__output__   ldr	x30, [sp], #16
__output__   ret
```

The called symbol `sumPosEltsScaledByIndex` is an indirect function (ifunc) which points to the resolver.

```assembly
__output__ .weak	sumPosEltsScaledByIndex
__output__ .type	sumPosEltsScaledByIndex,@gnu_indirect_function
__output__ .set sumPosEltsScaledByIndex, sumPosEltsScaledByIndex.resolver
```

The names `sumPosEltsScaledByIndex._Msve` and `sumPosEltsScaledByIndex.default` correspond to the function versions of `sumPosEltsScaledByIndex`.

See the [Arm C Language Extensions](https://arm-software.github.io/acle/main/acle.html#name-mangling) for further information on the name mangling rules.
