simd/motion

Code

main.odin ¶
147 linesSource

1package simd_example
2
3import "base:intrinsics"
4import "core:fmt"
5import "core:math/rand"
6import "core:simd"
7import "core:time"
8
9NUM_DATA :: #config(NUM, 10_000_000) // The amount of data points to generate and process
10NUM_REPETITIONS :: #config(REP, 100) // The number of times to run each proc, for performance measurement
11WIDTH :: #config(WIDTH, 16)
12
13// A single moving point. A point has a position and a velocity, and in this example they will
14// travel in a straight line and bounce when reaching the edge of a bounding rectangle.
15Point :: struct {
16	// Components are separate fields so that they'll be stored in separate arrays when used as #soa
17	pos_x, pos_y, vel_x, vel_y: f32,
18}
19
20// Updates a SoA slice of points, with given the bounds and delta-time.
21// This implementation uses straightforward scalar logic.
22update_points_scalar :: proc (points: #soa[]Point, bounds: [2][2]f32, dt: f32) {
23	for &point in points {
24		if point.pos_x <= bounds[0].x { point.vel_x = +abs(point.vel_x) }
25		if point.pos_x >= bounds[1].x { point.vel_x = -abs(point.vel_x) }
26		if point.pos_y <= bounds[0].y { point.vel_y = +abs(point.vel_y) }
27		if point.pos_y >= bounds[1].y { point.vel_y = -abs(point.vel_y) }
28
29		point.pos_x += dt * point.vel_x
30		point.pos_y += dt * point.vel_y
31	}
32}
33
34// Updates a SoA slice of points, with given the bounds and delta-time
35// This implementation uses SIMD logic.
36update_points_simd :: proc (points: #soa[]Point, bounds: [2][2]f32, dt: f32) {
37	process_chunk :: proc (points: #soa[]Point, mask: #simd[WIDTH]u32, bounds: [2][2]f32, dt: f32) {
38		px_ptr := cast(^#simd[WIDTH]f32)points.pos_x
39		py_ptr := cast(^#simd[WIDTH]f32)points.pos_y
40		vx_ptr := cast(^#simd[WIDTH]f32)points.vel_x
41		vy_ptr := cast(^#simd[WIDTH]f32)points.vel_y
42
43		// Read the vectors of positions and velocities, as specified by the mask. Any values that
44		// are masked out are just populated with 0 (from the second parameter).
45		// 
46		// Each component is in a separate vector, so the math is vaguely similar to the math for
47		// doing just one at a time, as opposed to the "vectors" in the traditional sense
48		// corresponding to a SIMD vector. This allows for better parallelism, especially in 2D.
49		// Using a SIMD vector to represent a 2D vector would give relatively little benefit (2x
50		// parallelism at most, often less), whereas using a SIMD vector to represent a single
51		// component across multiple vectors allows this to scale (processing 4/8/16+ points at a
52		// time).
53		px := simd.masked_load(px_ptr, cast(#simd[WIDTH]f32)0, mask)
54		py := simd.masked_load(py_ptr, cast(#simd[WIDTH]f32)0, mask)
55		vx := simd.masked_load(vx_ptr, cast(#simd[WIDTH]f32)0, mask)
56		vy := simd.masked_load(vy_ptr, cast(#simd[WIDTH]f32)0, mask)
57
58		// Select elements where the X position is less than the minimum, replace the corresponding
59		// velocity with its absolute value.
60		// 
61		// Rather than doing this with normal conditionals, it's done via lane-wise comparisons and
62		// select. This means the absolute value is calculated for every element regardless of
63		// whether it's used, but this allows for better parallelism.
64		// 
65		// If the computation is something more expensive, you can use a conditional to simplify
66		// cases where all/none are matched via simd.reduce_and/reduce_or on the mask, but in this
67		// case it's not worth it.
68		min_x_mask := simd.lanes_le(px, cast(#simd[WIDTH]f32)bounds[0].x)
69		vx = simd.select(min_x_mask, +simd.abs(vx), vx)
70
71		// As above, but with the Y component
72		min_y_mask := simd.lanes_le(py, cast(#simd[WIDTH]f32)bounds[0].y)
73		vy = simd.select(min_y_mask, +simd.abs(vy), vy)
74
75		// As above, but with the maximums
76		max_x_mask := simd.lanes_ge(px, cast(#simd[WIDTH]f32)bounds[1].x)
77		vx = simd.select(max_x_mask, -simd.abs(vx), vx)
78
79		max_y_mask := simd.lanes_ge(py, cast(#simd[WIDTH]f32)bounds[1].y)
80		vy = simd.select(max_y_mask, -simd.abs(vy), vy)
81
82		// Updating the positions, at least, is straightforward
83		px += dt * vx
84		py += dt * vy
85
86		// Write the modified positions/velocities back
87		// As with reads, only the positions that are selected in the mask are actually written to
88		// in memory
89		simd.masked_store(px_ptr, px, mask)
90		simd.masked_store(py_ptr, py, mask)
91		simd.masked_store(vx_ptr, vx, mask)
92		simd.masked_store(vy_ptr, vy, mask)
93	}
94
95	points := points
96	for len(points) >= WIDTH {
97		process_chunk(points, max(u32), bounds, dt)
98		points = points[WIDTH:]
99	}
100
101	if len(points) > 0 {
102		index := iota(#simd[WIDTH]i32)
103		mask := simd.lanes_lt(index, cast(#simd[WIDTH]i32)len(points))
104		process_chunk(points, mask, bounds, dt)
105	}
106}
107
108main :: proc() {
109	if ODIN_OPTIMIZATION_MODE <= .Minimal {
110		fmt.println("WARNING: For best results, run benchmarks in an optimized build!")
111	}
112
113	bounds := [2][2]f32 {
114		{-100, -100},
115		{+100, +100},
116	}
117	dt := f32(0.1)
118
119	points := make(#soa[]Point, NUM_DATA, context.temp_allocator)
120	for &point in points {
121		point.pos_x = rand.float32_range(bounds[0].x, bounds[1].x)
122		point.pos_y = rand.float32_range(bounds[0].y, bounds[1].y)
123		point.vel_x = rand.float32_range(-1, +1)
124		point.vel_y = rand.float32_range(-1, +1)
125	}
126
127	fmt.printfln("Motion (Scalar): %v", benchmark(update_points_scalar, points, bounds, dt))
128	fmt.printfln("Motion (SIMD): %v", benchmark(update_points_simd, points, bounds, dt))
129}
130
131benchmark :: proc (p: proc (points: #soa[]Point, bounds: [2][2]f32, dt: f32), points: #soa[]Point, bounds: [2][2]f32, dt: f32) -> time.Duration {
132	best_elapsed := max(time.Duration)
133	for _ in 0..<NUM_REPETITIONS {
134		start := time.tick_now()
135		p(points, bounds, dt)
136		best_elapsed = min(time.tick_since(start), best_elapsed)
137	}
138	return best_elapsed
139}
140
141iota :: proc ($V: typeid/#simd[$N]$E) -> (result: V) {
142	for i in 0..<N {
143		result = simd.replace(result, i, E(i))
144	}
145	return
146}
147

Declarations Used 14