Repository navigation
Expand file tree
/
Copy pathexample_encode_test.go
More file actions
201 lines (173 loc) · 6.23 KB
/
Copy pathexample_encode_test.go
File metadata and controls
201 lines (173 loc) · 6.23 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
package simd_test
import (
"fmt"
"github.com/sebishogun/simd"
)
// Encodings: quantization, the narrow float formats, bit packing, run-length
// and the varint widths. These are the operations a column store or an
// inference runtime spends its time in.
// The columnar compression pipeline, in the order the pieces are meant to be
// used: differences first, then zigzag so small negatives stay small, then
// pack to the width that actually fits.
func ExampleBitPackInto() {
a := []uint32{1, 2, 3, 4, 5, 6, 7, 0}
// Three bits hold 0..7, so eight values need 24 bits — one 32-bit word.
// Unpacking needs one word MORE than packing: a value whose bits straddle a
// word boundary reads the next word, and the last one straddles whenever
// the total is not a multiple of 32. Size for the reader.
packed := make([]uint32, (len(a)*3+31)/32+1)
simd.BitPackInto(packed, a, 3)
back := make([]uint32, len(a))
simd.BitUnpackInto(back, packed, 3)
fmt.Println(back)
// Output: [1 2 3 4 5 6 7 0]
}
// VarintSize gives the exact encoded length of a whole slice in one
// vectorized pass, so an encoder sizes its buffer once instead of growing it.
//
// Writing the bytes is serial and always will be — where value i lands depends
// on the width of every value before it — but asking how wide each one is
// vectorizes, and that is the part that lets the allocation happen once.
func ExampleVarintSize() {
a := []uint64{1, 300, 70000}
fmt.Println(simd.VarintSize(a), "bytes")
// Output: 6 bytes
}
// VarintLenInto gives the per-value widths. Prefix-summed, they are every
// value's offset in the stream, which is what makes the writes independent.
func ExampleVarintLenInto() {
lens := make([]int32, 3)
simd.VarintLenInto(lens, []uint64{1, 300, 70000})
fmt.Println(lens)
// Output: [1 2 3]
}
func ExampleAppendVarints() {
buf := simd.AppendVarints(nil, []uint64{1, 300})
fmt.Printf("% x\n", buf)
// Output: 01 ac 02
}
// RunLengthEncodeInt32 turns a run of equal values into a value and a count.
// The run boundaries are found with a vectorized pass; the scratch slice holds
// them so nothing is allocated.
func ExampleRunLengthEncodeInt32() {
a := []int32{7, 7, 7, 9, 9, 4}
values := make([]int32, len(a))
lengths := make([]int32, len(a))
n := simd.RunLengthEncodeInt32(values, lengths, a, make([]bool, len(a)))
fmt.Println(values[:n], lengths[:n])
// Output: [7 9 4] [3 2 1]
}
// RunStartsInto marks every element that begins a run, which is the
// vectorizable half of run-length encoding.
func ExampleRunStartsInto() {
dst := make([]bool, 6)
simd.RunStartsInto(dst, []int32{7, 7, 7, 9, 9, 4})
fmt.Println(dst)
// Output: [true false false true false true]
}
// Quantization to int8 with a scale and zero point: the operation an inference
// runtime does to every weight and activation.
func ExampleQuantizePerChannelInt8() {
// Two channels of two values, each with its own scale.
a := []float32{1, 2, 30, 40}
scale := []float32{0.5, 10}
zero := []int32{0, 0}
dst := make([]int8, len(a))
simd.QuantizePerChannelInt8(dst, a, scale, zero, 2, 2)
fmt.Println(dst)
// Output: [2 4 3 4]
}
// The int8 matrix multiply accumulates into int32, because the products of
// two int8 values overflow int8 immediately. Requantize brings it back down.
func ExampleQMatMulInt8Into() {
// [1 2] * [1 0] = [1 2]
// [3 4] [0 1] [3 4]
a := []int8{1, 2, 3, 4}
b := []int8{1, 0, 0, 1}
acc := make([]int32, 4)
simd.QMatMulInt8Into(acc, a, b, 2, 2, 2)
fmt.Println(acc)
// q = round(acc*scale) + zeroPoint, rounding half to EVEN — so 1*0.5
// becomes 0 and 3*0.5 becomes 2, which is what the runtimes this
// interoperates with do and what round-half-away-from-zero would not.
out := make([]int8, 4)
simd.RequantizeInt8Into(out, acc, 0.5, 0)
fmt.Println(out)
// Output:
// [1 2 3 4]
// [0 1 2 2]
}
// float16 and bfloat16 are storage formats: half the bytes, and the
// conversion is what the vector unit is for.
func ExampleFloat16ToFloat32Into() {
dst := make([]float32, 2)
simd.Float16ToFloat32Into(dst, []uint16{0x3c00, 0x4000}) // 1.0, 2.0
fmt.Println(dst)
// Output: [1 2]
}
// e4m3 trades exponent range for mantissa, which is the trade inference
// weights want. It has no infinity: the largest value is 448.
func ExampleFloat32ToFloat8E4M3Into() {
dst := make([]byte, 2)
simd.Float32ToFloat8E4M3Into(dst, []float32{1, 2})
back := make([]float32, 2)
simd.Float8E4M3ToFloat32Into(back, dst)
fmt.Println(back)
// Output: [1 2]
}
// Grayscale uses the libjpeg BT.601 weights in Q16, so it agrees with every
// other implementation of the same conversion to the bit.
func ExampleGrayscaleInto() {
dst := make([]byte, 3)
simd.GrayscaleInto(dst,
[]byte{255, 0, 0}, // r
[]byte{0, 255, 0}, // g
[]byte{0, 0, 255}) // b
fmt.Println(dst)
// Output: [76 150 29]
}
func ExampleRGBToUVInto() {
u := make([]byte, 1)
v := make([]byte, 1)
simd.RGBToUVInto(u, v, []byte{255}, []byte{0}, []byte{0})
fmt.Println(u[0], v[0])
// Output: 85 255
}
// A counter-based generator: element i depends on i alone, which is what makes
// it vectorizable, reproducible across architectures, and splittable across
// goroutines without any shared state.
func ExampleRandomInto() {
a := make([]float64, 4)
simd.RandomInto(a, 42)
b := make([]float64, 4)
simd.RandomInto(b, 42)
fmt.Println(a[0] == b[0], a[0] >= 0 && a[0] < 1)
// Output: true true
}
// RequantizeInt8Into brings an int32 accumulator back to int8:
// q = round(acc*scale) + zeroPoint, rounding half to even and saturating
// rather than wrapping.
func ExampleRequantizeInt8Into() {
dst := make([]int8, 4)
simd.RequantizeInt8Into(dst, []int32{100, 200, 300, 100000}, 0.5, 0)
fmt.Println(dst)
// Output: [50 100 127 127]
}
func ExampleZigzagEncodeInt32Into() {
deltas := []int32{0, -1, 1, -2, 2}
enc := make([]uint32, len(deltas))
simd.ZigzagEncodeInt32Into(enc, deltas)
// Small magnitudes of either sign became small unsigned values, which is
// what makes the varint of each one a single byte.
fmt.Println(enc)
// Output: [0 1 2 3 4]
}
func ExampleQuantizeInt8() {
// A symmetric per-tensor scale, the common case for weights.
w := []float32{-1.0, -0.5, -0.25, 0, 0.25, 0.5, 1.0}
scale := float32(1.0 / 127)
q := make([]int8, len(w))
simd.QuantizeInt8(q, w, scale, 0)
fmt.Println(q)
// Output: [-127 -64 -32 0 32 64 127]
}