Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 10 additions & 2 deletions src/encoding/hex/hex.go
Original file line number Diff line number Diff line change
Expand Up @@ -43,8 +43,12 @@ func EncodedLen(n int) int { return n * 2 }
// of bytes written to dst, but this value is always [EncodedLen](len(src)).
// Encode implements hexadecimal encoding.
func Encode(dst, src []byte) int {
j := 0
for _, v := range src {
i := 0
if haveSIMD && len(src) >= 16 {
i = encodeSIMD(dst, src)
}
j := 2 * i
for _, v := range src[i:] {
dst[j] = hextable[v>>4]
dst[j+1] = hextable[v&0x0f]
j += 2
Expand Down Expand Up @@ -86,6 +90,10 @@ func DecodedLen(x int) int { return x / 2 }
// of bytes decoded before the error.
func Decode(dst, src []byte) (int, error) {
i, j := 0, 0
if haveSIMD && len(src) >= 32 {
j = decodeSIMD(dst, src)
i = j / 2
}
for ; j < len(src)-1; j += 2 {
p := src[j]
q := src[j+1]
Expand Down
17 changes: 17 additions & 0 deletions src/encoding/hex/hex_nosimd.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,17 @@
// Copyright 2026 The Go Authors. All rights reserved.
// Use of this source code is governed by a BSD-style
// license that can be found in the LICENSE file.

//go:build !((amd64 || arm64) && goexperiment.simd)

package hex

const haveSIMD = false

func encodeSIMD(dst, src []byte) int {
panic("unreachable")
}

func decodeSIMD(dst, src []byte) int {
panic("unreachable")
}
93 changes: 93 additions & 0 deletions src/encoding/hex/hex_simd_amd64.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,93 @@
// Copyright 2026 The Go Authors. All rights reserved.
// Use of this source code is governed by a BSD-style
// license that can be found in the LICENSE file.

//go:build goexperiment.simd

package hex

import "simd/archsimd"

// haveSIMD reports whether the CPU supports the AVX2 instructions
// used by encodeSIMD and decodeSIMD.
var haveSIMD = archsimd.X86.AVX2()

var hexDigits32 = [32]uint8{
'0', '1', '2', '3', '4', '5', '6', '7', '8', '9', 'a', 'b', 'c', 'd', 'e', 'f',
'0', '1', '2', '3', '4', '5', '6', '7', '8', '9', 'a', 'b', 'c', 'd', 'e', 'f',
}

// encodeSIMD encodes the whole 16-byte blocks of src that fit in dst
// and returns the number of bytes of src it encoded.
func encodeSIMD(dst, src []byte) int {
n := len(src)
digits := archsimd.LoadUint8x32Array(&hexDigits32)
spread := archsimd.BroadcastUint16x16(0x1001)
for len(src) >= 32 && len(dst) >= 64 {
encodeBlock(digits, spread, (*[16]uint8)(src), (*[32]uint8)(dst))
encodeBlock(digits, spread, (*[16]uint8)(src[16:]), (*[32]uint8)(dst[32:]))
src, dst = src[32:], dst[64:]
}
if len(src) >= 16 && len(dst) >= 32 {
encodeBlock(digits, spread, (*[16]uint8)(src), (*[32]uint8)(dst))
src = src[16:]
}
archsimd.ClearAVXUpperBits()
return n - len(src)
}

// encodeBlock widens each byte 0xHL of src to the uint16 0x00HL.
// Multiplied by 0x1001 that is 0xL0HL, and shifted right by 4 it is
// 0x0L0H: the two nibbles in output order, which one in-lane byte
// shuffle turns into digits.
func encodeBlock(digits archsimd.Uint8x32, spread archsimd.Uint16x16, src *[16]uint8, dst *[32]uint8) {
w := archsimd.LoadUint8x16Array(src).ExtendToUint16().Mul(spread).ShiftAllRight(4)
digits.PermuteOrZeroGrouped(w.AsUint8x32().AsInt8x32()).StoreArray(dst)
}

// pairWeights makes VPMADDUBSW compute 16*first + second for each pair
// of nibbles.
var pairWeights = [32]int8{
16, 1, 16, 1, 16, 1, 16, 1, 16, 1, 16, 1, 16, 1, 16, 1,
16, 1, 16, 1, 16, 1, 16, 1, 16, 1, 16, 1, 16, 1, 16, 1,
}

// packBytes moves the low byte of each uint16 of the low 128-bit lane
// to bytes 0-7 and of the high lane to bytes 8-15.
var packBytes = [32]int8{
0, 2, 4, 6, 8, 10, 12, 14, -1, -1, -1, -1, -1, -1, -1, -1,
-1, -1, -1, -1, -1, -1, -1, -1, 0, 2, 4, 6, 8, 10, 12, 14,
}

// decodeSIMD decodes the whole 32-character blocks of src that fit in
// dst, up to the first block holding a non-hex character, and returns
// the number of characters of src it decoded.
//
// It uses algorithm 3 of
// http://0x80.pl/notesen/2022-01-17-validating-hex-parse.html:
// a digit maps to 0-9 and a letter of either case to 10-15, anything
// else to more than 15 on both paths, so the smaller of the two is the
// nibble.
func decodeSIMD(dst, src []byte) int {
n := len(src)
c6 := archsimd.BroadcastUint8x32(0xc6)
six := archsimd.BroadcastUint8x32(6)
f0 := archsimd.BroadcastUint8x32(0xf0)
upper := archsimd.BroadcastUint8x32(0xdf)
bigA := archsimd.BroadcastUint8x32('A')
ten := archsimd.BroadcastUint8x32(10)
weights := archsimd.LoadInt8x32Array(&pairWeights)
pack := archsimd.LoadInt8x32Array(&packBytes)
for len(src) >= 32 && len(dst) >= 16 {
c := archsimd.LoadUint8x32Array((*[32]uint8)(src))
nib := c.Add(c6).SubSaturated(six).Sub(f0).Min(c.And(upper).Sub(bigA).AddSaturated(ten))
if !nib.And(f0).IsZero() {
break
}
b := nib.DotProductPairsSaturated(weights).AsUint8x32().PermuteOrZeroGrouped(pack)
b.GetLo().Or(b.GetHi()).StoreArray((*[16]uint8)(dst))
src, dst = src[32:], dst[16:]
}
archsimd.ClearAVXUpperBits()
return n - len(src)
}
56 changes: 56 additions & 0 deletions src/encoding/hex/hex_simd_arm64.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,56 @@
// Copyright 2026 The Go Authors. All rights reserved.
// Use of this source code is governed by a BSD-style
// license that can be found in the LICENSE file.

//go:build goexperiment.simd

package hex

import "simd/archsimd"

// haveSIMD is true because NEON is part of the arm64 baseline.
const haveSIMD = true

var hexDigits16 = [16]uint8{'0', '1', '2', '3', '4', '5', '6', '7', '8', '9', 'a', 'b', 'c', 'd', 'e', 'f'}

// encodeSIMD encodes the whole 16-byte blocks of src that fit in dst
// and returns the number of bytes of src it encoded.
func encodeSIMD(dst, src []byte) int {
n := len(src)
digits := archsimd.LoadUint8x16Array(&hexDigits16)
lowNibble := archsimd.BroadcastUint8x16(0x0f)
for len(src) >= 16 && len(dst) >= 32 {
v := archsimd.LoadUint8x16Array((*[16]uint8)(src))
hi, lo := digits.LookupOrZero(v.ShiftAllRight(4)), digits.LookupOrZero(v.And(lowNibble))
hi.InterleaveLo(lo).StoreArray((*[16]uint8)(dst))
hi.InterleaveHi(lo).StoreArray((*[16]uint8)(dst[16:]))
src, dst = src[16:], dst[32:]
}
return n - len(src)
}

// decodeSIMD decodes the whole 32-character blocks of src that fit in
// dst, up to the first block holding a non-hex character, and returns
// the number of characters of src it decoded. It computes the nibbles
// as decodeSIMD does on amd64.
func decodeSIMD(dst, src []byte) int {
n := len(src)
c6 := archsimd.BroadcastUint8x16(0xc6)
six := archsimd.BroadcastUint8x16(6)
f0 := archsimd.BroadcastUint8x16(0xf0)
upper := archsimd.BroadcastUint8x16(0xdf)
bigA := archsimd.BroadcastUint8x16('A')
ten := archsimd.BroadcastUint8x16(10)
for len(src) >= 32 && len(dst) >= 16 {
c1 := archsimd.LoadUint8x16Array((*[16]uint8)(src))
c2 := archsimd.LoadUint8x16Array((*[16]uint8)(src[16:]))
n1 := c1.Add(c6).SubSaturated(six).Sub(f0).Min(c1.And(upper).Sub(bigA).AddSaturated(ten))
n2 := c2.Add(c6).SubSaturated(six).Sub(f0).Min(c2.And(upper).Sub(bigA).AddSaturated(ten))
if n1.Max(n2).ReduceMax() > 15 {
break
}
n1.ConcatEven(n2).ShiftAllLeft(4).Or(n1.ConcatOdd(n2)).StoreArray((*[16]uint8)(dst))
src, dst = src[32:], dst[16:]
}
return n - len(src)
}
97 changes: 97 additions & 0 deletions src/encoding/hex/hex_simd_portable.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,97 @@
// Copyright 2026 The Go Authors. All rights reserved.
// Use of this source code is governed by a BSD-style
// license that can be found in the LICENSE file.

//go:build goexperiment.simd

package hex

import (
"internal/byteorder"
"simd"
)

// encodePortable and decodePortable are encodeSIMD and decodeSIMD written
// with the portable simd package. Only tests and benchmarks use them, to
// compare the two.

// encodePortable encodes the whole vectors of src that fit in dst and
// returns the number of bytes of src it encoded. The simd package cannot
// move bytes between lanes, so each 64-bit lane computes the digits of its
// own 8 bytes, and scalar stores put the two halves in order.
func encodePortable(dst, src []byte) int {
vl := simd.VectorBitSize() / 8
if simd.Emulated() || len(src) < vl {
return 0
}
n := len(src)
var lo, hi [8]uint64
low32 := simd.BroadcastUint64s(0xffffffff)
m16 := simd.BroadcastUint64s(0x0000ffff0000ffff)
m8 := simd.BroadcastUint64s(0x00ff00ff00ff00ff)
nibble := simd.BroadcastUint64s(0x000f000f000f000f)
six := simd.BroadcastUint64s(0x0606060606060606)
ones := simd.BroadcastUint64s(0x0101010101010101)
zero := simd.BroadcastUint64s(0x3030303030303030)
gap := simd.BroadcastUint64s(0x2727272727272727)
// digits spreads the 4 bytes in the low half of each lane to 16-bit
// slots, puts their nibbles in output order and adds '0', or 'a'-10
// to the nibbles above 9.
digits := func(x simd.Uint64s) simd.Uint64s {
x = x.Or(x.ShiftAllLeft(16)).And(m16)
x = x.Or(x.ShiftAllLeft(8)).And(m8)
n := x.ShiftAllRight(4).And(nibble).Or(x.And(nibble).ShiftAllLeft(8))
m := n.Add(six).ShiftAllRight(4).And(ones)
return n.Add(zero).Add(m.ShiftAllLeft(8).Sub(m).And(gap))
}
for len(src) >= vl && len(dst) >= 2*vl {
x := simd.LoadUint8s(src).ReshapeToUint64s()
digits(x.And(low32)).Store(lo[:])
digits(x.ShiftAllRight(32)).Store(hi[:])
for k := range vl / 8 {
byteorder.LEPutUint64(dst[16*k:], lo[k])
byteorder.LEPutUint64(dst[16*k+8:], hi[k])
}
src, dst = src[vl:], dst[2*vl:]
}
return n - len(src)
}

// decodePortable decodes the whole vectors of src that fit in dst, up to
// the first vector holding a non-hex character, and returns the number of
// characters of src it decoded. It computes the nibbles as decodeSIMD does
// and packs each 64-bit lane in place; the simd package cannot narrow, so
// scalar stores join the 4-byte halves.
func decodePortable(dst, src []byte) int {
vl := simd.VectorBitSize() / 8
if simd.Emulated() || len(src) < vl {
return 0
}
n := len(src)
var out [8]uint64
c6 := simd.BroadcastUint8s(0xc6)
six := simd.BroadcastUint8s(6)
f0 := simd.BroadcastUint8s(0xf0)
upper := simd.BroadcastUint8s(0xdf)
bigA := simd.BroadcastUint8s('A')
ten := simd.BroadcastUint8s(10)
fifteen := simd.BroadcastUint8s(15)
m8 := simd.BroadcastUint64s(0x00ff00ff00ff00ff)
m16 := simd.BroadcastUint64s(0x0000ffff0000ffff)
for len(src) >= vl && len(dst) >= vl/2 {
c := simd.LoadUint8s(src)
nib := c.Add(c6).SubSaturated(six).Sub(f0).Min(c.And(upper).Sub(bigA).AddSaturated(ten))
if !nib.Max(fifteen).Equal(fifteen).All() {
break
}
v := nib.ReshapeToUint64s()
v = v.ShiftAllLeft(4).Or(v.ShiftAllRight(8)).And(m8)
v = v.Or(v.ShiftAllRight(8)).And(m16)
v.Or(v.ShiftAllRight(16)).Store(out[:])
for k := range vl / 16 {
byteorder.LEPutUint64(dst[8*k:], out[2*k]&0xffffffff|out[2*k+1]<<32)
}
src, dst = src[vl:], dst[vl/2:]
}
return n - len(src)
}
Loading
Loading