Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 4 additions & 6 deletions src/ImageSharp/Common/Helpers/ColorNumerics.cs
Original file line number Diff line number Diff line change
Expand Up @@ -16,17 +16,15 @@ internal static class ColorNumerics
/// Vector for converting pixel to gray value as specified by
/// ITU-R Recommendation BT.709.
/// </summary>
private static readonly Vector4 Bt709 = new(.2126f, .7152f, .0722f, 0.0f);
public static readonly Vector4 Bt709 = new(.2126f, .7152f, .0722f, 0.0f);

/// <summary>
/// Convert a pixel value to grayscale using ITU-R Recommendation BT.709.
/// Gets unrounded, unsaturated luminance using ITU-R Recommendation BT.709.
/// </summary>
/// <param name="vector">The vector to get the luminance from.</param>
/// <param name="luminanceLevels">
/// The number of luminance levels (256 for 8 bit, 65536 for 16 bit grayscale images).
/// </param>
/// <returns>The unrounded luminance.</returns>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public static int GetBT709Luminance(Vector4 vector, int luminanceLevels) => (int)MathF.Round(Vector4.Dot(vector, Bt709) * (luminanceLevels - 1));
public static float GetBT709Luminance(Vector4 vector) => Vector4.Dot(vector, Bt709);

/// <summary>
/// Gets the luminance from the rgb components using the formula
Expand Down
30 changes: 30 additions & 0 deletions src/ImageSharp/Common/Helpers/Numerics.cs
Original file line number Diff line number Diff line change
Expand Up @@ -1066,6 +1066,36 @@ public static nuint Vector512Count<TVector>(int length)
where TVector : struct
=> (uint)length / (uint)Vector512<TVector>.Count;

/// <summary>
/// Gets the count of vectors that safely fit into a span whose element type matches the vector lane type.
/// </summary>
/// <typeparam name="TVector">The type of the span elements and vector lanes.</typeparam>
/// <param name="span">The given span.</param>
/// <returns>Count of vectors that safely fit into the span.</returns>
public static nuint Vector128Count<TVector>(this ReadOnlySpan<TVector> span)
where TVector : struct
=> (uint)span.Length / (uint)Vector128<TVector>.Count;

/// <summary>
/// Gets the count of vectors that safely fit into a span whose element type matches the vector lane type.
/// </summary>
/// <typeparam name="TVector">The type of the span elements and vector lanes.</typeparam>
/// <param name="span">The given span.</param>
/// <returns>Count of vectors that safely fit into the span.</returns>
public static nuint Vector256Count<TVector>(this ReadOnlySpan<TVector> span)
where TVector : struct
=> (uint)span.Length / (uint)Vector256<TVector>.Count;

/// <summary>
/// Gets the count of vectors that safely fit into a span whose element type matches the vector lane type.
/// </summary>
/// <typeparam name="TVector">The type of the span elements and vector lanes.</typeparam>
/// <param name="span">The given span.</param>
/// <returns>Count of vectors that safely fit into the span.</returns>
public static nuint Vector512Count<TVector>(this ReadOnlySpan<TVector> span)
where TVector : struct
=> (uint)span.Length / (uint)Vector512<TVector>.Count;

/// <summary>
/// Clamps a floating-point component while mapping NaN to the lower bound.
/// </summary>
Expand Down
150 changes: 150 additions & 0 deletions src/ImageSharp/Common/Helpers/SimdUtils.FloatPlanes.cs
Original file line number Diff line number Diff line change
@@ -0,0 +1,150 @@
// Copyright (c) Six Labors.
// Licensed under the Six Labors Split License.

using System.Numerics;
using System.Runtime.CompilerServices;
using System.Runtime.InteropServices;
using System.Runtime.Intrinsics;
using SixLabors.ImageSharp.Common.Helpers;

namespace SixLabors.ImageSharp;

/// <content>
/// Converts planar floating-point components to and from contiguous four-component vectors.
/// </content>
internal static partial class SimdUtils
{
/// <summary>
/// Interleaves three equally sized component planes and an optional fourth plane into four-component vectors.
/// Values are copied without changing their floating-point representation.
/// </summary>
/// <param name="component0">The values for <see cref="Vector4.X"/>.</param>
/// <param name="component1">The values for <see cref="Vector4.Y"/>.</param>
/// <param name="component2">The values for <see cref="Vector4.Z"/>.</param>
/// <param name="component3">The values for <see cref="Vector4.W"/>, or an empty span to use 1 for every value.</param>
/// <param name="destination">The destination vectors.</param>
internal static void InterleaveFloatPlanes(
ReadOnlySpan<float> component0,
ReadOnlySpan<float> component1,
ReadOnlySpan<float> component2,
ReadOnlySpan<float> component3,
Span<Vector4> destination)
{
Guard.IsTrue(component1.Length == component0.Length, nameof(component1), "Components must be of same size!");
Guard.IsTrue(component2.Length == component0.Length, nameof(component2), "Components must be of same size!");
Guard.IsTrue(component3.IsEmpty || component3.Length == component0.Length, nameof(component3), "Components must be of same size!");
Guard.DestinationShouldNotBeTooShort(component0, destination, nameof(destination));

ref float c0 = ref MemoryMarshal.GetReference(component0);
ref float c1 = ref MemoryMarshal.GetReference(component1);
ref float c2 = ref MemoryMarshal.GetReference(component2);
ref float c3 = ref MemoryMarshal.GetReference(component3);
ref float d = ref Unsafe.As<Vector4, float>(ref MemoryMarshal.GetReference(destination));
bool hasComponent3 = !component3.IsEmpty;
int i = 0;

if (Vector128.IsHardwareAccelerated)
{
// Load four values from each plane: v0=[c0.0,c0.1,c0.2,c0.3] through
// v3=[c3.0,c3.1,c3.2,c3.3]. An absent fourth plane supplies four 1.0 values.
// The unpack helpers transpose component bits without arithmetic, preserving
// signed zero, infinities, and NaN payload bits.
for (; i <= component0.Length - 4; i += 4)
{
Vector128<float> v0 = Vector128.LoadUnsafe(ref c0, (nuint)i);
Vector128<float> v1 = Vector128.LoadUnsafe(ref c1, (nuint)i);
Vector128<float> v2 = Vector128.LoadUnsafe(ref c2, (nuint)i);
Vector128<float> v3 = hasComponent3 ? Vector128.LoadUnsafe(ref c3, (nuint)i) : Vector128.Create(1F);

// The 32-bit zips produce c01Low=[c0.0,c1.0,c0.1,c1.1] and
// c01High=[c0.2,c1.2,c0.3,c1.3], with equivalent pairs for components 2 and 3.
Vector128<float> c01Low = Vector128_.UnpackLow(v0, v1);
Vector128<float> c01High = Vector128_.UnpackHigh(v0, v1);
Vector128<float> c23Low = Vector128_.UnpackLow(v2, v3);
Vector128<float> c23High = Vector128_.UnpackHigh(v2, v3);

// Each 64-bit zip joins the two pairs for one Vector4. The stores emit
// four consecutive vectors in component order 0, 1, 2, 3.
Vector128.StoreUnsafe(Vector128_.UnpackLow(c01Low.AsDouble(), c23Low.AsDouble()).AsSingle(), ref d, (nuint)(i * 4));
Vector128.StoreUnsafe(Vector128_.UnpackHigh(c01Low.AsDouble(), c23Low.AsDouble()).AsSingle(), ref d, (nuint)((i + 1) * 4));
Vector128.StoreUnsafe(Vector128_.UnpackLow(c01High.AsDouble(), c23High.AsDouble()).AsSingle(), ref d, (nuint)((i + 2) * 4));
Vector128.StoreUnsafe(Vector128_.UnpackHigh(c01High.AsDouble(), c23High.AsDouble()).AsSingle(), ref d, (nuint)((i + 3) * 4));
}
}

// Fewer than four remaining pixels cannot be loaded as a full register. The tail
// writes the identical component order, including the implicit fourth value.
for (; i < component0.Length; i++)
{
destination[i] = new Vector4(component0[i], component1[i], component2[i], hasComponent3 ? component3[i] : 1F);
}
}

/// <summary>
/// Deinterleaves four-component vectors into equally sized component planes.
/// Values are copied without changing their floating-point representation.
/// </summary>
/// <param name="source">The source vectors.</param>
/// <param name="component0">The destination for <see cref="Vector4.X"/> values.</param>
/// <param name="component1">The destination for <see cref="Vector4.Y"/> values.</param>
/// <param name="component2">The destination for <see cref="Vector4.Z"/> values.</param>
/// <param name="component3">The destination for <see cref="Vector4.W"/> values.</param>
internal static void DeinterleaveFloatPlanes(
ReadOnlySpan<Vector4> source,
Span<float> component0,
Span<float> component1,
Span<float> component2,
Span<float> component3)
{
Guard.IsTrue(component1.Length == component0.Length, nameof(component1), "Components must be of same size!");
Guard.IsTrue(component2.Length == component0.Length, nameof(component2), "Components must be of same size!");
Guard.IsTrue(component3.Length == component0.Length, nameof(component3), "Components must be of same size!");
Guard.DestinationShouldNotBeTooShort(source, component0, nameof(component0));

ref float s = ref Unsafe.As<Vector4, float>(ref MemoryMarshal.GetReference(source));
ref float c0 = ref MemoryMarshal.GetReference(component0);
ref float c1 = ref MemoryMarshal.GetReference(component1);
ref float c2 = ref MemoryMarshal.GetReference(component2);
ref float c3 = ref MemoryMarshal.GetReference(component3);
int i = 0;

if (Vector128.IsHardwareAccelerated)
{
// Each loaded register is one vector: p0=[c0.0,c1.0,c2.0,c3.0] through
// p3=[c0.3,c1.3,c2.3,c3.3]. The unpack helpers make this a bitwise
// transpose, preserving nonfinite values and signed zero exactly.
for (; i <= source.Length - 4; i += 4)
{
Vector128<float> p0 = Vector128.LoadUnsafe(ref s, (nuint)(i * 4));
Vector128<float> p1 = Vector128.LoadUnsafe(ref s, (nuint)((i + 1) * 4));
Vector128<float> p2 = Vector128.LoadUnsafe(ref s, (nuint)((i + 2) * 4));
Vector128<float> p3 = Vector128.LoadUnsafe(ref s, (nuint)((i + 3) * 4));

// The first 32-bit zips pair adjacent vectors. c01Low contains
// [c0.0,c0.1,c1.0,c1.1], and c01High contains the next two vectors.
// c23Low and c23High hold the corresponding third and fourth components.
Vector128<float> c01Low = Vector128_.UnpackLow(p0, p1);
Vector128<float> c01High = Vector128_.UnpackLow(p2, p3);
Vector128<float> c23Low = Vector128_.UnpackHigh(p0, p1);
Vector128<float> c23High = Vector128_.UnpackHigh(p2, p3);

// A 64-bit zip combines the matching two-component groups into
// four values for each component plane in source vector order.
Vector128.StoreUnsafe(Vector128_.UnpackLow(c01Low.AsDouble(), c01High.AsDouble()).AsSingle(), ref c0, (nuint)i);
Vector128.StoreUnsafe(Vector128_.UnpackHigh(c01Low.AsDouble(), c01High.AsDouble()).AsSingle(), ref c1, (nuint)i);
Vector128.StoreUnsafe(Vector128_.UnpackLow(c23Low.AsDouble(), c23High.AsDouble()).AsSingle(), ref c2, (nuint)i);
Vector128.StoreUnsafe(Vector128_.UnpackHigh(c23Low.AsDouble(), c23High.AsDouble()).AsSingle(), ref c3, (nuint)i);
}
}

// The scalar remainder uses the same component mapping for up to three pixels.
for (; i < source.Length; i++)
{
Vector4 value = source[i];
component0[i] = value.X;
component1[i] = value.Y;
component2[i] = value.Z;
component3[i] = value.W;
}
}
}
41 changes: 22 additions & 19 deletions src/ImageSharp/Common/Helpers/SimdUtils.HwIntrinsics.cs
Original file line number Diff line number Diff line change
Expand Up @@ -940,16 +940,20 @@ internal static void FloatToByteSaturate(
ref Vector512<byte> destinationBase = ref Unsafe.As<byte, Vector512<byte>>(ref MemoryMarshal.GetReference(destination));

Vector512<float> scale = Vector512.Create(scaleFactor);
Vector512<float> lowerBound = Vector512<float>.Zero;
Vector512<float> upperBound = Vector512.Create(byte.MaxValue / scaleFactor);
Vector512<int> mask = PermuteMaskDeinterleave16x32();

for (nuint i = 0; i < n; i++)
{
ref Vector512<float> s = ref Unsafe.Add(ref sourceBase, i * 4);

Vector512<float> f0 = scale * s;
Vector512<float> f1 = scale * Unsafe.Add(ref s, 1);
Vector512<float> f2 = scale * Unsafe.Add(ref s, 2);
Vector512<float> f3 = scale * Unsafe.Add(ref s, 3);
// Float-to-int conversion maps infinities and overflow to an invalid integer.
// Clamp in the float domain first so SIMD agrees with the scalar byte saturation rule.
Vector512<float> f0 = scale * Numerics.Clamp(s, lowerBound, upperBound);
Vector512<float> f1 = scale * Numerics.Clamp(Unsafe.Add(ref s, 1), lowerBound, upperBound);
Vector512<float> f2 = scale * Numerics.Clamp(Unsafe.Add(ref s, 2), lowerBound, upperBound);
Vector512<float> f3 = scale * Numerics.Clamp(Unsafe.Add(ref s, 3), lowerBound, upperBound);

Vector512<int> w0 = Vector512_.ConvertToInt32RoundAwayFromZero(f0);
Vector512<int> w1 = Vector512_.ConvertToInt32RoundAwayFromZero(f1);
Expand All @@ -974,16 +978,19 @@ internal static void FloatToByteSaturate(
ref Vector256<byte> destinationBase = ref Unsafe.As<byte, Vector256<byte>>(ref MemoryMarshal.GetReference(destination));

Vector256<float> scale = Vector256.Create(scaleFactor);
Vector256<float> lowerBound = Vector256<float>.Zero;
Vector256<float> upperBound = Vector256.Create(byte.MaxValue / scaleFactor);
Vector256<int> mask = PermuteMaskDeinterleave8x32();

for (nuint i = 0; i < n; i++)
{
ref Vector256<float> s = ref Unsafe.Add(ref sourceBase, i * 4);

Vector256<float> f0 = scale * s;
Vector256<float> f1 = scale * Unsafe.Add(ref s, 1);
Vector256<float> f2 = scale * Unsafe.Add(ref s, 2);
Vector256<float> f3 = scale * Unsafe.Add(ref s, 3);
// Clamp before integer conversion so infinity and overflow reach the byte endpoint.
Vector256<float> f0 = scale * Numerics.Clamp(s, lowerBound, upperBound);
Vector256<float> f1 = scale * Numerics.Clamp(Unsafe.Add(ref s, 1), lowerBound, upperBound);
Vector256<float> f2 = scale * Numerics.Clamp(Unsafe.Add(ref s, 2), lowerBound, upperBound);
Vector256<float> f3 = scale * Numerics.Clamp(Unsafe.Add(ref s, 3), lowerBound, upperBound);

Vector256<int> w0 = Vector256_.ConvertToInt32RoundAwayFromZero(f0);
Vector256<int> w1 = Vector256_.ConvertToInt32RoundAwayFromZero(f1);
Expand All @@ -1009,28 +1016,24 @@ internal static void FloatToByteSaturate(
ref Vector128<byte> destinationBase = ref Unsafe.As<byte, Vector128<byte>>(ref MemoryMarshal.GetReference(destination));

Vector128<float> scale = Vector128.Create(scaleFactor);
Vector128<int> min = Vector128<int>.Zero;
Vector128<int> max = Vector128.Create((int)byte.MaxValue);
Vector128<float> lowerBound = Vector128<float>.Zero;
Vector128<float> upperBound = Vector128.Create(byte.MaxValue / scaleFactor);

for (nuint i = 0; i < n; i++)
{
ref Vector128<float> s = ref Unsafe.Add(ref sourceBase, i * 4);

Vector128<float> f0 = scale * s;
Vector128<float> f1 = scale * Unsafe.Add(ref s, 1);
Vector128<float> f2 = scale * Unsafe.Add(ref s, 2);
Vector128<float> f3 = scale * Unsafe.Add(ref s, 3);
// Clamp before integer conversion so infinity and overflow reach the byte endpoint.
Vector128<float> f0 = scale * Numerics.Clamp(s, lowerBound, upperBound);
Vector128<float> f1 = scale * Numerics.Clamp(Unsafe.Add(ref s, 1), lowerBound, upperBound);
Vector128<float> f2 = scale * Numerics.Clamp(Unsafe.Add(ref s, 2), lowerBound, upperBound);
Vector128<float> f3 = scale * Numerics.Clamp(Unsafe.Add(ref s, 3), lowerBound, upperBound);

Vector128<int> w0 = Vector128_.ConvertToInt32RoundAwayFromZero(f0);
Vector128<int> w1 = Vector128_.ConvertToInt32RoundAwayFromZero(f1);
Vector128<int> w2 = Vector128_.ConvertToInt32RoundAwayFromZero(f2);
Vector128<int> w3 = Vector128_.ConvertToInt32RoundAwayFromZero(f3);

w0 = Vector128.Clamp(w0, min, max);
w1 = Vector128.Clamp(w1, min, max);
w2 = Vector128.Clamp(w2, min, max);
w3 = Vector128.Clamp(w3, min, max);

Vector128<ushort> u0 = Vector128.Narrow(w0, w1).AsUInt16();
Vector128<ushort> u1 = Vector128.Narrow(w2, w3).AsUInt16();

Expand Down
40 changes: 40 additions & 0 deletions src/ImageSharp/Common/Helpers/Vector128Utilities.cs
Original file line number Diff line number Diff line change
Expand Up @@ -582,6 +582,46 @@ public static Vector128<long> UnpackHigh(Vector128<long> left, Vector128<long> r
return Vector128.Create(left.GetUpper(), right.GetUpper());
}

/// <summary>
/// Interleaves the high 64-bit floating-point components of two vectors.
/// </summary>
/// <param name="left">The first vector.</param>
/// <param name="right">The second vector.</param>
/// <returns>The interleaved high components.</returns>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public static Vector128<double> UnpackHigh(Vector128<double> left, Vector128<double> right)
=> UnpackHigh(left.AsInt64(), right.AsInt64()).AsDouble();

/// <summary>
/// Interleaves the low 64-bit floating-point components of two vectors.
/// </summary>
/// <param name="left">The first vector.</param>
/// <param name="right">The second vector.</param>
/// <returns>The interleaved low components.</returns>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public static Vector128<double> UnpackLow(Vector128<double> left, Vector128<double> right)
=> UnpackLow(left.AsInt64(), right.AsInt64()).AsDouble();

/// <summary>
/// Interleaves the high 32-bit floating-point components of two vectors.
/// </summary>
/// <param name="left">The first vector.</param>
/// <param name="right">The second vector.</param>
/// <returns>The interleaved high components.</returns>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public static Vector128<float> UnpackHigh(Vector128<float> left, Vector128<float> right)
=> UnpackHigh(left.AsInt32(), right.AsInt32()).AsSingle();

/// <summary>
/// Interleaves the low 32-bit floating-point components of two vectors.
/// </summary>
/// <param name="left">The first vector.</param>
/// <param name="right">The second vector.</param>
/// <returns>The interleaved low components.</returns>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public static Vector128<float> UnpackLow(Vector128<float> left, Vector128<float> right)
=> UnpackLow(left.AsInt32(), right.AsInt32()).AsSingle();

/// <summary>
/// Unpack and interleave 64-bit integers from the low half of <paramref name="left"/> and <paramref name="right"/>
/// and store the results in the result.
Expand Down
Loading
Loading