mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-01 14:58:14 +00:00
Merge pull request #21131 from hrydgard/more-morph-optimizations
Optimize non-JIT morph vertex decoders
This commit is contained in:
4 files changed
+227
-83
No files matched your search
+64
-17
@@ -202,14 +202,6 @@ struct Vec4F32 {
|
||||
bits = _mm_srai_epi32(_mm_unpacklo_epi16(bits, bits), 16);
|
||||
return Vec4F32 { _mm_mul_ps(_mm_cvtepi32_ps(bits), _mm_set1_ps(1.0f / 32768.0f)) };
|
||||
}
|
||||
void Store(float *dst) { _mm_storeu_ps(dst, v); }
|
||||
void Store2(float *dst) { _mm_storel_epi64((__m128i *)dst, _mm_castps_si128(v)); }
|
||||
void StoreAligned (float *dst) { _mm_store_ps(dst, v); }
|
||||
void Store3(float *dst) {
|
||||
// TODO: There might be better ways.
|
||||
_mm_store_pd((double *)dst, _mm_castps_pd(v));
|
||||
_mm_store_ss(dst + 2, _mm_shuffle_ps(v, v, _MM_SHUFFLE(2, 2, 2, 2)));
|
||||
}
|
||||
|
||||
static Vec4F32 LoadConvertS16(const int16_t *src) { // Note: will load 8 bytes
|
||||
__m128i value = _mm_loadl_epi64((const __m128i *)src);
|
||||
@@ -224,6 +216,15 @@ struct Vec4F32 {
|
||||
return Vec4F32{ _mm_cvtepi32_ps(_mm_srai_epi32(_mm_unpacklo_epi16(value16, value16), 24)) };
|
||||
}
|
||||
|
||||
// NOTE: Does not normalize to 0..255 range.
|
||||
static Vec4F32 LoadConvertU8(const uint8_t *src) { // Note: will load 8 bytes
|
||||
__m128i value = _mm_loadl_epi64((const __m128i *)src);
|
||||
__m128i zero = _mm_setzero_si128();
|
||||
__m128i value16 = _mm_unpacklo_epi8(value, zero);
|
||||
// 16-bit to 32-bit, use the upper words and an arithmetic shift right to sign extend
|
||||
return Vec4F32{ _mm_cvtepi32_ps(_mm_unpacklo_epi16(value16, zero)) };
|
||||
}
|
||||
|
||||
static Vec4F32 LoadF24x3_One(const uint32_t *src) {
|
||||
alignas(16) static const uint32_t mask[4] = { 0xFFFFFFFF, 0xFFFFFFFF, 0xFFFFFFFF, 0x0 };
|
||||
alignas(16) static const float onelane3[4] = { 0.0f, 0.0f, 0.0f, 1.0f };
|
||||
@@ -232,6 +233,20 @@ struct Vec4F32 {
|
||||
return Vec4F32{ _mm_or_ps(_mm_and_ps(value, _mm_load_ps((const float *)mask)), _mm_load_ps(onelane3)) };
|
||||
}
|
||||
|
||||
void Store(float *dst) { _mm_storeu_ps(dst, v); }
|
||||
void Store2(float *dst) { _mm_storel_epi64((__m128i *)dst, _mm_castps_si128(v)); }
|
||||
void StoreAligned(float *dst) { _mm_store_ps(dst, v); }
|
||||
void Store3(float *dst) {
|
||||
// This seems to be the best way with SSE2.
|
||||
_mm_storel_pd((double *)dst, _mm_castps_pd(v));
|
||||
_mm_store_ss(dst + 2, _mm_shuffle_ps(v, v, _MM_SHUFFLE(2, 2, 2, 2)));
|
||||
}
|
||||
void StoreConvertToU8(uint8_t *dst) {
|
||||
__m128i zero = _mm_setzero_si128();
|
||||
__m128i ivalue = _mm_packus_epi16(_mm_packs_epi32(_mm_cvttps_epi32(v), zero), zero);
|
||||
_mm_storeu_si32(dst, ivalue);
|
||||
}
|
||||
|
||||
static Vec4F32 FromVec4S32(Vec4S32 other) { return Vec4F32{ _mm_cvtepi32_ps(other.v) }; }
|
||||
|
||||
Vec4F32 operator +(Vec4F32 other) const { return Vec4F32{ _mm_add_ps(v, other.v) }; }
|
||||
@@ -244,7 +259,8 @@ struct Vec4F32 {
|
||||
void operator *=(Vec4F32 other) { v = _mm_mul_ps(v, other.v); }
|
||||
void operator /=(Vec4F32 other) { v = _mm_div_ps(v, other.v); }
|
||||
void operator &=(Vec4S32 other) { v = _mm_and_ps(v, _mm_castsi128_ps(other.v)); }
|
||||
Vec4F32 operator *(float f) const { return Vec4F32{ _mm_mul_ps(v, _mm_set1_ps(f)) }; }
|
||||
Vec4F32 operator *(float f) const { return Vec4F32{_mm_mul_ps(v, _mm_set1_ps(f))}; }
|
||||
void operator *=(float f) { v = _mm_mul_ps(v, _mm_set1_ps(f)); }
|
||||
// NOTE: May be slow.
|
||||
float operator[](size_t index) const { return ((float *)&v)[index]; }
|
||||
|
||||
@@ -298,6 +314,10 @@ struct Vec4F32 {
|
||||
Vec4S32 CompareEq(Vec4F32 other) const { return Vec4S32{ _mm_castps_si128(_mm_cmpeq_ps(v, other.v)) }; }
|
||||
Vec4S32 CompareLt(Vec4F32 other) const { return Vec4S32{ _mm_castps_si128(_mm_cmplt_ps(v, other.v)) }; }
|
||||
Vec4S32 CompareGt(Vec4F32 other) const { return Vec4S32{ _mm_castps_si128(_mm_cmpgt_ps(v, other.v)) }; }
|
||||
|
||||
template<int i> float GetLane() const {
|
||||
return _mm_cvtss_f32(_mm_shuffle_ps(v, v, _MM_SHUFFLE(i, i, i, i)));
|
||||
}
|
||||
};
|
||||
|
||||
inline Vec4S32 Vec4S32FromF32(Vec4F32 f) { return Vec4S32{ _mm_cvttps_epi32(f.v) }; }
|
||||
@@ -530,14 +550,6 @@ struct Vec4F32 {
|
||||
return Vec4F32 { vcvtq_n_f32_s32(vmovl_s16(vld1_s16(src)), 15) };
|
||||
}
|
||||
static Vec4F32 LoadAligned(const float *src) { return Vec4F32{ vld1q_f32(src) }; }
|
||||
void Store(float *dst) { vst1q_f32(dst, v); }
|
||||
void Store2(float *dst) { vst1_f32(dst, vget_low_f32(v)); }
|
||||
void StoreAligned(float *dst) { vst1q_f32(dst, v); }
|
||||
void Store3(float *dst) {
|
||||
// TODO: There might be better ways. Try to avoid this when possible.
|
||||
vst1_f32(dst, vget_low_f32(v));
|
||||
dst[2] = vgetq_lane_f32(v, 2);
|
||||
}
|
||||
|
||||
static Vec4F32 LoadConvertS16(const int16_t *src) {
|
||||
int16x4_t value = vld1_s16(src);
|
||||
@@ -550,6 +562,12 @@ struct Vec4F32 {
|
||||
return Vec4F32{ vcvtq_f32_s32(vmovl_s16(value16)) };
|
||||
}
|
||||
|
||||
static Vec4F32 LoadConvertU8(const uint8_t *src) { // Note: will load 8 bytes, not 4. Only the first 4 bytes will be used.
|
||||
uint8x8_t value = vld1_u8(src);
|
||||
uint16x4_t value16 = vget_low_u16(vmovl_u8(value));
|
||||
return Vec4F32{ vcvtq_f32_u32(vmovl_u16(value16)) };
|
||||
}
|
||||
|
||||
static Vec4F32 LoadF24x3_One(const uint32_t *src) {
|
||||
return Vec4F32{ vsetq_lane_f32(1.0f, vreinterpretq_f32_u32(vshlq_n_u32(vld1q_u32(src), 8)), 3) };
|
||||
}
|
||||
@@ -558,6 +576,26 @@ struct Vec4F32 {
|
||||
return Vec4F32{ vcvtq_f32_s32(other.v) };
|
||||
}
|
||||
|
||||
void Store(float *dst) { vst1q_f32(dst, v); }
|
||||
void Store2(float *dst) { vst1_f32(dst, vget_low_f32(v)); }
|
||||
void StoreAligned(float *dst) { vst1q_f32(dst, v); }
|
||||
void Store3(float *dst) {
|
||||
// TODO: There might be better ways. Try to avoid this when possible.
|
||||
vst1_f32(dst, vget_low_f32(v));
|
||||
#if PPSSPP_ARCH(ARM64_NEON)
|
||||
vst1q_lane_f32(dst + 2, v, 2);
|
||||
#else
|
||||
dst[2] = vgetq_lane_f32(v, 2);
|
||||
#endif
|
||||
}
|
||||
void StoreConvertToU8(uint8_t *dest) {
|
||||
uint32x4_t ivalue32 = vcvtq_u32_f32(v);
|
||||
uint16x4_t ivalue16 = vqmovn_u32(ivalue32);
|
||||
uint8x8_t ivalue8 = vqmovn_u16(vcombine_u16(ivalue16, ivalue16)); // Is there no way to avoid the combine here?
|
||||
uint32_t value = vget_lane_u32(vreinterpret_u32_u8(ivalue8), 0);
|
||||
memcpy(dest, &value, sizeof(uint32_t));
|
||||
}
|
||||
|
||||
// NOTE: May be slow.
|
||||
float operator[](size_t index) const { return ((float *)&v)[index]; }
|
||||
|
||||
@@ -577,6 +615,7 @@ struct Vec4F32 {
|
||||
#endif
|
||||
void operator &=(Vec4S32 other) { v = vreinterpretq_f32_s32(vandq_s32(vreinterpretq_s32_f32(v), other.v)); }
|
||||
Vec4F32 operator *(float f) const { return Vec4F32{ vmulq_f32(v, vdupq_n_f32(f)) }; }
|
||||
void operator *=(float f) { v = vmulq_f32(v, vdupq_n_f32(f)); }
|
||||
|
||||
Vec4F32 Mul(float f) const { return Vec4F32{ vmulq_f32(v, vdupq_n_f32(f)) }; }
|
||||
|
||||
@@ -659,6 +698,10 @@ struct Vec4F32 {
|
||||
#endif
|
||||
return Vec4F32{ sum };
|
||||
}
|
||||
|
||||
template<int i> float GetLane() const {
|
||||
return vgetq_lane_f32(v, i);
|
||||
}
|
||||
};
|
||||
|
||||
inline Vec4S32 Vec4S32FromF32(Vec4F32 f) { return Vec4S32{ vcvtq_s32_f32(f.v) }; }
|
||||
@@ -1109,6 +1152,10 @@ struct Vec4F32 {
|
||||
|
||||
return Vec4F32{ { x, y, z, 1.0f } };
|
||||
}
|
||||
|
||||
template<int i> float GetLane() const {
|
||||
return v[i];
|
||||
}
|
||||
};
|
||||
|
||||
inline bool AnyZeroSignBit(Vec4S32 value) {
|
||||
|
||||
@@ -23,6 +23,7 @@
|
||||
|
||||
#include "Common/CommonTypes.h"
|
||||
#include "Common/Data/Convert/ColorConv.h"
|
||||
#include "Common/Math/CrossSIMD.h"
|
||||
#include "Common/Log.h"
|
||||
#include "Common/LogReporting.h"
|
||||
#include "Core/Config.h"
|
||||
@@ -414,57 +415,67 @@ void VertexDecoder::Step_TcFloatPrescale(const VertexDecoder *dec, const u8 *ptr
|
||||
void VertexDecoder::Step_TcU8MorphToFloat(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float uv[2]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
float w = gstate_c.morphWeights[n];
|
||||
const u8 *uvdata = (const u8 *)(ptr + dec->onesize_*n + dec->tcoff);
|
||||
const int onesize = dec->onesize_;
|
||||
const u8 *uvdata = (const u8 *)(ptr + dec->tcoff);
|
||||
|
||||
uv[0] += (float)uvdata[0] * (1.f / 128.f) * w;
|
||||
uv[1] += (float)uvdata[1] * (1.f / 128.f) * w;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
const float w = gstate_c.morphWeights[n];
|
||||
|
||||
uv[0] += (float)uvdata[0] * w;
|
||||
uv[1] += (float)uvdata[1] * w;
|
||||
|
||||
uvdata += onesize;
|
||||
}
|
||||
|
||||
float *out = (float *)(decoded + dec->decFmt.uvoff);
|
||||
out[0] = uv[0];
|
||||
out[1] = uv[1];
|
||||
out[0] = uv[0] * (1.f / 128.f);
|
||||
out[1] = uv[1] * (1.f / 128.f);
|
||||
}
|
||||
|
||||
// Just two channels, barely worth SIMD.
|
||||
void VertexDecoder::Step_TcU16MorphToFloat(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float uv[2]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
float w = gstate_c.morphWeights[n];
|
||||
const u16_le *uvdata = (const u16_le *)(ptr + dec->onesize_*n + dec->tcoff);
|
||||
const int onesize = dec->onesize_;
|
||||
const u8 *b_uvdata = ptr + dec->tcoff;
|
||||
|
||||
uv[0] += (float)uvdata[0] * (1.f / 32768.f) * w;
|
||||
uv[1] += (float)uvdata[1] * (1.f / 32768.f) * w;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
const float w = gstate_c.morphWeights[n];
|
||||
const u16 *uvdata = (const u16 *)(b_uvdata);
|
||||
uv[0] += (float)uvdata[0] * w;
|
||||
uv[1] += (float)uvdata[1] * w;
|
||||
b_uvdata += onesize;
|
||||
}
|
||||
|
||||
float *out = (float *)(decoded + dec->decFmt.uvoff);
|
||||
out[0] = uv[0];
|
||||
out[1] = uv[1];
|
||||
out[0] = uv[0] * (1.f / 32768.f);
|
||||
out[1] = uv[1] * (1.f / 32768.f);
|
||||
}
|
||||
|
||||
void VertexDecoder::Step_TcU16DoubleMorphToFloat(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float uv[2]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
float w = gstate_c.morphWeights[n];
|
||||
const u16_le *uvdata = (const u16_le *)(ptr + dec->onesize_*n + dec->tcoff);
|
||||
const float w = gstate_c.morphWeights[n];
|
||||
const u16_le *uvdata = (const u16_le *)(ptr + onesize * n + dec->tcoff);
|
||||
|
||||
uv[0] += (float)uvdata[0] * (1.f / 16384.f) * w;
|
||||
uv[1] += (float)uvdata[1] * (1.f / 16384.f) * w;
|
||||
uv[0] += (float)uvdata[0] * w;
|
||||
uv[1] += (float)uvdata[1] * w;
|
||||
}
|
||||
|
||||
float *out = (float *)(decoded + dec->decFmt.uvoff);
|
||||
out[0] = uv[0];
|
||||
out[1] = uv[1];
|
||||
out[0] = uv[0] * (1.f / 16384.f);
|
||||
out[1] = uv[1] * (1.f / 16384.f);
|
||||
}
|
||||
|
||||
void VertexDecoder::Step_TcFloatMorph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float uv[2]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
float w = gstate_c.morphWeights[n];
|
||||
const float_le *uvdata = (const float_le *)(ptr + dec->onesize_*n + dec->tcoff);
|
||||
const float w = gstate_c.morphWeights[n];
|
||||
const float_le *uvdata = (const float_le *)(ptr + onesize*n + dec->tcoff);
|
||||
|
||||
uv[0] += (float)uvdata[0] * w;
|
||||
uv[1] += (float)uvdata[1] * w;
|
||||
@@ -478,57 +489,65 @@ void VertexDecoder::Step_TcFloatMorph(const VertexDecoder *dec, const u8 *ptr, u
|
||||
void VertexDecoder::Step_TcU8PrescaleMorph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float uv[2]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
const u8 *uvdata = (const u8 *)(ptr + dec->tcoff);
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
const float w = gstate_c.morphWeights[n] * (1.f / 128.f);
|
||||
const u8 *uvdata = (const u8 *)(ptr + dec->onesize_*n + dec->tcoff);
|
||||
const float w = gstate_c.morphWeights[n];
|
||||
|
||||
uv[0] += (float)uvdata[0] * w;
|
||||
uv[1] += (float)uvdata[1] * w;
|
||||
}
|
||||
|
||||
uvdata += onesize;
|
||||
}
|
||||
float *out = (float *)(decoded + dec->decFmt.uvoff);
|
||||
out[0] = uv[0] * dec->prescaleUV_->uScale + dec->prescaleUV_->uOff;
|
||||
out[1] = uv[1] * dec->prescaleUV_->vScale + dec->prescaleUV_->vOff;
|
||||
out[0] = uv[0] * dec->prescaleUV_->uScale * (1.f / 128.f) + dec->prescaleUV_->uOff;
|
||||
out[1] = uv[1] * dec->prescaleUV_->vScale * (1.f / 128.f) + dec->prescaleUV_->vOff;
|
||||
}
|
||||
|
||||
void VertexDecoder::Step_TcU16PrescaleMorph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float uv[2]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
const u8 *b_uvdata = ptr + dec->tcoff;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
const float w = gstate_c.morphWeights[n] * (1.f / 32768.f);
|
||||
const u16_le *uvdata = (const u16_le *)(ptr + dec->onesize_*n + dec->tcoff);
|
||||
const float w = gstate_c.morphWeights[n];
|
||||
const u16_le *uvdata = (const u16_le *)(b_uvdata);
|
||||
|
||||
uv[0] += (float)uvdata[0] * w;
|
||||
uv[1] += (float)uvdata[1] * w;
|
||||
|
||||
b_uvdata += onesize;
|
||||
}
|
||||
|
||||
float *out = (float *)(decoded + dec->decFmt.uvoff);
|
||||
out[0] = uv[0] * dec->prescaleUV_->uScale + dec->prescaleUV_->uOff;
|
||||
out[1] = uv[1] * dec->prescaleUV_->vScale + dec->prescaleUV_->vOff;
|
||||
out[0] = uv[0] * dec->prescaleUV_->uScale * (1.f / 32768.f) + dec->prescaleUV_->uOff;
|
||||
out[1] = uv[1] * dec->prescaleUV_->vScale * (1.f / 32768.f) + dec->prescaleUV_->vOff;
|
||||
}
|
||||
|
||||
void VertexDecoder::Step_TcU16DoublePrescaleMorph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float uv[2]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
const float w = gstate_c.morphWeights[n] * (1.f / 16384.f);
|
||||
const u16_le *uvdata = (const u16_le *)(ptr + dec->onesize_*n + dec->tcoff);
|
||||
const float w = gstate_c.morphWeights[n];
|
||||
const u16_le *uvdata = (const u16_le *)(ptr + onesize * n + dec->tcoff);
|
||||
|
||||
uv[0] += (float)uvdata[0] * w;
|
||||
uv[1] += (float)uvdata[1] * w;
|
||||
}
|
||||
|
||||
float *out = (float *)(decoded + dec->decFmt.uvoff);
|
||||
out[0] = uv[0] * dec->prescaleUV_->uScale + dec->prescaleUV_->uOff;
|
||||
out[1] = uv[1] * dec->prescaleUV_->vScale + dec->prescaleUV_->vOff;
|
||||
out[0] = uv[0] * dec->prescaleUV_->uScale * (1.f / 16384.f) + dec->prescaleUV_->uOff;
|
||||
out[1] = uv[1] * dec->prescaleUV_->vScale * (1.f / 16384.f) + dec->prescaleUV_->vOff;
|
||||
}
|
||||
|
||||
void VertexDecoder::Step_TcFloatPrescaleMorph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float uv[2] = { 0, 0 };
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
float w = gstate_c.morphWeights[n];
|
||||
const float_le *uvdata = (const float_le *)(ptr + dec->onesize_*n + dec->tcoff);
|
||||
const float w = gstate_c.morphWeights[n];
|
||||
const float_le *uvdata = (const float_le *)(ptr + onesize * n + dec->tcoff);
|
||||
|
||||
uv[0] += (float)uvdata[0] * w;
|
||||
uv[1] += (float)uvdata[1] * w;
|
||||
@@ -580,9 +599,10 @@ void VertexDecoder::Step_Color8888(const VertexDecoder *dec, const u8 *ptr, u8 *
|
||||
void VertexDecoder::Step_Color565Morph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float col[3]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
float w = gstate_c.morphWeights[n];
|
||||
u16 cdata = *(const u16_le *)(ptr + dec->onesize_*n + dec->coloff);
|
||||
u16 cdata = *(const u16_le *)(ptr + onesize * n + dec->coloff);
|
||||
col[0] += w * (cdata & 0x1f) * (255.0f / 31.0f);
|
||||
col[1] += w * ((cdata >> 5) & 0x3f) * (255.0f / 63.0f);
|
||||
col[2] += w * ((cdata >> 11) & 0x1f) * (255.0f / 31.0f);
|
||||
@@ -598,9 +618,10 @@ void VertexDecoder::Step_Color565Morph(const VertexDecoder *dec, const u8 *ptr,
|
||||
void VertexDecoder::Step_Color5551Morph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float col[4]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
float w = gstate_c.morphWeights[n];
|
||||
u16 cdata = *(const u16_le *)(ptr + dec->onesize_*n + dec->coloff);
|
||||
u16 cdata = *(const u16_le *)(ptr + onesize * n + dec->coloff);
|
||||
col[0] += w * (cdata & 0x1f) * (255.0f / 31.0f);
|
||||
col[1] += w * ((cdata >> 5) & 0x1f) * (255.0f / 31.0f);
|
||||
col[2] += w * ((cdata >> 10) & 0x1f) * (255.0f / 31.0f);
|
||||
@@ -610,15 +631,16 @@ void VertexDecoder::Step_Color5551Morph(const VertexDecoder *dec, const u8 *ptr,
|
||||
for (int i = 0; i < 4; i++) {
|
||||
c[i] = clamp_u8((int)col[i]);
|
||||
}
|
||||
gstate_c.vertexFullAlpha = gstate_c.vertexFullAlpha && (int)col[3] >= 255;
|
||||
gstate_c.vertexFullAlpha = gstate_c.vertexFullAlpha && col[3] >= 255.0f;
|
||||
}
|
||||
|
||||
void VertexDecoder::Step_Color4444Morph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float col[4]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
float w = gstate_c.morphWeights[n];
|
||||
u16 cdata = *(const u16_le *)(ptr + dec->onesize_*n + dec->coloff);
|
||||
u16 cdata = *(const u16_le *)(ptr + onesize * n + dec->coloff);
|
||||
for (int j = 0; j < 4; j++)
|
||||
col[j] += w * ((cdata >> (j * 4)) & 0xF) * (255.0f / 15.0f);
|
||||
}
|
||||
@@ -626,23 +648,40 @@ void VertexDecoder::Step_Color4444Morph(const VertexDecoder *dec, const u8 *ptr,
|
||||
for (int i = 0; i < 4; i++) {
|
||||
c[i] = clamp_u8((int)col[i]);
|
||||
}
|
||||
gstate_c.vertexFullAlpha = gstate_c.vertexFullAlpha && (int)col[3] >= 255;
|
||||
gstate_c.vertexFullAlpha = gstate_c.vertexFullAlpha && col[3] >= 255.0f;
|
||||
}
|
||||
|
||||
void VertexDecoder::Step_Color8888Morph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float col[4]{};
|
||||
const int onesize = dec->onesize_;
|
||||
const int morphcount = dec->morphcount;
|
||||
const int coloff = dec->coloff;
|
||||
const u8 *cdata = (const u8*)(ptr + coloff);
|
||||
#ifdef CROSSSIMD_SLOW
|
||||
float col[4]{};
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
float w = gstate_c.morphWeights[n];
|
||||
const u8 *cdata = (const u8*)(ptr + dec->onesize_*n + dec->coloff);
|
||||
const float w = gstate_c.morphWeights[n];
|
||||
for (int j = 0; j < 4; j++)
|
||||
col[j] += w * cdata[j];
|
||||
col[j] += (float)cdata[j] * w;
|
||||
cdata += onesize;
|
||||
}
|
||||
u8 *c = decoded + dec->decFmt.c0off;
|
||||
for (int i = 0; i < 4; i++) {
|
||||
c[i] = clamp_u8((int)col[i]);
|
||||
}
|
||||
gstate_c.vertexFullAlpha = gstate_c.vertexFullAlpha && (int)col[3] >= 255;
|
||||
#else
|
||||
const float *weights = gstate_c.morphWeights;
|
||||
Vec4F32 sum = Vec4F32::Zero();
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
sum += Vec4F32::LoadConvertU8(cdata) * weights[n];
|
||||
cdata += onesize;
|
||||
}
|
||||
|
||||
u8 *c = decoded + dec->decFmt.c0off;
|
||||
sum.StoreConvertToU8(c);
|
||||
|
||||
gstate_c.vertexFullAlpha = gstate_c.vertexFullAlpha && sum.GetLane<3>() >= 255.0f;
|
||||
#endif
|
||||
}
|
||||
|
||||
void VertexDecoder::Step_NormalS8(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
@@ -697,34 +736,56 @@ void VertexDecoder::Step_NormalFloatSkin(const VertexDecoder *dec, const u8 *ptr
|
||||
}
|
||||
|
||||
void VertexDecoder::Step_NormalS8Morph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
#ifdef CROSSSIMD_SLOW
|
||||
float acc[3]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
const s8 *bv = (const s8 *)(ptr + dec->nrmoff);
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
const s8 *bv = (const s8*)(ptr + dec->onesize_*n + dec->nrmoff);
|
||||
const float multiplier = gstate_c.morphWeights[n] * (1.0f / 128.0f);
|
||||
for (int j = 0; j < 3; j++)
|
||||
acc[j] += bv[j] * multiplier;
|
||||
bv += onesize;
|
||||
}
|
||||
float *normal = (float *)(decoded + dec->decFmt.nrmoff);
|
||||
memcpy(normal, acc, sizeof(float) * 3);
|
||||
#else
|
||||
Vec4F32 sum = Vec4F32::Zero();
|
||||
const float *weights = gstate_c.morphWeights;
|
||||
const int morphcount = dec->morphcount;
|
||||
const s8 *bv = (const s8*)(ptr + dec->nrmoff);
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
Vec4F32 w = Vec4F32::Splat(weights[n]);
|
||||
sum += Vec4F32::LoadConvertS8(bv) * w;
|
||||
bv += onesize;
|
||||
}
|
||||
sum = sum * (1.0f / 128.0f);
|
||||
float *normal = (float *)(decoded + dec->decFmt.nrmoff);
|
||||
sum.Store3(normal);
|
||||
#endif
|
||||
}
|
||||
|
||||
void VertexDecoder::Step_NormalS16Morph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float acc[3]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
const s16_le *sv = (const s16_le *)(ptr + dec->onesize_*n + dec->nrmoff);
|
||||
const s16_le *sv = (const s16_le *)(ptr + onesize * n + dec->nrmoff);
|
||||
const float multiplier = gstate_c.morphWeights[n] * (1.0f / 32768.0f);
|
||||
for (int j = 0; j < 3; j++)
|
||||
acc[j] += sv[j] * multiplier;
|
||||
}
|
||||
float *normal = (float *)(decoded + dec->decFmt.nrmoff);
|
||||
memcpy(normal, acc, sizeof(float) * 3);
|
||||
normal[0] = acc[0] * (1.0f / 32768.0f);
|
||||
normal[1] = acc[1] * (1.0f / 32768.0f);
|
||||
normal[2] = acc[2] * (1.0f / 32768.0f);
|
||||
}
|
||||
|
||||
void VertexDecoder::Step_NormalFloatMorph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float acc[3]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
#ifdef CROSSSIMD_SLOW
|
||||
float acc[3]{};
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
float multiplier = gstate_c.morphWeights[n];
|
||||
const float_le *fv = (const float_le *)(ptr + dec->onesize_*n + dec->nrmoff);
|
||||
@@ -733,13 +794,27 @@ void VertexDecoder::Step_NormalFloatMorph(const VertexDecoder *dec, const u8 *pt
|
||||
}
|
||||
float *normal = (float *)(decoded + dec->decFmt.nrmoff);
|
||||
memcpy(normal, acc, sizeof(float) * 3);
|
||||
#else
|
||||
Vec4F32 sum = Vec4F32::Zero();
|
||||
const float *weights = gstate_c.morphWeights;
|
||||
const u8 *bv = (ptr + dec->nrmoff);
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
Vec4F32 w = Vec4F32::Splat(weights[n]);
|
||||
sum += Vec4F32::Load((float *)bv) * w;
|
||||
bv += onesize;
|
||||
}
|
||||
float *normal = (float *)(decoded + dec->decFmt.nrmoff);
|
||||
sum.Store3(normal);
|
||||
#endif
|
||||
}
|
||||
|
||||
void VertexDecoder::Step_NormalS8MorphSkin(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float nrm[3]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
const s8 *bv = (const s8*)(ptr + dec->onesize_ * n + dec->nrmoff);
|
||||
const s8 *bv = (const s8*)(ptr + onesize * n + dec->nrmoff);
|
||||
const float multiplier = gstate_c.morphWeights[n] * (1.0f / 128.0f);
|
||||
for (int j = 0; j < 3; j++)
|
||||
nrm[j] += bv[j] * multiplier;
|
||||
@@ -751,8 +826,9 @@ void VertexDecoder::Step_NormalS8MorphSkin(const VertexDecoder *dec, const u8 *p
|
||||
void VertexDecoder::Step_NormalS16MorphSkin(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float nrm[3]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
const s16_le *sv = (const s16_le *)(ptr + dec->onesize_ * n + dec->nrmoff);
|
||||
const s16_le *sv = (const s16_le *)(ptr + onesize * n + dec->nrmoff);
|
||||
const float multiplier = gstate_c.morphWeights[n] * (1.0f / 32768.0f);
|
||||
for (int j = 0; j < 3; j++)
|
||||
nrm[j] += sv[j] * multiplier;
|
||||
@@ -764,9 +840,10 @@ void VertexDecoder::Step_NormalS16MorphSkin(const VertexDecoder *dec, const u8 *
|
||||
void VertexDecoder::Step_NormalFloatMorphSkin(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float nrm[3]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
float multiplier = gstate_c.morphWeights[n];
|
||||
const float_le *fv = (const float_le *)(ptr + dec->onesize_ * n + dec->nrmoff);
|
||||
const float_le *fv = (const float_le *)(ptr + onesize * n + dec->nrmoff);
|
||||
for (int j = 0; j < 3; j++)
|
||||
nrm[j] += fv[j] * multiplier;
|
||||
}
|
||||
@@ -859,24 +936,47 @@ void VertexDecoder::Step_PosS8Morph(const VertexDecoder *dec, const u8 *ptr, u8
|
||||
memcpy(v, acc, 12);
|
||||
}
|
||||
|
||||
// TODO: If we want to squeeze a little more performance here, we can specialize this
|
||||
// for some low morph counts, MotorStorm likes to use 1 and 2 (1 is almost nonsensical as a morph count,
|
||||
// but it will multiply the vertices with the morphweight[0]... So we could check that morphweight[0] == 1.0
|
||||
// and if so use the normal path, although not sure how expensive that check would be). Or just assume
|
||||
// that it's 1.0 in that case, but that seems dangerous.
|
||||
void VertexDecoder::Step_PosS16Morph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
const int onesize = dec->onesize_;
|
||||
#ifdef CROSSSIMD_SLOW
|
||||
float acc[3]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
const float multiplier = 1.0f / 32768.0f;
|
||||
const s16_le *sv = (const s16_le *)(ptr + dec->onesize_*n + dec->posoff);
|
||||
const s16_le *sv = (const s16_le *)(ptr + onesize * n + dec->posoff);
|
||||
for (int j = 0; j < 3; j++)
|
||||
acc[j] += (float)sv[j] * (multiplier * gstate_c.morphWeights[n]);
|
||||
}
|
||||
float *v = (float *)(decoded + dec->decFmt.posoff);
|
||||
memcpy(v, acc, 12);
|
||||
#else
|
||||
Vec4F32 sum = Vec4F32::Zero();
|
||||
const float *weights = gstate_c.morphWeights;
|
||||
const int morphcount = dec->morphcount;
|
||||
const s8 *bv = (const s8 *)(ptr + dec->posoff);
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
Vec4F32 w = Vec4F32::Splat(weights[n]);
|
||||
sum += Vec4F32::LoadConvertS16((s16_le *)bv) * w; // ARM could bake the 1/32768 factor in here.
|
||||
bv += onesize;
|
||||
}
|
||||
sum *= (1.0f / 32768.0f); // Could bake this factor into the weights, but perf gain would probably be meaningless.
|
||||
float *v = (float *)(decoded + dec->decFmt.posoff);
|
||||
// It would be fine to "Store4" here actually as the last component will end up being overwritten.
|
||||
sum.Store3(v);
|
||||
#endif
|
||||
}
|
||||
|
||||
void VertexDecoder::Step_PosFloatMorph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float acc[3]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
const float_le *fv = (const float_le *)(ptr + dec->onesize_*n + dec->posoff);
|
||||
const float_le *fv = (const float_le *)(ptr + onesize * n + dec->posoff);
|
||||
for (int j = 0; j < 3; j++)
|
||||
acc[j] += fv[j] * gstate_c.morphWeights[n];
|
||||
}
|
||||
@@ -887,9 +987,10 @@ void VertexDecoder::Step_PosFloatMorph(const VertexDecoder *dec, const u8 *ptr,
|
||||
void VertexDecoder::Step_PosS8MorphSkin(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float pos[3]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
const float multiplier = 1.0f / 128.0f;
|
||||
const s8 *sv = (const s8*)(ptr + dec->onesize_ * n + dec->posoff);
|
||||
const s8 *sv = (const s8*)(ptr + onesize * n + dec->posoff);
|
||||
for (int j = 0; j < 3; j++)
|
||||
pos[j] += (float)sv[j] * (multiplier * gstate_c.morphWeights[n]);
|
||||
}
|
||||
@@ -900,9 +1001,10 @@ void VertexDecoder::Step_PosS8MorphSkin(const VertexDecoder *dec, const u8 *ptr,
|
||||
void VertexDecoder::Step_PosS16MorphSkin(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float pos[3]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
const float multiplier = 1.0f / 32768.0f;
|
||||
const s16_le *sv = (const s16_le *)(ptr + dec->onesize_ * n + dec->posoff);
|
||||
const s16_le *sv = (const s16_le *)(ptr + onesize * n + dec->posoff);
|
||||
for (int j = 0; j < 3; j++)
|
||||
pos[j] += (float)sv[j] * (multiplier * gstate_c.morphWeights[n]);
|
||||
}
|
||||
@@ -913,8 +1015,9 @@ void VertexDecoder::Step_PosS16MorphSkin(const VertexDecoder *dec, const u8 *ptr
|
||||
void VertexDecoder::Step_PosFloatMorphSkin(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
|
||||
float pos[3]{};
|
||||
const int morphcount = dec->morphcount;
|
||||
const int onesize = dec->onesize_;
|
||||
for (int n = 0; n < morphcount; n++) {
|
||||
const float_le *fv = (const float_le *)(ptr + dec->onesize_ * n + dec->posoff);
|
||||
const float_le *fv = (const float_le *)(ptr + onesize * n + dec->posoff);
|
||||
for (int j = 0; j < 3; j++)
|
||||
pos[j] += fv[j] * gstate_c.morphWeights[n];
|
||||
}
|
||||
|
||||
@@ -1199,14 +1199,6 @@ inline void ConvertMatrix4x3To3x4Transposed(float *m4x4, const float *m4x3) {
|
||||
#endif
|
||||
}
|
||||
|
||||
inline void Transpose4x4(float out[16], const float in[16]) {
|
||||
for (int i = 0; i < 4; i++) {
|
||||
for (int j = 0; j < 4; j++) {
|
||||
out[i * 4 + j] = in[j * 4 + i];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
namespace Math3D {
|
||||
|
||||
template<typename T>
|
||||
|
||||
@@ -480,6 +480,7 @@ void DirButton::Draw(UIContext &dc) {
|
||||
if (compact) {
|
||||
// No folder icon, except "up"
|
||||
dc.PushScissor(bounds_);
|
||||
dc.SetFontStyle(*GetTextStyle(dc, UI::TextSize::Small));
|
||||
if (image == ImageID("I_FOLDER") || image == ImageID("I_FOLDER_PINNED")) {
|
||||
dc.DrawTextRect(text, bounds_.Inset(5, 2), style.fgColor, ALIGN_VCENTER | FLAG_WRAP_TEXT);
|
||||
if (pinned_) {
|
||||
@@ -490,6 +491,7 @@ void DirButton::Draw(UIContext &dc) {
|
||||
} else {
|
||||
dc.Draw()->DrawImage(image, bounds_.centerX(), bounds_.centerY(), gridStyle_ ? g_Config.fGameGridScale : 1.0, style.fgColor, ALIGN_CENTER);
|
||||
}
|
||||
dc.SetFontStyle(dc.GetTheme().uiFont);
|
||||
dc.PopScissor();
|
||||
} else {
|
||||
bool scissor = false;
|
||||
|
||||
Reference in new issue
Block a user