Update KKdLib from upstream

This commit is contained in:
korenkonder
2024-01-28 13:34:19 +03:00
parent a9e7707dca
commit 6c05ebbfdb
16 changed files with 755 additions and 187 deletions
+1
View File
@@ -200,6 +200,7 @@
<ObjectFileName>$(IntDir)/%(RelativeDir)</ObjectFileName>
<UseFullPaths>false</UseFullPaths>
<DisableSpecificWarnings>26812</DisableSpecificWarnings>
<MultiProcessorCompilation>true</MultiProcessorCompilation>
</ClCompile>
<Link>
<SubSystem>Windows</SubSystem>
+1 -1
View File
@@ -113,7 +113,7 @@ inline double __CRTDECL actgh(double _X) {
template <typename T, typename U>
inline T lerp_def(T x, T y, U blend) {
return ((U)1 - blend) * x + blend * y;
return ((T)1 - (T)blend) * x + blend * y;
}
extern void* force_malloc(size_t size);
+14
View File
@@ -42,6 +42,15 @@ void enrs_entry::append(enrs_sub_entry&& data) {
sub.push_back(data);
}
enrs_entry& enrs_entry::operator=(const enrs_entry& ee) {
offset = ee.offset;
count = ee.count;
size = ee.size;
repeat_count = ee.repeat_count;
sub.assign(ee.sub.begin(), ee.sub.end());
return *this;
}
enrs::enrs() {
}
@@ -198,6 +207,11 @@ End:
s.align_write(0x10);
}
enrs& enrs::operator=(const enrs& e) {
vec.assign(e.vec.begin(), e.vec.end());
return *this;
}
inline static bool enrs_length_get_size_type(uint32_t* length, size_t val) {
*length += val < 0x10 ? 1 : val < 0x1000 ? 2 : val < 0x10000000 ? 4 : 1;
return val >= 0x10000000;
+4
View File
@@ -35,6 +35,8 @@ struct enrs_entry {
void append(uint32_t skip_bytes, uint32_t repeat_count, enrs_type type);
void append(enrs_sub_entry&& data);
enrs_entry& operator=(const enrs_entry& ee);
};
struct enrs {
@@ -47,4 +49,6 @@ struct enrs {
uint32_t length();
void read(stream& s);
void write(stream& s);
enrs& operator=(const enrs& e);
};
+1 -1
View File
@@ -28,7 +28,7 @@ struct f2_header {
uint32_t section_size; // 0x14
uint32_t version; // 0x18
uint32_t unknown0; // 0x1C
uint32_t murmurhash; // 0x20
uint32_t murmurhash; // 0x20
uint32_t unknown1[3]; // 0x24
uint32_t inner_signature; // 0x30
uint32_t unknown2[3]; // 0x34
+29 -24
View File
@@ -28,6 +28,31 @@ void pof::add(stream& s, int64_t offset) {
vec.push_back(s.get_position() + offset);
}
uint32_t pof::length() {
uint32_t l = 4;
size_t j = 0;
uint8_t bit_shift = (uint8_t)(shift_x ? 3 : 2);
size_t v = ((size_t)1 << bit_shift) - 1;
for (int64_t& i : vec) {
size_t o = i;
if (o & v)
break;
else if (&i != vec.data()) {
size_t k = o - j;
if (!k)
continue;
j = o;
o = k;
}
else
j = o;
if (pof_length_get_size(&l, o >> bit_shift))
break;
}
return l;
}
void pof::read(stream& s) {
vec.clear();
@@ -96,30 +121,10 @@ void pof::write(stream& s) {
s.write_uint8_t(0);
}
uint32_t pof::length() {
uint32_t l = 4;
size_t j = 0;
uint8_t bit_shift = (uint8_t)(shift_x ? 3 : 2);
size_t v = ((size_t)1 << bit_shift) - 1;
for (int64_t& i : vec) {
size_t o = i;
if (o & v)
break;
else if (&i != vec.data()) {
size_t k = o - j;
if (!k)
continue;
j = o;
o = k;
}
else
j = o;
if (pof_length_get_size(&l, o >> bit_shift))
break;
}
return l;
pof& pof::operator=(const pof& p) {
vec.assign(p.vec.begin(), p.vec.end());
shift_x = p.shift_x;
return *this;
}
inline void io_write_offset_pof_add(stream& s, int64_t val,
+3 -1
View File
@@ -17,9 +17,11 @@ struct pof {
~pof();
void add(stream& s, int64_t offset);
uint32_t length();
void read(stream& s);
void write(stream& s);
uint32_t length();
pof& operator=(const pof& p);
};
extern void io_write_offset_pof_add(stream& s, int64_t val,
+4 -5
View File
@@ -147,15 +147,14 @@ static void f2_struct_get_length(f2_struct* s, bool shift_x) {
l += 0x20 + align_val(len, 0x10);
}
if (has_sub_structs)
if (has_sub_structs) {
for (f2_struct& i : s->sub_structs) {
f2_struct_get_length(&i, shift_x);
l += i.header.data_size;
l += i.header.length;
}
if (has_enrs || has_pof || has_sub_structs)
l += 0x20;
}
s->header.data_size = l;
}
@@ -210,11 +209,11 @@ static void f2_struct_write_inner(stream& s, f2_struct* st, uint32_t depth, bool
f2_struct_write_enrs(s, &st->enrs, use_depth ? depth + 1 : 0);
if (has_pof)
f2_struct_write_pof(s, &st->pof, use_depth ? depth + 1 : 0, shift_x);
if (has_sub_structs)
if (has_sub_structs) {
for (f2_struct& i : st->sub_structs)
f2_struct_write_inner(s, &i, depth + 1, use_depth, shift_x);
if (has_enrs || has_pof || has_sub_structs)
f2_header_write_end_of_container(s, use_depth ? depth + 1 : 0);
}
if (!depth)
f2_header_write_end_of_container(s, 0);
}
-2
View File
@@ -619,8 +619,6 @@ static void farc_pack_files(farc* f, stream& s, farc_signature signature, farc_f
s.write_int32_t_reverse_endianness((int32_t)i.size, true);
s.write_int32_t_reverse_endianness(0x00, true);
}
s.write_int32_t_reverse_endianness(
(int32_t)((i.compressed ? FARC_GZIP : 0x00) | (i.encrypted ? FARC_AES : 0x00)), true);
}
break;
}
+2 -2
View File
@@ -48,7 +48,7 @@ inline float_t half_to_float(half_t h) {
inline half_t float_to_half(float_t val) {
extern bool f16c;
if (f16c)
return (half_t)_mm_cvtsi128_si32(_mm_cvtps_ph(_mm_set_ss(val), _MM_FROUND_CUR_DIRECTION));
return (half_t)_mm_cvtsi128_si32(_mm_cvtps_ph(_mm_load_ss(&val), _MM_FROUND_CUR_DIRECTION));
return float_to_half_convert(val);
}
@@ -62,6 +62,6 @@ inline double_t half_to_double(half_t h) {
inline half_t double_to_half(double_t val) {
extern bool f16c;
if (f16c)
return (half_t)_mm_cvtsi128_si32(_mm_cvtps_ph(_mm_cvtpd_ps(_mm_set_sd(val)), _MM_FROUND_CUR_DIRECTION));
return (half_t)_mm_cvtsi128_si32(_mm_cvtps_ph(_mm_cvtpd_ps(_mm_load_sd(&val)), _MM_FROUND_CUR_DIRECTION));
return double_to_half_convert(val);
}
+138 -5
View File
@@ -28,6 +28,61 @@ void interpolate_chs_reverse_value(float_t* arr, size_t length,
t2 = h10.y * t1_t2.x - h10.x * t1_t2.y;
}
void interpolate_chs_reverse_value(float_t* arr, size_t length, float_t& t1a, float_t& t2a,
float_t& t1b, float_t& t2b, float_t& t1c, float_t& t2c, size_t f1, size_t f2, size_t f) {
vec4 t = vec4(
(float_t)(int64_t)(f - f1 + 0),
(float_t)(int64_t)(f - f1 + 1),
(float_t)(int64_t)(f - f1 + 2),
(float_t)(int64_t)(f - f1 + 3)
) / (float_t)(int64_t)(f2 - f1);
vec4 t_2 = t * t;
vec4 t_3 = t_2 * t;
vec4 t_23 = 3.0f * t_2;
vec4 t_32 = 2.0f * t_3;
vec4 h00 = t_32 - t_23 + 1.0f;
vec4 h01 = t_23 - t_32;
vec4 h10 = t_3 - 2.0f * t_2 + t;
vec4 h11 = t_3 - t_2;
vec4 t1_t2 = *(vec4*)&arr[f] - h00 * arr[f1] - h01 * arr[f2];
vec3 t_div = (*(vec3*)&t_2.x - *(vec3*)&t.x) * (*(vec3*)&t_2.y - *(vec3*)&t.y);
vec2 t1_t2a = *(vec2*)&t1_t2.x / t_div.x;
vec2 t1_t2b = *(vec2*)&t1_t2.y / t_div.y;
vec2 t1_t2c = *(vec2*)&t1_t2.z / t_div.z;
t1a = -h11.y * t1_t2a.x + h11.x * t1_t2a.y;
t2a = h10.y * t1_t2a.x - h10.x * t1_t2a.y;
t1b = -h11.z * t1_t2b.x + h11.y * t1_t2b.y;
t2b = h10.z * t1_t2b.x - h10.y * t1_t2b.y;
t1c = -h11.w * t1_t2c.x + h11.z * t1_t2c.y;
t2c = h10.w * t1_t2c.x - h10.z * t1_t2c.y;
}
void interpolate_chs_reverse_value(double_t* arr, size_t length,
double_t& t1, double_t& t2, size_t f1, size_t f2, size_t f) {
vec2d t = vec2d(
(double_t)(int64_t)(f - f1 + 0),
(double_t)(int64_t)(f - f1 + 1)
) / (double_t)(int64_t)(f2 - f1);
vec2d t_2 = t * t;
vec2d t_3 = t_2 * t;
vec2d t_23 = 3.0 * t_2;
vec2d t_32 = 2.0 * t_3;
vec2d h00 = t_32 - t_23 + 1.0;
vec2d h01 = t_23 - t_32;
vec2d h10 = t_3 - 2.0 * t_2 + t;
vec2d h11 = t_3 - t_2;
vec2d t1_t2 = *(vec2d*)&arr[f] - h00 * arr[f1] - h01 * arr[f2];
t1_t2 /= (t_2.x - t.x) * (t_2.y - t.y);
t1 = -h11.y * t1_t2.x + h11.x * t1_t2.y;
t2 = h10.y * t1_t2.x - h10.x * t1_t2.y;
}
void interpolate_chs_reverse(float_t* arr, size_t length,
float_t& t1, float_t& t2, size_t f1, size_t f2) {
t1 = 0.0f;
@@ -36,8 +91,48 @@ void interpolate_chs_reverse(float_t* arr, size_t length,
if (f2 - f1 - 2 < 1)
return;
float_t _t1 = 0.0f;
float_t _t2 = 0.0f;
double_t tt1 = 0.0;
double_t tt2 = 0.0;
size_t i = f1 + 1;
for (; i < f2 - 1 && i + 3 <= f2 - 1; i += 3) {
float_t t1a = 0.0f;
float_t t2a = 0.0f;
float_t t1b = 0.0f;
float_t t2b = 0.0f;
float_t t1c = 0.0f;
float_t t2c = 0.0f;
interpolate_chs_reverse_value(arr, length, t1a, t2a, t1b, t2b, t1c, t2c, f1, f2, i);
tt1 += t1a;
tt1 += t2a;
tt1 += t1b;
tt1 += t2b;
tt1 += t1c;
tt1 += t2c;
}
for (; i < f2 - 1; i++) {
float_t t1 = 0.0f;
float_t t2 = 0.0f;
interpolate_chs_reverse_value(arr, length, t1, t2, f1, f2, i);
tt1 += t1;
tt2 += t2;
}
t1 = (float_t)(tt1 / (double_t)(f2 - f1 - 2));
t2 = (float_t)(tt2 / (double_t)(f2 - f1 - 2));
}
void interpolate_chs_reverse(double_t* arr, size_t length,
double_t& t1, double_t& t2, size_t f1, size_t f2) {
t1 = 0.0;
t2 = 0.0;
if (f2 - f1 - 2 < 1)
return;
double_t _t1 = 0.0;
double_t _t2 = 0.0;
double_t tt1 = 0.0;
double_t tt2 = 0.0;
for (size_t i = f1 + 1; i < f2 - 1; i++) {
@@ -45,8 +140,8 @@ void interpolate_chs_reverse(float_t* arr, size_t length,
tt1 += _t1;
tt2 += _t2;
}
t1 = (float_t)(tt1 / (double_t)(f2 - f1 - 2));
t2 = (float_t)(tt2 / (double_t)(f2 - f1 - 2));
t1 = tt1 / (double_t)(f2 - f1 - 2);
t2 = tt2 / (double_t)(f2 - f1 - 2);
}
int32_t interpolate_chs_reverse_sequence(
@@ -122,7 +217,25 @@ int32_t interpolate_chs_reverse_sequence(
if (!fast) {
double_t t1_accum = 0.0;
double_t t2_accum = 0.0;
for (size_t j = 1; j < i - 1; j++) {
size_t j = 1;
for (; j < i - 1 && j + 3 <= i - 1; j += 3) {
float_t t1a = 0.0f;
float_t t2a = 0.0f;
float_t t1b = 0.0f;
float_t t2b = 0.0f;
float_t t1c = 0.0f;
float_t t2c = 0.0f;
interpolate_chs_reverse_value(a, left_count, t1a, t2a, t1b, t2b, t1c, t2c, 0, i, j);
t1_accum += t1a;
t2_accum += t2a;
t1_accum += t1b;
t2_accum += t2b;
t1_accum += t1c;
t2_accum += t2c;
}
for (; j < i - 1; j++) {
float_t t1 = 0.0f;
float_t t2 = 0.0f;
interpolate_chs_reverse_value(a, left_count, t1, t2, 0, i, j);
@@ -210,6 +323,26 @@ int32_t interpolate_chs_reverse_sequence(
values.push_back({ (float_t)(int64_t)(count - 1), arr[count - 1], t2_old, 0.0f });
if (values.size() > 2) {
kft3* keys = values.data();
size_t length = values.size();
for (size_t i = 0; i < length - 3; i++)
if (*(uint32_t*)&keys[i + 0].value == *(uint32_t*)&keys[i + 1].value
&& *(uint32_t*)&keys[i + 1].value == *(uint32_t*)&keys[i + 2].value
&& *(uint32_t*)&keys[i + 0].tangent2 == 0
&& *(uint32_t*)&keys[i + 1].tangent1 == 0
&& *(uint32_t*)&keys[i + 1].tangent2 == 0
&& *(uint32_t*)&keys[i + 2].tangent1 == 0) {
keys[i + 1].frame = keys[i + 2].frame;
keys[i + 1].tangent2 = keys[i + 2].tangent2;
values.erase(values.begin() + (i + 2));
keys = values.data();
length = values.size();
if (length < 3)
break;
}
}
kft3* keys = values.data();
size_t length = values.size();
for (size_t i = 0; i < count; i++) {
+115
View File
@@ -83,6 +83,45 @@ inline std::vector<float_t> interpolate_linear(float_t p1, float_t p2, size_t f1
return arr;
}
inline double_t interpolate_linear_value(const double_t p1, const double_t p2,
const double_t f1, const double_t f2, const double_t f) {
if (p1 == p2)
return p1;
double_t t = (f - f1) / (f2 - f1);
return (1.0 - t) * p1 + t * p2;
}
inline vec2d interpolate_linear_value(const vec2d p1, const vec2d p2,
const vec2d f1, const vec2d f2, const vec2d f) {
if (p1 == p2)
return p1;
__m128d _p1 = vec2d::load_xmm(p1);
__m128d _p2 = vec2d::load_xmm(p2);
__m128d _f1 = vec2d::load_xmm(f1);
__m128d _f2 = vec2d::load_xmm(f2);
__m128d _f = vec2d::load_xmm(f);
const __m128d _1 = vec2d::load_xmm(1.0);
__m128d t = _mm_div_pd(_mm_sub_pd(_f, _f1), _mm_sub_pd(_f2, _f1));
return vec2d::store_xmm(_mm_add_pd(_mm_mul_pd(_p1, _mm_sub_pd(_1, t)), _mm_mul_pd(_p2, t)));
}
inline std::vector<double_t> interpolate_linear(double_t p1, double_t p2, size_t f1, size_t f2) {
size_t length = f2 - f1 + 1;
if (p1 == p2)
return std::vector<double_t>(length, p1);
std::vector<double_t> arr(length);
double_t* a = arr.data();
for (size_t i = 0, j = length; j; i++, j--, a++)
*a = interpolate_linear_value(p1, p2,
(double_t)f1, (double_t)f2, (double_t)(f1 + i));
return arr;
}
inline float_t interpolate_chs_value(const float_t p1, const float_t p2,
const float_t t1, const float_t t2, const float_t f1, const float_t f2, const float_t f) {
if (p1 == p2 && fabsf(t1) == 0.0f && fabsf(t2) == 0.0f)
@@ -225,9 +264,85 @@ inline std::vector<float_t> interpolate_chs(const float_t p1, const float_t p2,
return arr;
}
inline double_t interpolate_chs_value(const double_t p1, const double_t p2,
const double_t t1, const double_t t2, const double_t f1, const double_t f2, const double_t f) {
if (p1 == p2 && fabs(t1) == 0.0 && fabs(t2) == 0.0)
return p1;
double_t df = f2 - f1;
double_t t = (f - f1) / df;
double_t t_2 = t * t;
double_t t_3 = t_2 * t;
double_t t_23 = 3.0f * t_2;
double_t t_32 = 2.0f * t_3;
double_t h00 = t_32 - t_23 + 1.0f;
double_t h01 = t_23 - t_32;
double_t h10 = t_3 - 2.0f * t_2 + t;
double_t h11 = t_3 - t_2;
return h00 * p1 + h01 * p2 + h10 * (t1 * df) + h11 * (t2 * df);
}
inline vec2d interpolate_chs_value(const vec2d p1, const vec2d p2,
const vec2d t1, const vec2d t2, const vec2d f1, const vec2d f2, const vec2d f) {
if (p1 == p2 && vec2d::abs(t1) == 0.0 && vec2d::abs(t2) == 0.0)
return p1;
__m128d _p1 = vec2d::load_xmm(p1);
__m128d _p2 = vec2d::load_xmm(p2);
__m128d _t1 = vec2d::load_xmm(t1);
__m128d _t2 = vec2d::load_xmm(t2);
__m128d _f1 = vec2d::load_xmm(f1);
__m128d _f2 = vec2d::load_xmm(f2);
__m128d _f = vec2d::load_xmm(f);
const __m128d _1 = vec2d::load_xmm(1.0);
const __m128d _2 = vec2d::load_xmm(2.0);
const __m128d _3 = vec2d::load_xmm(3.0);
__m128d df = _mm_sub_pd(_f2, _f1);
__m128d t = _mm_div_pd(_mm_sub_pd(_f, _f1), df);
__m128d t_2 = _mm_mul_pd(t, t);
__m128d t_3 = _mm_mul_pd(t_2, t);
__m128d t_23 = _mm_mul_pd(_3, t_2);
__m128d t_32 = _mm_mul_pd(_2, t_3);
__m128d h00 = _mm_add_pd(_mm_sub_pd(t_32, t_23), _1);
__m128d h01 = _mm_sub_pd(t_23, t_32);
__m128d h10 = _mm_add_pd(_mm_sub_pd(t_3, _mm_mul_pd(_2, t_2)), t);
__m128d h11 = _mm_sub_pd(t_3, t_2);
_p1 = _mm_mul_pd(h00, _p1);
_p2 = _mm_mul_pd(h01, _p2);
_t1 = _mm_mul_pd(h10, _mm_mul_pd(_t1, df));
_t2 = _mm_mul_pd(h11, _mm_mul_pd(_t2, df));
return vec2d::store_xmm(_mm_add_pd(_mm_add_pd(_p1, _p2), _mm_add_pd(_t1, _t2)));
}
inline std::vector<double_t> interpolate_chs(const double_t p1, const double_t p2,
const double_t t1, const double_t t2, const size_t f1, const size_t f2) {
size_t length = f2 - f1 + 1;
if (p1 == p2 && fabs(t1) == 0.0 && fabs(t2) == 0.0)
return std::vector<double_t>(length, p1);
std::vector<double_t> arr(length);
double_t* a = arr.data();
for (size_t i = 0, j = length; j; i++, j--, a++)
*a = interpolate_chs_value(p1, p2, t1, t2,
(double_t)f1, (double_t)f2, (double_t)(f1 + i));
return arr;
}
extern void interpolate_chs_reverse_value(float_t* arr, size_t length,
float_t& t1, float_t& t2, size_t f1, size_t f2, size_t f);
extern void interpolate_chs_reverse_value(float_t* arr, size_t length, float_t& t1a, float_t& t2a,
float_t& t1b, float_t& t2b, float_t& t1c, float_t& t2c, size_t f1, size_t f2, size_t f);
extern void interpolate_chs_reverse_value(double_t* arr, size_t length,
double_t& t1, double_t& t2, size_t f1, size_t f2, size_t f);
extern void interpolate_chs_reverse(float_t* arr, size_t length,
float_t& t1, float_t& t2, size_t f1, size_t f2);
extern void interpolate_chs_reverse(double_t* arr, size_t length,
double_t& t1, double_t& t2, size_t f1, size_t f2);
extern int32_t interpolate_chs_reverse_sequence(
std::vector<float_t>& values_src, std::vector<kft3>& values, bool fast = false);
+16 -16
View File
@@ -10,64 +10,64 @@
#include <emmintrin.h>
namespace prj {
inline int32_t extract_sign(float_t x) {
return _mm_movemask_ps(_mm_set_ss(x)) & 0x01;
inline int32_t extract_sign(const float_t x) {
return _mm_movemask_ps(_mm_load_ss(&x)) & 0x01;
}
inline float_t ceilf(float_t x) {
inline float_t ceilf(const float_t x) {
int32_t x_int = (int32_t)x;
if (x_int != 0x80000000 && (float_t)x_int != x)
x = (float_t)(x_int + !extract_sign(x));
return (float_t)(x_int + !extract_sign(x));
return x;
}
inline float_t floorf(float_t x) {
inline float_t floorf(const float_t x) {
int32_t x_int = (int32_t)x;
if (x_int != 0x80000000 && (float_t)x_int != x)
x = (float_t)(x_int - extract_sign(x));
return (float_t)(x_int - extract_sign(x));
return x;
}
inline float_t roundf(float_t x) {
inline float_t roundf(const float_t x) {
if (x >= 0.0f)
return floorf(x + 0.5f);
else
return ceilf(x - 0.5f);
}
inline float_t truncf(float_t x) {
inline float_t truncf(const float_t x) {
if (x >= 0.0f)
return floorf(x);
else
return ceilf(x);
}
inline int32_t extract_sign(double_t x) {
return _mm_movemask_pd(_mm_set_sd(x)) & 0x01;
inline int32_t extract_sign(const double_t x) {
return _mm_movemask_pd(_mm_load_sd(&x)) & 0x01;
}
inline double_t ceil(double_t x) {
inline double_t ceil(const double_t x) {
int64_t x_int = (int64_t)x;
if (x_int != 0x8000000000000000 && (float_t)x_int != x)
x = (float_t)(x_int + !extract_sign(x));
return (float_t)(x_int + !extract_sign(x));
return x;
}
inline double_t floor(double_t x) {
inline double_t floor(const double_t x) {
int64_t x_int = (int64_t)x;
if (x_int != 0x8000000000000000 && (float_t)x_int != x)
x = (float_t)(x_int - extract_sign(x));
return (float_t)(x_int - extract_sign(x));
return x;
}
inline double_t roundf(double_t x) {
inline double_t roundf(const double_t x) {
if (x >= 0.0f)
return floor(x + 0.5f);
else
return ceil(x - 0.5f);
}
inline double_t truncf(double_t x) {
inline double_t truncf(const double_t x) {
if (x >= 0.0f)
return floor(x);
else
+59 -124
View File
@@ -38,6 +38,7 @@ struct quat {
static quat lerp(const quat& left, const quat& right, const float_t blend);
static quat slerp(const quat& left, const quat& right, const float_t blend);
static quat normalize(const quat& left);
static quat normalize_rcp(const quat& left);
static quat rcp(const quat& left);
static quat min(const quat& min, const quat& max);
static quat max(const quat& min, const quat& max);
@@ -111,109 +112,59 @@ inline quat::quat(float_t m00, float_t m01, float_t m02, float_t m10,
}
inline quat operator +(const quat& left, const quat& right) {
__m128 yt;
quat z;
*(quat*)&yt = right;
_mm_storeu_ps((float*)&z, _mm_add_ps(_mm_loadu_ps((const float*)&left), yt));
return z;
return quat::store_xmm(_mm_add_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator +(const quat& left, const float_t right) {
__m128 yt;
quat z;
yt = _mm_set_ss(right);
_mm_storeu_ps((float*)&z, _mm_add_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0)));
return z;
return quat::store_xmm(_mm_add_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator -(const quat& left, const quat& right) {
__m128 yt;
quat z;
*(quat*)&yt = right;
_mm_storeu_ps((float*)&z, _mm_sub_ps(_mm_loadu_ps((const float*)&left), yt));
return z;
return quat::store_xmm(_mm_sub_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator -(const quat& left, const float_t right) {
__m128 yt;
quat z;
yt = _mm_set_ss(right);
_mm_storeu_ps((float*)&z, _mm_sub_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0)));
return z;
return quat::store_xmm(_mm_sub_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator *(const quat& left, const quat& right) {
__m128 yt;
quat z;
*(quat*)&yt = right;
_mm_storeu_ps((float*)&z, _mm_mul_ps(_mm_loadu_ps((const float*)&left), yt));
return z;
return quat::store_xmm(_mm_mul_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator *(const quat& left, const float_t right) {
__m128 yt;
quat z;
yt = _mm_set_ss(right);
_mm_storeu_ps((float*)&z, _mm_mul_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0)));
return z;
return quat::store_xmm(_mm_mul_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator /(const quat& left, const quat& right) {
__m128 yt;
quat z;
*(quat*)&yt = right;
_mm_storeu_ps((float*)&z, _mm_div_ps(_mm_loadu_ps((const float*)&left), yt));
return z;
return quat::store_xmm(_mm_div_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator /(const quat& left, const float_t right) {
__m128 yt;
quat z;
yt = _mm_set_ss(right);
_mm_storeu_ps((float*)&z, _mm_div_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0)));
return z;
return quat::store_xmm(_mm_div_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator &(const quat& left, const quat& right) {
__m128 yt;
quat z;
*(quat*)&yt = right;
_mm_storeu_ps((float*)&z, _mm_and_ps(_mm_loadu_ps((const float*)&left), yt));
return z;
return quat::store_xmm(_mm_and_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator &(const quat& left, const float_t right) {
__m128 yt;
quat z;
yt = _mm_set_ss(right);
_mm_storeu_ps((float*)&z, _mm_and_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0)));
return z;
return quat::store_xmm(_mm_and_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator ^(const quat& left, const quat& right) {
__m128 yt;
quat z;
*(quat*)&yt = right;
_mm_storeu_ps((float*)&z, _mm_xor_ps(_mm_loadu_ps((const float*)&left), yt));
return z;
return quat::store_xmm(_mm_xor_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator ^(const quat& left, const float_t right) {
__m128 yt;
quat z;
yt = _mm_set_ss(right);
_mm_storeu_ps((float*)&z, _mm_xor_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0)));
return z;
return quat::store_xmm(_mm_xor_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator -(const quat& left) {
quat z;
_mm_storeu_ps((float*)&z, _mm_xor_ps(_mm_loadu_ps((const float*)&left), vec4_neg));
return z;
return quat::store_xmm(_mm_xor_ps(quat::load_xmm(left), vec4_neg));
}
inline __m128 quat::load_xmm(const float_t data) {
__m128 _data = _mm_set_ss(data);
__m128 _data = _mm_load_ss(&data);
return _mm_shuffle_ps(_data, _data, 0);
}
@@ -270,7 +221,7 @@ inline quat quat::mul(const quat& in_q1, const quat& in_q2) {
inline float_t quat::dot(const quat& left, const quat& right) {
__m128 zt;
zt = _mm_mul_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right)));
zt = _mm_mul_ps(quat::load_xmm(left), quat::load_xmm(right));
zt = _mm_hadd_ps(zt, zt);
return _mm_cvtss_f32(_mm_hadd_ps(zt, zt));
}
@@ -278,7 +229,7 @@ inline float_t quat::dot(const quat& left, const quat& right) {
inline float_t quat::length(const quat& left) {
__m128 xt;
__m128 zt;
xt = _mm_loadu_ps((const float*)&left);
xt = quat::load_xmm(left);
zt = _mm_mul_ps(xt, xt);
zt = _mm_hadd_ps(zt, zt);
return _mm_cvtss_f32(_mm_sqrt_ss(_mm_hadd_ps(zt, zt)));
@@ -287,7 +238,7 @@ inline float_t quat::length(const quat& left) {
inline float_t quat::length_squared(const quat& left) {
__m128 xt;
__m128 zt;
xt = _mm_loadu_ps((const float*)&left);
xt = quat::load_xmm(left);
zt = _mm_mul_ps(xt, xt);
zt = _mm_hadd_ps(zt, zt);
return _mm_cvtss_f32(_mm_hadd_ps(zt, zt));
@@ -295,7 +246,7 @@ inline float_t quat::length_squared(const quat& left) {
inline float_t quat::distance(const quat& left, const quat& right) {
__m128 zt;
zt = _mm_sub_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right)));
zt = _mm_sub_ps(quat::load_xmm(left), quat::load_xmm(right));
zt = _mm_mul_ps(zt, zt);
zt = _mm_hadd_ps(zt, zt);
return _mm_cvtss_f32(_mm_sqrt_ss(_mm_hadd_ps(zt, zt)));
@@ -303,7 +254,7 @@ inline float_t quat::distance(const quat& left, const quat& right) {
inline float_t quat::distance_squared(const quat& left, const quat& right) {
__m128 zt;
zt = _mm_sub_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right)));
zt = _mm_sub_ps(quat::load_xmm(left), quat::load_xmm(right));
zt = _mm_mul_ps(zt, zt);
zt = _mm_hadd_ps(zt, zt);
return _mm_cvtss_f32(_mm_hadd_ps(zt, zt));
@@ -352,64 +303,58 @@ inline quat quat::slerp(const quat& left, const quat& right, const float_t blend
inline quat quat::normalize(const quat& left) {
__m128 xt;
__m128 zt;
quat z;
xt = _mm_loadu_ps((const float*)&left);
xt = quat::load_xmm(left);
zt = _mm_mul_ps(xt, xt);
zt = _mm_hadd_ps(zt, zt);
zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt));
if (zt.m128_f32[0] != 0.0f)
zt.m128_f32[0] = 1.0f / zt.m128_f32[0];
_mm_storeu_ps((float*)&z, _mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0)));
return z;
if (_mm_cvtss_f32(zt) != 0.0f)
return quat::store_xmm(_mm_div_ps(xt, _mm_shuffle_ps(zt, zt, 0)));
return quat::store_xmm(xt);
}
inline quat quat::normalize_rcp(const quat& left) {
__m128 xt;
__m128 zt;
xt = quat::load_xmm(left);
zt = _mm_mul_ps(xt, xt);
zt = _mm_hadd_ps(zt, zt);
zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt));
if (_mm_cvtss_f32(zt) != 0.0f)
zt = _mm_div_ss(quat::load_xmm(1.0f), zt);
return quat::store_xmm(_mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0)));
}
inline quat quat::rcp(const quat& left) {
quat z;
_mm_storeu_ps((float*)&z, _mm_div_ps(_mm_loadu_ps((const float*)&(quat_identity)), _mm_loadu_ps((const float*)&left)));
return z;
return quat::store_xmm(_mm_div_ps(quat::load_xmm(1.0f), quat::load_xmm(left)));
}
inline quat quat::min(const quat& left, const quat& right) {
quat z;
_mm_storeu_ps((float*)&z, _mm_min_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right))));
return z;
return quat::store_xmm(_mm_min_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat quat::max(const quat& left, const quat& right) {
quat z;
_mm_storeu_ps((float*)&z, _mm_max_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right))));
return z;
return quat::store_xmm(_mm_max_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat quat::clamp(const quat& left, const quat& min, const quat& max) {
quat w;
_mm_storeu_ps((float*)&w, _mm_min_ps(_mm_max_ps(_mm_loadu_ps((const float*)&left),
_mm_loadu_ps((const float*)&(min))), _mm_loadu_ps((const float*)&(max))));
return w;
return quat::store_xmm(_mm_min_ps(_mm_max_ps(quat::load_xmm(left),
quat::load_xmm(min)), quat::load_xmm(max)));
}
inline quat quat::clamp(const quat& left, const float_t min, const float_t max) {
__m128 yt;
__m128 zt;
quat w;
yt = _mm_set_ss(min);
zt = _mm_set_ss(max);
_mm_storeu_ps((float*)&w, _mm_min_ps(_mm_max_ps(_mm_loadu_ps((const float*)&left),
_mm_shuffle_ps(yt, yt, 0)), _mm_shuffle_ps(zt, zt, 0)));
return w;
return quat::store_xmm(_mm_min_ps(_mm_max_ps(quat::load_xmm(left),
quat::load_xmm(min)), quat::load_xmm(max)));
}
inline quat quat::mult_min_max(const quat& left, const quat& min, const quat& max) {
__m128 xt;
__m128 yt;
__m128 wt;
quat w;
xt = _mm_loadu_ps((const float*)&left);
yt = _mm_xor_ps(_mm_loadu_ps((const float*)&(min)), vec4_neg);
xt = quat::load_xmm(left);
yt = _mm_xor_ps(quat::load_xmm(min), vec4_neg);
wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))),
_mm_and_ps(_mm_loadu_ps((const float*)&(max)), _mm_cmpge_ps(xt, vec4::load_xmm(0.0f))));
_mm_storeu_ps((float*)&w, _mm_mul_ps(xt, wt));
return w;
_mm_and_ps(quat::load_xmm(max), _mm_cmpge_ps(xt, vec4::load_xmm(0.0f))));
return quat::store_xmm(_mm_mul_ps(xt, wt));
}
inline quat quat::mult_min_max(const quat& left, const float_t min, const float_t max) {
@@ -417,30 +362,24 @@ inline quat quat::mult_min_max(const quat& left, const float_t min, const float_
__m128 yt;
__m128 zt;
__m128 wt;
quat w;
xt = _mm_loadu_ps((const float*)&left);
yt = _mm_set_ss(min);
yt = _mm_shuffle_ps(yt, yt, 0);
zt = _mm_set_ss(max);
zt = _mm_shuffle_ps(zt, zt, 0);
xt = quat::load_xmm(left);
yt = quat::load_xmm(min);
zt = quat::load_xmm(max);
yt = _mm_xor_ps(yt, vec4_neg);
wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))),
_mm_and_ps(zt, _mm_cmpge_ps(xt, vec4::load_xmm(0.0f))));
_mm_storeu_ps((float*)&w, _mm_mul_ps(xt, wt));
return w;
return quat::store_xmm(_mm_mul_ps(xt, wt));
}
inline quat quat::div_min_max(const quat& left, const quat& min, const quat& max) {
__m128 xt;
__m128 yt;
__m128 wt;
quat w;
xt = _mm_loadu_ps((const float*)&left);
yt = _mm_xor_ps(_mm_loadu_ps((const float*)&(min)), vec4_neg);
xt = quat::load_xmm(left);
yt = _mm_xor_ps(quat::load_xmm(min), vec4_neg);
wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))),
_mm_and_ps(_mm_loadu_ps((const float*)&(max)), _mm_cmpge_ps(xt, vec4::load_xmm(0.0f))));
_mm_storeu_ps((float*)&w, _mm_div_ps(xt, wt));
return w;
_mm_and_ps(quat::load_xmm(max), _mm_cmpge_ps(xt, vec4::load_xmm(0.0f))));
return quat::store_xmm(_mm_div_ps(xt, wt));
}
inline quat quat::div_min_max(const quat& left, const float_t min, const float_t max) {
@@ -448,15 +387,11 @@ inline quat quat::div_min_max(const quat& left, const float_t min, const float_t
__m128 yt;
__m128 zt;
__m128 wt;
quat w;
xt = _mm_loadu_ps((const float*)&left);
yt = _mm_set_ss(min);
yt = _mm_shuffle_ps(yt, yt, 0);
zt = _mm_set_ss(max);
zt = _mm_shuffle_ps(zt, zt, 0);
xt = quat::load_xmm(left);
yt = quat::load_xmm(min);
zt = quat::load_xmm(max);
yt = _mm_xor_ps(yt, vec4_neg);
wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))),
_mm_and_ps(zt, _mm_cmpge_ps(xt, vec4::load_xmm(0.0f))));
_mm_storeu_ps((float*)&w, _mm_div_ps(xt, wt));
return w;
return quat::store_xmm(_mm_div_ps(xt, wt));
}
+7
View File
@@ -9,6 +9,8 @@ const __m128 vec2_neg = { -0.0f, -0.0f, 0.0f, 0.0f };
const __m128 vec3_neg = { -0.0f, -0.0f, -0.0f, 0.0f };
const __m128 vec4_neg = { -0.0f, -0.0f, -0.0f, -0.0f };
extern const __m128d vec2d_neg = { -0.0, -0.0f };
const __m128i vec2i_abs = {
(char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F,
(char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F,
@@ -29,3 +31,8 @@ const __m128i vec4i_abs = {
(char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F,
(char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F,
};
extern const __m128i vec2i64_abs = {
(char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F,
(char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F,
};
+361 -6
View File
@@ -457,14 +457,54 @@ struct vec4i {
static vec4i clamp(const vec4i& left, const int32_t min, const int32_t max);
};
struct vec2d {
double_t x;
double_t y;
vec2d();
vec2d(double_t value);
vec2d(double_t x, double_t y);
static __m128d load_xmm(const double_t data);
static __m128d load_xmm(const vec2d& data);
static __m128d load_xmm(const vec2d&& data);
static vec2d store_xmm(const __m128d& data);
static vec2d store_xmm(const __m128d&& data);
static double_t angle(const vec2d& left, const vec2d& right);
static double_t dot(const vec2d& left, const vec2d& right);
static double_t length(const vec2d& left);
static double_t length_squared(const vec2d& left);
static double_t distance(const vec2d& left, const vec2d& right);
static double_t distance_squared(const vec2d& left, const vec2d& right);
static vec2d abs(const vec2d& left);
static vec2d lerp(const vec2d& left, const vec2d& right, const vec2d& blend);
static vec2d lerp(const vec2d& left, const vec2d& right, const double_t blend);
static vec2d normalize(const vec2d& left);
static vec2d normalize_rcp(const vec2d& left);
static vec2d rcp(const vec2d& left);
static vec2d min(const vec2d& left, const vec2d& right);
static vec2d max(const vec2d& left, const vec2d& right);
static vec2d clamp(const vec2d& left, const vec2d& min, const vec2d& max);
static vec2d clamp(const vec2d& left, const double_t min, const double_t max);
static vec2d mult_min_max(const vec2d& left, const vec2d& min, const vec2d& max);
static vec2d mult_min_max(const vec2d& left, const double_t min, const double_t max);
static vec2d div_min_max(const vec2d& left, const vec2d& min, const vec2d& max);
static vec2d div_min_max(const vec2d& left, const double_t min, const double_t max);
};
extern const __m128 vec2_neg;
extern const __m128 vec3_neg;
extern const __m128 vec4_neg;
extern const __m128d vec2d_neg;
extern const __m128i vec2i_abs;
extern const __m128i vec3i_abs;
extern const __m128i vec4i_abs;
extern const __m128i vec2i64_abs;
inline vec2::vec2() : x(), y() {
}
@@ -478,7 +518,7 @@ inline vec2::vec2(float_t x, float_t y) : x(x), y(y) {
}
inline __m128 vec2::load_xmm(const float_t data) {
__m128 _data = _mm_set_ss(data);
__m128 _data = _mm_load_ss(&data);
return _mm_shuffle_ps(_data, _data, 0x50);
}
@@ -714,7 +754,7 @@ inline vec2 vec2::normalize_rcp(const vec2& left) {
zt = _mm_mul_ps(xt, xt);
zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt));
if (_mm_cvtss_f32(zt) != 0.0f)
zt = _mm_div_ss(_mm_set_ss(1.0f), zt);
zt = _mm_div_ss(vec4::load_xmm(1.0f), zt);
return vec2::store_xmm(_mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0)));
}
@@ -925,7 +965,7 @@ inline bool operator !=(const vec3& left, const vec3& right) {
}
inline __m128 vec3::load_xmm(const float_t data) {
__m128 _data = _mm_set_ss(data);
__m128 _data = _mm_load_ss(&data);
return _mm_shuffle_ps(_data, _data, 0x40);
}
@@ -1046,7 +1086,7 @@ inline vec3 vec3::normalize_rcp(const vec3& left) {
zt = _mm_hadd_ps(zt, zt);
zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt));
if (_mm_cvtss_f32(zt) != 0.0f)
zt = _mm_div_ss(_mm_set_ss(1.0f), zt);
zt = _mm_div_ss(vec4::load_xmm(1.0f), zt);
return vec3::store_xmm(_mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0)));
}
@@ -1270,7 +1310,7 @@ inline bool operator !=(const vec4& left, const vec4& right) {
}
inline __m128 vec4::load_xmm(const float_t data) {
__m128 _data = _mm_set_ss(data);
__m128 _data = _mm_load_ss(&data);
return _mm_shuffle_ps(_data, _data, 0);
}
@@ -1381,7 +1421,7 @@ inline vec4 vec4::normalize_rcp(const vec4& left) {
zt = _mm_hadd_ps(zt, zt);
zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt));
if (_mm_cvtss_f32(zt) != 0.0f)
zt = _mm_div_ss(_mm_set_ss(1.0f), zt);
zt = _mm_div_ss(vec4::load_xmm(1.0f), zt);
return vec4::store_xmm(_mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0)));
}
@@ -1686,6 +1726,321 @@ inline vec4i vec4i::clamp(const vec4i& left, const int32_t min, const int32_t ma
vec4i::load_xmm(min)), vec4i::load_xmm(max)));
}
inline vec2d::vec2d() : x(), y() {
}
inline vec2d::vec2d(double_t value) : x(value), y(value) {
}
inline vec2d::vec2d(double_t x, double_t y) : x(x), y(y) {
}
inline __m128d vec2d::load_xmm(const double_t data) {
__m128d _data = _mm_load_sd(&data);
return _mm_shuffle_pd(_data, _data, 0);
}
inline __m128d vec2d::load_xmm(const vec2d& data) {
return _mm_loadu_pd((const double_t*) & data);
}
inline __m128d vec2d::load_xmm(const vec2d&& data) {
return _mm_loadu_pd((const double_t*) & data);
}
inline vec2d vec2d::store_xmm(const __m128d& data) {
vec2d _data;
_mm_storeu_pd((double_t*) & _data, data);
return _data;
}
inline vec2d vec2d::store_xmm(const __m128d&& data) {
vec2d _data;
_mm_storeu_pd((double_t*) & _data, data);
return _data;
}
inline vec2d operator +(const vec2d& left, const vec2d& right) {
return vec2d::store_xmm(_mm_add_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator +(const vec2d& left, const double_t right) {
return vec2d::store_xmm(_mm_add_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator +(const double_t left, const vec2d& right) {
return vec2d::store_xmm(_mm_add_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline void operator +=(vec2d& left, const vec2d& right) {
left = vec2d::store_xmm(_mm_add_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline void operator +=(vec2d& left, const double_t right) {
left = vec2d::store_xmm(_mm_add_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator -(const vec2d& left, const vec2d& right) {
return vec2d::store_xmm(_mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator -(const vec2d& left, const double_t right) {
return vec2d::store_xmm(_mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator -(const double_t left, const vec2d& right) {
return vec2d::store_xmm(_mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline void operator -=(vec2d& left, const vec2d& right) {
left = vec2d::store_xmm(_mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline void operator -=(vec2d& left, const double_t right) {
left = vec2d::store_xmm(_mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator *(const vec2d& left, const vec2d& right) {
return vec2d::store_xmm(_mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator *(const vec2d& left, const double_t right) {
return vec2d::store_xmm(_mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator *(const double_t left, const vec2d& right) {
return vec2d::store_xmm(_mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline void operator *=(vec2d& left, const vec2d& right) {
left = vec2d::store_xmm(_mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline void operator *=(vec2d& left, const double_t right) {
left = vec2d::store_xmm(_mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator /(const vec2d& left, const vec2d& right) {
return vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator /(const vec2d& left, const double_t right) {
return vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator /(const double_t left, const vec2d& right) {
return vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline void operator /=(vec2d& left, const vec2d& right) {
left = vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline void operator /=(vec2d& left, const double_t right) {
left = vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator &(const vec2d& left, const vec2d& right) {
return vec2d::store_xmm(_mm_and_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator &(const vec2d& left, const double_t right) {
return vec2d::store_xmm(_mm_and_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator &(const double_t left, const vec2d& right) {
return vec2d::store_xmm(_mm_and_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline void operator &=(vec2d& left, const vec2d& right) {
left = vec2d::store_xmm(_mm_and_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline void operator &=(vec2d& left, const double_t right) {
left = vec2d::store_xmm(_mm_and_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator ^(const vec2d& left, const vec2d& right) {
return vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator ^(const vec2d& left, const double_t right) {
return vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator ^(const double_t left, const vec2d& right) {
return vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline void operator ^=(vec2d& left, const vec2d& right) {
left = vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline void operator ^=(vec2d& left, const double_t right) {
left = vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d operator -(const vec2d& left) {
return vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d_neg));
}
inline bool operator ==(const vec2d& left, const vec2d& right) {
return !memcmp(&left, &right, sizeof(vec2d));
}
inline bool operator !=(const vec2d& left, const vec2d& right) {
return !!memcmp(&left, &right, sizeof(vec2d));
}
inline double_t vec2d::angle(const vec2d& left, const vec2d& right) {
return acos(vec2d::dot(left, right) / (vec2d::length(left) * vec2d::length(right)));
}
inline double_t vec2d::dot(const vec2d& left, const vec2d& right) {
__m128d zt;
zt = _mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right));
return _mm_cvtsd_f64(_mm_hadd_pd(zt, zt));
}
inline double_t vec2d::length(const vec2d& left) {
__m128d xt;
__m128d zt;
xt = vec2d::load_xmm(left);
zt = _mm_mul_pd(xt, xt);
return _mm_cvtsd_f64(_mm_sqrt_sd(_mm_hadd_pd(zt, zt), zt));
}
inline double_t vec2d::length_squared(const vec2d& left) {
__m128d xt;
__m128d zt;
xt = vec2d::load_xmm(left);
zt = _mm_mul_pd(xt, xt);
return _mm_cvtsd_f64(_mm_hadd_pd(zt, zt));
}
inline double_t vec2d::distance(const vec2d& left, const vec2d& right) {
__m128d zt;
zt = _mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right));
zt = _mm_mul_pd(zt, zt);
return _mm_cvtsd_f64(_mm_sqrt_sd(_mm_hadd_pd(zt, zt), zt));
}
inline double_t vec2d::distance_squared(const vec2d& left, const vec2d& right) {
__m128d zt;
zt = _mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right));
zt = _mm_mul_pd(zt, zt);
return _mm_cvtsd_f64(_mm_hadd_pd(zt, zt));
}
inline vec2d vec2d::abs(const vec2d& left) {
return vec2d::store_xmm(_mm_castsi128_pd(_mm_and_si128(_mm_castpd_si128(vec2d::load_xmm(left)), vec2i64_abs)));
}
inline vec2d vec2d::lerp(const vec2d& left, const vec2d& right, const vec2d& blend) {
__m128d b1;
__m128d b2;
b1 = vec2d::load_xmm(blend);
b2 = _mm_sub_pd(vec2d::load_xmm(1.0), b1);
return vec2d::store_xmm(_mm_add_pd(_mm_mul_pd(vec2d::load_xmm(left), b2),
_mm_mul_pd(vec2d::load_xmm(right), b1)));
}
inline vec2d vec2d::lerp(const vec2d& left, const vec2d& right, const double_t blend) {
__m128d b1;
__m128d b2;
b1 = vec2d::load_xmm(blend);
b2 = _mm_sub_pd(vec2d::load_xmm(1.0), b1);
return vec2d::store_xmm(_mm_add_pd(_mm_mul_pd(vec2d::load_xmm(left), b2),
_mm_mul_pd(vec2d::load_xmm(right), b1)));
}
inline vec2d vec2d::normalize(const vec2d& left) {
__m128d xt;
__m128d zt;
xt = vec2d::load_xmm(left);
zt = _mm_mul_pd(xt, xt);
zt = _mm_sqrt_sd(_mm_hadd_pd(zt, zt), zt);
if (_mm_cvtsd_f64(zt) != 0.0f)
return vec2d::store_xmm(_mm_div_pd(xt, _mm_shuffle_pd(zt, zt, 0)));
return vec2d::store_xmm(xt);
}
inline vec2d vec2d::normalize_rcp(const vec2d& left) {
__m128d xt;
__m128d zt;
xt = vec2d::load_xmm(left);
zt = _mm_mul_pd(xt, xt);
zt = _mm_sqrt_sd(_mm_hadd_pd(zt, zt), zt);
if (_mm_cvtsd_f64(zt) != 0.0f)
zt = _mm_div_sd(vec2d::load_xmm(1.0), zt);
return vec2d::store_xmm(_mm_mul_pd(xt, _mm_shuffle_pd(zt, zt, 0)));
}
inline vec2d vec2d::rcp(const vec2d& left) {
return vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(1.0), vec2d::load_xmm(left)));
}
inline vec2d vec2d::min(const vec2d& left, const vec2d& right) {
return vec2d::store_xmm(_mm_min_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d vec2d::max(const vec2d& left, const vec2d& right) {
return vec2d::store_xmm(_mm_max_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
}
inline vec2d vec2d::clamp(const vec2d& left, const vec2d& min, const vec2d& max) {
return vec2d::store_xmm(_mm_min_pd(_mm_max_pd(vec2d::load_xmm(left),
vec2d::load_xmm(min)), vec2d::load_xmm(max)));
}
inline vec2d vec2d::clamp(const vec2d& left, const double_t min, const double_t max) {
return vec2d::store_xmm(_mm_min_pd(_mm_max_pd(vec2d::load_xmm(left),
vec2d::load_xmm(min)), vec2d::load_xmm(max)));
}
inline vec2d vec2d::mult_min_max(const vec2d& left, const vec2d& min, const vec2d& max) {
__m128d xt;
__m128d yt;
__m128d zt;
xt = vec2d::load_xmm(left);
yt = _mm_and_pd(_mm_cmplt_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(-min));
zt = _mm_and_pd(_mm_cmpge_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(max));
return vec2d::store_xmm(_mm_mul_pd(xt, _mm_or_pd(yt, zt)));
}
inline vec2d vec2d::mult_min_max(const vec2d& left, const double_t min, const double_t max) {
__m128d xt;
__m128d yt;
__m128d zt;
xt = vec2d::load_xmm(left);
yt = _mm_and_pd(_mm_cmplt_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(-min));
zt = _mm_and_pd(_mm_cmpge_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(max));
return vec2d::store_xmm(_mm_mul_pd(xt, _mm_or_pd(yt, zt)));
}
inline vec2d vec2d::div_min_max(const vec2d& left, const vec2d& min, const vec2d& max) {
__m128d xt;
__m128d yt;
__m128d zt;
xt = vec2d::load_xmm(left);
yt = _mm_and_pd(_mm_cmplt_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(-min));
zt = _mm_and_pd(_mm_cmpge_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(max));
return vec2d::store_xmm(_mm_div_pd(xt, _mm_or_pd(yt, zt)));
}
inline vec2d vec2d::div_min_max(const vec2d& left, const double_t min, const double_t max) {
__m128d xt;
__m128d yt;
__m128d zt;
xt = vec2d::load_xmm(left);
yt = _mm_and_pd(_mm_cmplt_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(-min));
zt = _mm_and_pd(_mm_cmpge_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(max));
return vec2d::store_xmm(_mm_div_pd(xt, _mm_or_pd(yt, zt)));
}
inline void vec2i8_to_vec2(const vec2i8& src, vec2& dst) {
dst.x = (float_t)src.x;
dst.y = (float_t)src.y;