diff --git a/src/KKdLib/KKdLib.vcxproj b/src/KKdLib/KKdLib.vcxproj index bec7190..a02db4c 100644 --- a/src/KKdLib/KKdLib.vcxproj +++ b/src/KKdLib/KKdLib.vcxproj @@ -200,6 +200,7 @@ $(IntDir)/%(RelativeDir) false 26812 + true Windows diff --git a/src/KKdLib/default.hpp b/src/KKdLib/default.hpp index c23e28f..ad90eea 100644 --- a/src/KKdLib/default.hpp +++ b/src/KKdLib/default.hpp @@ -113,7 +113,7 @@ inline double __CRTDECL actgh(double _X) { template inline T lerp_def(T x, T y, U blend) { - return ((U)1 - blend) * x + blend * y; + return ((T)1 - (T)blend) * x + blend * y; } extern void* force_malloc(size_t size); diff --git a/src/KKdLib/f2/enrs.cpp b/src/KKdLib/f2/enrs.cpp index 31dacc5..75fc40c 100644 --- a/src/KKdLib/f2/enrs.cpp +++ b/src/KKdLib/f2/enrs.cpp @@ -42,6 +42,15 @@ void enrs_entry::append(enrs_sub_entry&& data) { sub.push_back(data); } +enrs_entry& enrs_entry::operator=(const enrs_entry& ee) { + offset = ee.offset; + count = ee.count; + size = ee.size; + repeat_count = ee.repeat_count; + sub.assign(ee.sub.begin(), ee.sub.end()); + return *this; +} + enrs::enrs() { } @@ -198,6 +207,11 @@ End: s.align_write(0x10); } +enrs& enrs::operator=(const enrs& e) { + vec.assign(e.vec.begin(), e.vec.end()); + return *this; +} + inline static bool enrs_length_get_size_type(uint32_t* length, size_t val) { *length += val < 0x10 ? 1 : val < 0x1000 ? 2 : val < 0x10000000 ? 4 : 1; return val >= 0x10000000; diff --git a/src/KKdLib/f2/enrs.hpp b/src/KKdLib/f2/enrs.hpp index aed551c..665c7d2 100644 --- a/src/KKdLib/f2/enrs.hpp +++ b/src/KKdLib/f2/enrs.hpp @@ -35,6 +35,8 @@ struct enrs_entry { void append(uint32_t skip_bytes, uint32_t repeat_count, enrs_type type); void append(enrs_sub_entry&& data); + + enrs_entry& operator=(const enrs_entry& ee); }; struct enrs { @@ -47,4 +49,6 @@ struct enrs { uint32_t length(); void read(stream& s); void write(stream& s); + + enrs& operator=(const enrs& e); }; diff --git a/src/KKdLib/f2/header.hpp b/src/KKdLib/f2/header.hpp index a955252..5674499 100644 --- a/src/KKdLib/f2/header.hpp +++ b/src/KKdLib/f2/header.hpp @@ -28,7 +28,7 @@ struct f2_header { uint32_t section_size; // 0x14 uint32_t version; // 0x18 uint32_t unknown0; // 0x1C - uint32_t murmurhash; // 0x20 + uint32_t murmurhash; // 0x20 uint32_t unknown1[3]; // 0x24 uint32_t inner_signature; // 0x30 uint32_t unknown2[3]; // 0x34 diff --git a/src/KKdLib/f2/pof.cpp b/src/KKdLib/f2/pof.cpp index 700e11c..e2087c6 100644 --- a/src/KKdLib/f2/pof.cpp +++ b/src/KKdLib/f2/pof.cpp @@ -28,6 +28,31 @@ void pof::add(stream& s, int64_t offset) { vec.push_back(s.get_position() + offset); } +uint32_t pof::length() { + uint32_t l = 4; + size_t j = 0; + uint8_t bit_shift = (uint8_t)(shift_x ? 3 : 2); + size_t v = ((size_t)1 << bit_shift) - 1; + + for (int64_t& i : vec) { + size_t o = i; + if (o & v) + break; + else if (&i != vec.data()) { + size_t k = o - j; + if (!k) + continue; + j = o; + o = k; + } + else + j = o; + + if (pof_length_get_size(&l, o >> bit_shift)) + break; + } + return l; +} void pof::read(stream& s) { vec.clear(); @@ -96,30 +121,10 @@ void pof::write(stream& s) { s.write_uint8_t(0); } -uint32_t pof::length() { - uint32_t l = 4; - size_t j = 0; - uint8_t bit_shift = (uint8_t)(shift_x ? 3 : 2); - size_t v = ((size_t)1 << bit_shift) - 1; - - for (int64_t& i : vec) { - size_t o = i; - if (o & v) - break; - else if (&i != vec.data()) { - size_t k = o - j; - if (!k) - continue; - j = o; - o = k; - } - else - j = o; - - if (pof_length_get_size(&l, o >> bit_shift)) - break; - } - return l; +pof& pof::operator=(const pof& p) { + vec.assign(p.vec.begin(), p.vec.end()); + shift_x = p.shift_x; + return *this; } inline void io_write_offset_pof_add(stream& s, int64_t val, diff --git a/src/KKdLib/f2/pof.hpp b/src/KKdLib/f2/pof.hpp index 2985ce5..51a7616 100644 --- a/src/KKdLib/f2/pof.hpp +++ b/src/KKdLib/f2/pof.hpp @@ -17,9 +17,11 @@ struct pof { ~pof(); void add(stream& s, int64_t offset); + uint32_t length(); void read(stream& s); void write(stream& s); - uint32_t length(); + + pof& operator=(const pof& p); }; extern void io_write_offset_pof_add(stream& s, int64_t val, diff --git a/src/KKdLib/f2/struct.cpp b/src/KKdLib/f2/struct.cpp index 4727597..b28191f 100644 --- a/src/KKdLib/f2/struct.cpp +++ b/src/KKdLib/f2/struct.cpp @@ -147,15 +147,14 @@ static void f2_struct_get_length(f2_struct* s, bool shift_x) { l += 0x20 + align_val(len, 0x10); } - if (has_sub_structs) + if (has_sub_structs) { for (f2_struct& i : s->sub_structs) { f2_struct_get_length(&i, shift_x); l += i.header.data_size; l += i.header.length; } - - if (has_enrs || has_pof || has_sub_structs) l += 0x20; + } s->header.data_size = l; } @@ -210,11 +209,11 @@ static void f2_struct_write_inner(stream& s, f2_struct* st, uint32_t depth, bool f2_struct_write_enrs(s, &st->enrs, use_depth ? depth + 1 : 0); if (has_pof) f2_struct_write_pof(s, &st->pof, use_depth ? depth + 1 : 0, shift_x); - if (has_sub_structs) + if (has_sub_structs) { for (f2_struct& i : st->sub_structs) f2_struct_write_inner(s, &i, depth + 1, use_depth, shift_x); - if (has_enrs || has_pof || has_sub_structs) f2_header_write_end_of_container(s, use_depth ? depth + 1 : 0); + } if (!depth) f2_header_write_end_of_container(s, 0); } diff --git a/src/KKdLib/farc.cpp b/src/KKdLib/farc.cpp index 690b3e8..7f1b741 100644 --- a/src/KKdLib/farc.cpp +++ b/src/KKdLib/farc.cpp @@ -619,8 +619,6 @@ static void farc_pack_files(farc* f, stream& s, farc_signature signature, farc_f s.write_int32_t_reverse_endianness((int32_t)i.size, true); s.write_int32_t_reverse_endianness(0x00, true); } - s.write_int32_t_reverse_endianness( - (int32_t)((i.compressed ? FARC_GZIP : 0x00) | (i.encrypted ? FARC_AES : 0x00)), true); } break; } diff --git a/src/KKdLib/half_t.hpp b/src/KKdLib/half_t.hpp index 40b7ffd..f8f5493 100644 --- a/src/KKdLib/half_t.hpp +++ b/src/KKdLib/half_t.hpp @@ -48,7 +48,7 @@ inline float_t half_to_float(half_t h) { inline half_t float_to_half(float_t val) { extern bool f16c; if (f16c) - return (half_t)_mm_cvtsi128_si32(_mm_cvtps_ph(_mm_set_ss(val), _MM_FROUND_CUR_DIRECTION)); + return (half_t)_mm_cvtsi128_si32(_mm_cvtps_ph(_mm_load_ss(&val), _MM_FROUND_CUR_DIRECTION)); return float_to_half_convert(val); } @@ -62,6 +62,6 @@ inline double_t half_to_double(half_t h) { inline half_t double_to_half(double_t val) { extern bool f16c; if (f16c) - return (half_t)_mm_cvtsi128_si32(_mm_cvtps_ph(_mm_cvtpd_ps(_mm_set_sd(val)), _MM_FROUND_CUR_DIRECTION)); + return (half_t)_mm_cvtsi128_si32(_mm_cvtps_ph(_mm_cvtpd_ps(_mm_load_sd(&val)), _MM_FROUND_CUR_DIRECTION)); return double_to_half_convert(val); } diff --git a/src/KKdLib/interpolation.cpp b/src/KKdLib/interpolation.cpp index ae4276a..59d7fd9 100644 --- a/src/KKdLib/interpolation.cpp +++ b/src/KKdLib/interpolation.cpp @@ -28,6 +28,61 @@ void interpolate_chs_reverse_value(float_t* arr, size_t length, t2 = h10.y * t1_t2.x - h10.x * t1_t2.y; } +void interpolate_chs_reverse_value(float_t* arr, size_t length, float_t& t1a, float_t& t2a, + float_t& t1b, float_t& t2b, float_t& t1c, float_t& t2c, size_t f1, size_t f2, size_t f) { + vec4 t = vec4( + (float_t)(int64_t)(f - f1 + 0), + (float_t)(int64_t)(f - f1 + 1), + (float_t)(int64_t)(f - f1 + 2), + (float_t)(int64_t)(f - f1 + 3) + ) / (float_t)(int64_t)(f2 - f1); + vec4 t_2 = t * t; + vec4 t_3 = t_2 * t; + vec4 t_23 = 3.0f * t_2; + vec4 t_32 = 2.0f * t_3; + + vec4 h00 = t_32 - t_23 + 1.0f; + vec4 h01 = t_23 - t_32; + vec4 h10 = t_3 - 2.0f * t_2 + t; + vec4 h11 = t_3 - t_2; + + vec4 t1_t2 = *(vec4*)&arr[f] - h00 * arr[f1] - h01 * arr[f2]; + vec3 t_div = (*(vec3*)&t_2.x - *(vec3*)&t.x) * (*(vec3*)&t_2.y - *(vec3*)&t.y); + vec2 t1_t2a = *(vec2*)&t1_t2.x / t_div.x; + vec2 t1_t2b = *(vec2*)&t1_t2.y / t_div.y; + vec2 t1_t2c = *(vec2*)&t1_t2.z / t_div.z; + + t1a = -h11.y * t1_t2a.x + h11.x * t1_t2a.y; + t2a = h10.y * t1_t2a.x - h10.x * t1_t2a.y; + t1b = -h11.z * t1_t2b.x + h11.y * t1_t2b.y; + t2b = h10.z * t1_t2b.x - h10.y * t1_t2b.y; + t1c = -h11.w * t1_t2c.x + h11.z * t1_t2c.y; + t2c = h10.w * t1_t2c.x - h10.z * t1_t2c.y; +} + +void interpolate_chs_reverse_value(double_t* arr, size_t length, + double_t& t1, double_t& t2, size_t f1, size_t f2, size_t f) { + vec2d t = vec2d( + (double_t)(int64_t)(f - f1 + 0), + (double_t)(int64_t)(f - f1 + 1) + ) / (double_t)(int64_t)(f2 - f1); + vec2d t_2 = t * t; + vec2d t_3 = t_2 * t; + vec2d t_23 = 3.0 * t_2; + vec2d t_32 = 2.0 * t_3; + + vec2d h00 = t_32 - t_23 + 1.0; + vec2d h01 = t_23 - t_32; + vec2d h10 = t_3 - 2.0 * t_2 + t; + vec2d h11 = t_3 - t_2; + + vec2d t1_t2 = *(vec2d*)&arr[f] - h00 * arr[f1] - h01 * arr[f2]; + t1_t2 /= (t_2.x - t.x) * (t_2.y - t.y); + + t1 = -h11.y * t1_t2.x + h11.x * t1_t2.y; + t2 = h10.y * t1_t2.x - h10.x * t1_t2.y; +} + void interpolate_chs_reverse(float_t* arr, size_t length, float_t& t1, float_t& t2, size_t f1, size_t f2) { t1 = 0.0f; @@ -36,8 +91,48 @@ void interpolate_chs_reverse(float_t* arr, size_t length, if (f2 - f1 - 2 < 1) return; - float_t _t1 = 0.0f; - float_t _t2 = 0.0f; + double_t tt1 = 0.0; + double_t tt2 = 0.0; + + size_t i = f1 + 1; + for (; i < f2 - 1 && i + 3 <= f2 - 1; i += 3) { + float_t t1a = 0.0f; + float_t t2a = 0.0f; + float_t t1b = 0.0f; + float_t t2b = 0.0f; + float_t t1c = 0.0f; + float_t t2c = 0.0f; + interpolate_chs_reverse_value(arr, length, t1a, t2a, t1b, t2b, t1c, t2c, f1, f2, i); + tt1 += t1a; + tt1 += t2a; + tt1 += t1b; + tt1 += t2b; + tt1 += t1c; + tt1 += t2c; + } + + for (; i < f2 - 1; i++) { + float_t t1 = 0.0f; + float_t t2 = 0.0f; + interpolate_chs_reverse_value(arr, length, t1, t2, f1, f2, i); + tt1 += t1; + tt2 += t2; + } + + t1 = (float_t)(tt1 / (double_t)(f2 - f1 - 2)); + t2 = (float_t)(tt2 / (double_t)(f2 - f1 - 2)); +} + +void interpolate_chs_reverse(double_t* arr, size_t length, + double_t& t1, double_t& t2, size_t f1, size_t f2) { + t1 = 0.0; + t2 = 0.0; + + if (f2 - f1 - 2 < 1) + return; + + double_t _t1 = 0.0; + double_t _t2 = 0.0; double_t tt1 = 0.0; double_t tt2 = 0.0; for (size_t i = f1 + 1; i < f2 - 1; i++) { @@ -45,8 +140,8 @@ void interpolate_chs_reverse(float_t* arr, size_t length, tt1 += _t1; tt2 += _t2; } - t1 = (float_t)(tt1 / (double_t)(f2 - f1 - 2)); - t2 = (float_t)(tt2 / (double_t)(f2 - f1 - 2)); + t1 = tt1 / (double_t)(f2 - f1 - 2); + t2 = tt2 / (double_t)(f2 - f1 - 2); } int32_t interpolate_chs_reverse_sequence( @@ -122,7 +217,25 @@ int32_t interpolate_chs_reverse_sequence( if (!fast) { double_t t1_accum = 0.0; double_t t2_accum = 0.0; - for (size_t j = 1; j < i - 1; j++) { + + size_t j = 1; + for (; j < i - 1 && j + 3 <= i - 1; j += 3) { + float_t t1a = 0.0f; + float_t t2a = 0.0f; + float_t t1b = 0.0f; + float_t t2b = 0.0f; + float_t t1c = 0.0f; + float_t t2c = 0.0f; + interpolate_chs_reverse_value(a, left_count, t1a, t2a, t1b, t2b, t1c, t2c, 0, i, j); + t1_accum += t1a; + t2_accum += t2a; + t1_accum += t1b; + t2_accum += t2b; + t1_accum += t1c; + t2_accum += t2c; + } + + for (; j < i - 1; j++) { float_t t1 = 0.0f; float_t t2 = 0.0f; interpolate_chs_reverse_value(a, left_count, t1, t2, 0, i, j); @@ -210,6 +323,26 @@ int32_t interpolate_chs_reverse_sequence( values.push_back({ (float_t)(int64_t)(count - 1), arr[count - 1], t2_old, 0.0f }); + if (values.size() > 2) { + kft3* keys = values.data(); + size_t length = values.size(); + for (size_t i = 0; i < length - 3; i++) + if (*(uint32_t*)&keys[i + 0].value == *(uint32_t*)&keys[i + 1].value + && *(uint32_t*)&keys[i + 1].value == *(uint32_t*)&keys[i + 2].value + && *(uint32_t*)&keys[i + 0].tangent2 == 0 + && *(uint32_t*)&keys[i + 1].tangent1 == 0 + && *(uint32_t*)&keys[i + 1].tangent2 == 0 + && *(uint32_t*)&keys[i + 2].tangent1 == 0) { + keys[i + 1].frame = keys[i + 2].frame; + keys[i + 1].tangent2 = keys[i + 2].tangent2; + values.erase(values.begin() + (i + 2)); + keys = values.data(); + length = values.size(); + if (length < 3) + break; + } + } + kft3* keys = values.data(); size_t length = values.size(); for (size_t i = 0; i < count; i++) { diff --git a/src/KKdLib/interpolation.hpp b/src/KKdLib/interpolation.hpp index 58c5352..18f95c0 100644 --- a/src/KKdLib/interpolation.hpp +++ b/src/KKdLib/interpolation.hpp @@ -83,6 +83,45 @@ inline std::vector interpolate_linear(float_t p1, float_t p2, size_t f1 return arr; } +inline double_t interpolate_linear_value(const double_t p1, const double_t p2, + const double_t f1, const double_t f2, const double_t f) { + if (p1 == p2) + return p1; + + double_t t = (f - f1) / (f2 - f1); + return (1.0 - t) * p1 + t * p2; +} + +inline vec2d interpolate_linear_value(const vec2d p1, const vec2d p2, + const vec2d f1, const vec2d f2, const vec2d f) { + if (p1 == p2) + return p1; + + __m128d _p1 = vec2d::load_xmm(p1); + __m128d _p2 = vec2d::load_xmm(p2); + __m128d _f1 = vec2d::load_xmm(f1); + __m128d _f2 = vec2d::load_xmm(f2); + __m128d _f = vec2d::load_xmm(f); + + const __m128d _1 = vec2d::load_xmm(1.0); + + __m128d t = _mm_div_pd(_mm_sub_pd(_f, _f1), _mm_sub_pd(_f2, _f1)); + return vec2d::store_xmm(_mm_add_pd(_mm_mul_pd(_p1, _mm_sub_pd(_1, t)), _mm_mul_pd(_p2, t))); +} + +inline std::vector interpolate_linear(double_t p1, double_t p2, size_t f1, size_t f2) { + size_t length = f2 - f1 + 1; + if (p1 == p2) + return std::vector(length, p1); + + std::vector arr(length); + double_t* a = arr.data(); + for (size_t i = 0, j = length; j; i++, j--, a++) + *a = interpolate_linear_value(p1, p2, + (double_t)f1, (double_t)f2, (double_t)(f1 + i)); + return arr; +} + inline float_t interpolate_chs_value(const float_t p1, const float_t p2, const float_t t1, const float_t t2, const float_t f1, const float_t f2, const float_t f) { if (p1 == p2 && fabsf(t1) == 0.0f && fabsf(t2) == 0.0f) @@ -225,9 +264,85 @@ inline std::vector interpolate_chs(const float_t p1, const float_t p2, return arr; } +inline double_t interpolate_chs_value(const double_t p1, const double_t p2, + const double_t t1, const double_t t2, const double_t f1, const double_t f2, const double_t f) { + if (p1 == p2 && fabs(t1) == 0.0 && fabs(t2) == 0.0) + return p1; + + double_t df = f2 - f1; + double_t t = (f - f1) / df; + double_t t_2 = t * t; + double_t t_3 = t_2 * t; + double_t t_23 = 3.0f * t_2; + double_t t_32 = 2.0f * t_3; + + double_t h00 = t_32 - t_23 + 1.0f; + double_t h01 = t_23 - t_32; + double_t h10 = t_3 - 2.0f * t_2 + t; + double_t h11 = t_3 - t_2; + + return h00 * p1 + h01 * p2 + h10 * (t1 * df) + h11 * (t2 * df); +} + +inline vec2d interpolate_chs_value(const vec2d p1, const vec2d p2, + const vec2d t1, const vec2d t2, const vec2d f1, const vec2d f2, const vec2d f) { + if (p1 == p2 && vec2d::abs(t1) == 0.0 && vec2d::abs(t2) == 0.0) + return p1; + + __m128d _p1 = vec2d::load_xmm(p1); + __m128d _p2 = vec2d::load_xmm(p2); + __m128d _t1 = vec2d::load_xmm(t1); + __m128d _t2 = vec2d::load_xmm(t2); + __m128d _f1 = vec2d::load_xmm(f1); + __m128d _f2 = vec2d::load_xmm(f2); + __m128d _f = vec2d::load_xmm(f); + + const __m128d _1 = vec2d::load_xmm(1.0); + const __m128d _2 = vec2d::load_xmm(2.0); + const __m128d _3 = vec2d::load_xmm(3.0); + + __m128d df = _mm_sub_pd(_f2, _f1); + __m128d t = _mm_div_pd(_mm_sub_pd(_f, _f1), df); + __m128d t_2 = _mm_mul_pd(t, t); + __m128d t_3 = _mm_mul_pd(t_2, t); + __m128d t_23 = _mm_mul_pd(_3, t_2); + __m128d t_32 = _mm_mul_pd(_2, t_3); + + __m128d h00 = _mm_add_pd(_mm_sub_pd(t_32, t_23), _1); + __m128d h01 = _mm_sub_pd(t_23, t_32); + __m128d h10 = _mm_add_pd(_mm_sub_pd(t_3, _mm_mul_pd(_2, t_2)), t); + __m128d h11 = _mm_sub_pd(t_3, t_2); + + _p1 = _mm_mul_pd(h00, _p1); + _p2 = _mm_mul_pd(h01, _p2); + _t1 = _mm_mul_pd(h10, _mm_mul_pd(_t1, df)); + _t2 = _mm_mul_pd(h11, _mm_mul_pd(_t2, df)); + return vec2d::store_xmm(_mm_add_pd(_mm_add_pd(_p1, _p2), _mm_add_pd(_t1, _t2))); +} + +inline std::vector interpolate_chs(const double_t p1, const double_t p2, + const double_t t1, const double_t t2, const size_t f1, const size_t f2) { + size_t length = f2 - f1 + 1; + if (p1 == p2 && fabs(t1) == 0.0 && fabs(t2) == 0.0) + return std::vector(length, p1); + + std::vector arr(length); + double_t* a = arr.data(); + for (size_t i = 0, j = length; j; i++, j--, a++) + *a = interpolate_chs_value(p1, p2, t1, t2, + (double_t)f1, (double_t)f2, (double_t)(f1 + i)); + return arr; +} + extern void interpolate_chs_reverse_value(float_t* arr, size_t length, float_t& t1, float_t& t2, size_t f1, size_t f2, size_t f); +extern void interpolate_chs_reverse_value(float_t* arr, size_t length, float_t& t1a, float_t& t2a, + float_t& t1b, float_t& t2b, float_t& t1c, float_t& t2c, size_t f1, size_t f2, size_t f); +extern void interpolate_chs_reverse_value(double_t* arr, size_t length, + double_t& t1, double_t& t2, size_t f1, size_t f2, size_t f); extern void interpolate_chs_reverse(float_t* arr, size_t length, float_t& t1, float_t& t2, size_t f1, size_t f2); +extern void interpolate_chs_reverse(double_t* arr, size_t length, + double_t& t1, double_t& t2, size_t f1, size_t f2); extern int32_t interpolate_chs_reverse_sequence( std::vector& values_src, std::vector& values, bool fast = false); diff --git a/src/KKdLib/prj/math.hpp b/src/KKdLib/prj/math.hpp index f45de9f..8bdc3c5 100644 --- a/src/KKdLib/prj/math.hpp +++ b/src/KKdLib/prj/math.hpp @@ -10,64 +10,64 @@ #include namespace prj { - inline int32_t extract_sign(float_t x) { - return _mm_movemask_ps(_mm_set_ss(x)) & 0x01; + inline int32_t extract_sign(const float_t x) { + return _mm_movemask_ps(_mm_load_ss(&x)) & 0x01; } - inline float_t ceilf(float_t x) { + inline float_t ceilf(const float_t x) { int32_t x_int = (int32_t)x; if (x_int != 0x80000000 && (float_t)x_int != x) - x = (float_t)(x_int + !extract_sign(x)); + return (float_t)(x_int + !extract_sign(x)); return x; } - inline float_t floorf(float_t x) { + inline float_t floorf(const float_t x) { int32_t x_int = (int32_t)x; if (x_int != 0x80000000 && (float_t)x_int != x) - x = (float_t)(x_int - extract_sign(x)); + return (float_t)(x_int - extract_sign(x)); return x; } - inline float_t roundf(float_t x) { + inline float_t roundf(const float_t x) { if (x >= 0.0f) return floorf(x + 0.5f); else return ceilf(x - 0.5f); } - inline float_t truncf(float_t x) { + inline float_t truncf(const float_t x) { if (x >= 0.0f) return floorf(x); else return ceilf(x); } - inline int32_t extract_sign(double_t x) { - return _mm_movemask_pd(_mm_set_sd(x)) & 0x01; + inline int32_t extract_sign(const double_t x) { + return _mm_movemask_pd(_mm_load_sd(&x)) & 0x01; } - inline double_t ceil(double_t x) { + inline double_t ceil(const double_t x) { int64_t x_int = (int64_t)x; if (x_int != 0x8000000000000000 && (float_t)x_int != x) - x = (float_t)(x_int + !extract_sign(x)); + return (float_t)(x_int + !extract_sign(x)); return x; } - inline double_t floor(double_t x) { + inline double_t floor(const double_t x) { int64_t x_int = (int64_t)x; if (x_int != 0x8000000000000000 && (float_t)x_int != x) - x = (float_t)(x_int - extract_sign(x)); + return (float_t)(x_int - extract_sign(x)); return x; } - inline double_t roundf(double_t x) { + inline double_t roundf(const double_t x) { if (x >= 0.0f) return floor(x + 0.5f); else return ceil(x - 0.5f); } - inline double_t truncf(double_t x) { + inline double_t truncf(const double_t x) { if (x >= 0.0f) return floor(x); else diff --git a/src/KKdLib/quat.hpp b/src/KKdLib/quat.hpp index 09583de..24748ce 100644 --- a/src/KKdLib/quat.hpp +++ b/src/KKdLib/quat.hpp @@ -38,6 +38,7 @@ struct quat { static quat lerp(const quat& left, const quat& right, const float_t blend); static quat slerp(const quat& left, const quat& right, const float_t blend); static quat normalize(const quat& left); + static quat normalize_rcp(const quat& left); static quat rcp(const quat& left); static quat min(const quat& min, const quat& max); static quat max(const quat& min, const quat& max); @@ -111,109 +112,59 @@ inline quat::quat(float_t m00, float_t m01, float_t m02, float_t m10, } inline quat operator +(const quat& left, const quat& right) { - __m128 yt; - quat z; - *(quat*)&yt = right; - _mm_storeu_ps((float*)&z, _mm_add_ps(_mm_loadu_ps((const float*)&left), yt)); - return z; + return quat::store_xmm(_mm_add_ps(quat::load_xmm(left), quat::load_xmm(right))); } inline quat operator +(const quat& left, const float_t right) { - __m128 yt; - quat z; - yt = _mm_set_ss(right); - _mm_storeu_ps((float*)&z, _mm_add_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0))); - return z; + return quat::store_xmm(_mm_add_ps(quat::load_xmm(left), quat::load_xmm(right))); } inline quat operator -(const quat& left, const quat& right) { - __m128 yt; - quat z; - *(quat*)&yt = right; - _mm_storeu_ps((float*)&z, _mm_sub_ps(_mm_loadu_ps((const float*)&left), yt)); - return z; + return quat::store_xmm(_mm_sub_ps(quat::load_xmm(left), quat::load_xmm(right))); } inline quat operator -(const quat& left, const float_t right) { - __m128 yt; - quat z; - yt = _mm_set_ss(right); - _mm_storeu_ps((float*)&z, _mm_sub_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0))); - return z; + return quat::store_xmm(_mm_sub_ps(quat::load_xmm(left), quat::load_xmm(right))); } inline quat operator *(const quat& left, const quat& right) { - __m128 yt; - quat z; - *(quat*)&yt = right; - _mm_storeu_ps((float*)&z, _mm_mul_ps(_mm_loadu_ps((const float*)&left), yt)); - return z; + return quat::store_xmm(_mm_mul_ps(quat::load_xmm(left), quat::load_xmm(right))); } inline quat operator *(const quat& left, const float_t right) { - __m128 yt; - quat z; - yt = _mm_set_ss(right); - _mm_storeu_ps((float*)&z, _mm_mul_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0))); - return z; + return quat::store_xmm(_mm_mul_ps(quat::load_xmm(left), quat::load_xmm(right))); } inline quat operator /(const quat& left, const quat& right) { - __m128 yt; - quat z; - *(quat*)&yt = right; - _mm_storeu_ps((float*)&z, _mm_div_ps(_mm_loadu_ps((const float*)&left), yt)); - return z; + return quat::store_xmm(_mm_div_ps(quat::load_xmm(left), quat::load_xmm(right))); } inline quat operator /(const quat& left, const float_t right) { - __m128 yt; - quat z; - yt = _mm_set_ss(right); - _mm_storeu_ps((float*)&z, _mm_div_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0))); - return z; + return quat::store_xmm(_mm_div_ps(quat::load_xmm(left), quat::load_xmm(right))); } inline quat operator &(const quat& left, const quat& right) { - __m128 yt; - quat z; - *(quat*)&yt = right; - _mm_storeu_ps((float*)&z, _mm_and_ps(_mm_loadu_ps((const float*)&left), yt)); - return z; + return quat::store_xmm(_mm_and_ps(quat::load_xmm(left), quat::load_xmm(right))); } inline quat operator &(const quat& left, const float_t right) { - __m128 yt; - quat z; - yt = _mm_set_ss(right); - _mm_storeu_ps((float*)&z, _mm_and_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0))); - return z; + return quat::store_xmm(_mm_and_ps(quat::load_xmm(left), quat::load_xmm(right))); } inline quat operator ^(const quat& left, const quat& right) { - __m128 yt; - quat z; - *(quat*)&yt = right; - _mm_storeu_ps((float*)&z, _mm_xor_ps(_mm_loadu_ps((const float*)&left), yt)); - return z; + return quat::store_xmm(_mm_xor_ps(quat::load_xmm(left), quat::load_xmm(right))); } inline quat operator ^(const quat& left, const float_t right) { - __m128 yt; - quat z; - yt = _mm_set_ss(right); - _mm_storeu_ps((float*)&z, _mm_xor_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0))); - return z; + return quat::store_xmm(_mm_xor_ps(quat::load_xmm(left), quat::load_xmm(right))); } inline quat operator -(const quat& left) { - quat z; - _mm_storeu_ps((float*)&z, _mm_xor_ps(_mm_loadu_ps((const float*)&left), vec4_neg)); - return z; + return quat::store_xmm(_mm_xor_ps(quat::load_xmm(left), vec4_neg)); } inline __m128 quat::load_xmm(const float_t data) { - __m128 _data = _mm_set_ss(data); + __m128 _data = _mm_load_ss(&data); return _mm_shuffle_ps(_data, _data, 0); } @@ -270,7 +221,7 @@ inline quat quat::mul(const quat& in_q1, const quat& in_q2) { inline float_t quat::dot(const quat& left, const quat& right) { __m128 zt; - zt = _mm_mul_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right))); + zt = _mm_mul_ps(quat::load_xmm(left), quat::load_xmm(right)); zt = _mm_hadd_ps(zt, zt); return _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); } @@ -278,7 +229,7 @@ inline float_t quat::dot(const quat& left, const quat& right) { inline float_t quat::length(const quat& left) { __m128 xt; __m128 zt; - xt = _mm_loadu_ps((const float*)&left); + xt = quat::load_xmm(left); zt = _mm_mul_ps(xt, xt); zt = _mm_hadd_ps(zt, zt); return _mm_cvtss_f32(_mm_sqrt_ss(_mm_hadd_ps(zt, zt))); @@ -287,7 +238,7 @@ inline float_t quat::length(const quat& left) { inline float_t quat::length_squared(const quat& left) { __m128 xt; __m128 zt; - xt = _mm_loadu_ps((const float*)&left); + xt = quat::load_xmm(left); zt = _mm_mul_ps(xt, xt); zt = _mm_hadd_ps(zt, zt); return _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); @@ -295,7 +246,7 @@ inline float_t quat::length_squared(const quat& left) { inline float_t quat::distance(const quat& left, const quat& right) { __m128 zt; - zt = _mm_sub_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right))); + zt = _mm_sub_ps(quat::load_xmm(left), quat::load_xmm(right)); zt = _mm_mul_ps(zt, zt); zt = _mm_hadd_ps(zt, zt); return _mm_cvtss_f32(_mm_sqrt_ss(_mm_hadd_ps(zt, zt))); @@ -303,7 +254,7 @@ inline float_t quat::distance(const quat& left, const quat& right) { inline float_t quat::distance_squared(const quat& left, const quat& right) { __m128 zt; - zt = _mm_sub_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right))); + zt = _mm_sub_ps(quat::load_xmm(left), quat::load_xmm(right)); zt = _mm_mul_ps(zt, zt); zt = _mm_hadd_ps(zt, zt); return _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); @@ -352,64 +303,58 @@ inline quat quat::slerp(const quat& left, const quat& right, const float_t blend inline quat quat::normalize(const quat& left) { __m128 xt; __m128 zt; - quat z; - xt = _mm_loadu_ps((const float*)&left); + xt = quat::load_xmm(left); zt = _mm_mul_ps(xt, xt); zt = _mm_hadd_ps(zt, zt); zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt)); - if (zt.m128_f32[0] != 0.0f) - zt.m128_f32[0] = 1.0f / zt.m128_f32[0]; - _mm_storeu_ps((float*)&z, _mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0))); - return z; + if (_mm_cvtss_f32(zt) != 0.0f) + return quat::store_xmm(_mm_div_ps(xt, _mm_shuffle_ps(zt, zt, 0))); + return quat::store_xmm(xt); +} + +inline quat quat::normalize_rcp(const quat& left) { + __m128 xt; + __m128 zt; + xt = quat::load_xmm(left); + zt = _mm_mul_ps(xt, xt); + zt = _mm_hadd_ps(zt, zt); + zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt)); + if (_mm_cvtss_f32(zt) != 0.0f) + zt = _mm_div_ss(quat::load_xmm(1.0f), zt); + return quat::store_xmm(_mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0))); } inline quat quat::rcp(const quat& left) { - quat z; - _mm_storeu_ps((float*)&z, _mm_div_ps(_mm_loadu_ps((const float*)&(quat_identity)), _mm_loadu_ps((const float*)&left))); - return z; + return quat::store_xmm(_mm_div_ps(quat::load_xmm(1.0f), quat::load_xmm(left))); } inline quat quat::min(const quat& left, const quat& right) { - quat z; - _mm_storeu_ps((float*)&z, _mm_min_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right)))); - return z; + return quat::store_xmm(_mm_min_ps(quat::load_xmm(left), quat::load_xmm(right))); } inline quat quat::max(const quat& left, const quat& right) { - quat z; - _mm_storeu_ps((float*)&z, _mm_max_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right)))); - return z; + return quat::store_xmm(_mm_max_ps(quat::load_xmm(left), quat::load_xmm(right))); } inline quat quat::clamp(const quat& left, const quat& min, const quat& max) { - quat w; - _mm_storeu_ps((float*)&w, _mm_min_ps(_mm_max_ps(_mm_loadu_ps((const float*)&left), - _mm_loadu_ps((const float*)&(min))), _mm_loadu_ps((const float*)&(max)))); - return w; + return quat::store_xmm(_mm_min_ps(_mm_max_ps(quat::load_xmm(left), + quat::load_xmm(min)), quat::load_xmm(max))); } inline quat quat::clamp(const quat& left, const float_t min, const float_t max) { - __m128 yt; - __m128 zt; - quat w; - yt = _mm_set_ss(min); - zt = _mm_set_ss(max); - _mm_storeu_ps((float*)&w, _mm_min_ps(_mm_max_ps(_mm_loadu_ps((const float*)&left), - _mm_shuffle_ps(yt, yt, 0)), _mm_shuffle_ps(zt, zt, 0))); - return w; + return quat::store_xmm(_mm_min_ps(_mm_max_ps(quat::load_xmm(left), + quat::load_xmm(min)), quat::load_xmm(max))); } inline quat quat::mult_min_max(const quat& left, const quat& min, const quat& max) { __m128 xt; __m128 yt; __m128 wt; - quat w; - xt = _mm_loadu_ps((const float*)&left); - yt = _mm_xor_ps(_mm_loadu_ps((const float*)&(min)), vec4_neg); + xt = quat::load_xmm(left); + yt = _mm_xor_ps(quat::load_xmm(min), vec4_neg); wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))), - _mm_and_ps(_mm_loadu_ps((const float*)&(max)), _mm_cmpge_ps(xt, vec4::load_xmm(0.0f)))); - _mm_storeu_ps((float*)&w, _mm_mul_ps(xt, wt)); - return w; + _mm_and_ps(quat::load_xmm(max), _mm_cmpge_ps(xt, vec4::load_xmm(0.0f)))); + return quat::store_xmm(_mm_mul_ps(xt, wt)); } inline quat quat::mult_min_max(const quat& left, const float_t min, const float_t max) { @@ -417,30 +362,24 @@ inline quat quat::mult_min_max(const quat& left, const float_t min, const float_ __m128 yt; __m128 zt; __m128 wt; - quat w; - xt = _mm_loadu_ps((const float*)&left); - yt = _mm_set_ss(min); - yt = _mm_shuffle_ps(yt, yt, 0); - zt = _mm_set_ss(max); - zt = _mm_shuffle_ps(zt, zt, 0); + xt = quat::load_xmm(left); + yt = quat::load_xmm(min); + zt = quat::load_xmm(max); yt = _mm_xor_ps(yt, vec4_neg); wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))), _mm_and_ps(zt, _mm_cmpge_ps(xt, vec4::load_xmm(0.0f)))); - _mm_storeu_ps((float*)&w, _mm_mul_ps(xt, wt)); - return w; + return quat::store_xmm(_mm_mul_ps(xt, wt)); } inline quat quat::div_min_max(const quat& left, const quat& min, const quat& max) { __m128 xt; __m128 yt; __m128 wt; - quat w; - xt = _mm_loadu_ps((const float*)&left); - yt = _mm_xor_ps(_mm_loadu_ps((const float*)&(min)), vec4_neg); + xt = quat::load_xmm(left); + yt = _mm_xor_ps(quat::load_xmm(min), vec4_neg); wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))), - _mm_and_ps(_mm_loadu_ps((const float*)&(max)), _mm_cmpge_ps(xt, vec4::load_xmm(0.0f)))); - _mm_storeu_ps((float*)&w, _mm_div_ps(xt, wt)); - return w; + _mm_and_ps(quat::load_xmm(max), _mm_cmpge_ps(xt, vec4::load_xmm(0.0f)))); + return quat::store_xmm(_mm_div_ps(xt, wt)); } inline quat quat::div_min_max(const quat& left, const float_t min, const float_t max) { @@ -448,15 +387,11 @@ inline quat quat::div_min_max(const quat& left, const float_t min, const float_t __m128 yt; __m128 zt; __m128 wt; - quat w; - xt = _mm_loadu_ps((const float*)&left); - yt = _mm_set_ss(min); - yt = _mm_shuffle_ps(yt, yt, 0); - zt = _mm_set_ss(max); - zt = _mm_shuffle_ps(zt, zt, 0); + xt = quat::load_xmm(left); + yt = quat::load_xmm(min); + zt = quat::load_xmm(max); yt = _mm_xor_ps(yt, vec4_neg); wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))), _mm_and_ps(zt, _mm_cmpge_ps(xt, vec4::load_xmm(0.0f)))); - _mm_storeu_ps((float*)&w, _mm_div_ps(xt, wt)); - return w; + return quat::store_xmm(_mm_div_ps(xt, wt)); } diff --git a/src/KKdLib/vec.cpp b/src/KKdLib/vec.cpp index f499c76..375c958 100644 --- a/src/KKdLib/vec.cpp +++ b/src/KKdLib/vec.cpp @@ -9,6 +9,8 @@ const __m128 vec2_neg = { -0.0f, -0.0f, 0.0f, 0.0f }; const __m128 vec3_neg = { -0.0f, -0.0f, -0.0f, 0.0f }; const __m128 vec4_neg = { -0.0f, -0.0f, -0.0f, -0.0f }; +extern const __m128d vec2d_neg = { -0.0, -0.0f }; + const __m128i vec2i_abs = { (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F, (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F, @@ -29,3 +31,8 @@ const __m128i vec4i_abs = { (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F, (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F, }; + +extern const __m128i vec2i64_abs = { + (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F, + (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F, +}; diff --git a/src/KKdLib/vec.hpp b/src/KKdLib/vec.hpp index 3d68ce4..f605087 100644 --- a/src/KKdLib/vec.hpp +++ b/src/KKdLib/vec.hpp @@ -457,14 +457,54 @@ struct vec4i { static vec4i clamp(const vec4i& left, const int32_t min, const int32_t max); }; +struct vec2d { + double_t x; + double_t y; + + vec2d(); + vec2d(double_t value); + vec2d(double_t x, double_t y); + + static __m128d load_xmm(const double_t data); + static __m128d load_xmm(const vec2d& data); + static __m128d load_xmm(const vec2d&& data); + static vec2d store_xmm(const __m128d& data); + static vec2d store_xmm(const __m128d&& data); + + static double_t angle(const vec2d& left, const vec2d& right); + static double_t dot(const vec2d& left, const vec2d& right); + static double_t length(const vec2d& left); + static double_t length_squared(const vec2d& left); + static double_t distance(const vec2d& left, const vec2d& right); + static double_t distance_squared(const vec2d& left, const vec2d& right); + static vec2d abs(const vec2d& left); + static vec2d lerp(const vec2d& left, const vec2d& right, const vec2d& blend); + static vec2d lerp(const vec2d& left, const vec2d& right, const double_t blend); + static vec2d normalize(const vec2d& left); + static vec2d normalize_rcp(const vec2d& left); + static vec2d rcp(const vec2d& left); + static vec2d min(const vec2d& left, const vec2d& right); + static vec2d max(const vec2d& left, const vec2d& right); + static vec2d clamp(const vec2d& left, const vec2d& min, const vec2d& max); + static vec2d clamp(const vec2d& left, const double_t min, const double_t max); + static vec2d mult_min_max(const vec2d& left, const vec2d& min, const vec2d& max); + static vec2d mult_min_max(const vec2d& left, const double_t min, const double_t max); + static vec2d div_min_max(const vec2d& left, const vec2d& min, const vec2d& max); + static vec2d div_min_max(const vec2d& left, const double_t min, const double_t max); +}; + extern const __m128 vec2_neg; extern const __m128 vec3_neg; extern const __m128 vec4_neg; +extern const __m128d vec2d_neg; + extern const __m128i vec2i_abs; extern const __m128i vec3i_abs; extern const __m128i vec4i_abs; +extern const __m128i vec2i64_abs; + inline vec2::vec2() : x(), y() { } @@ -478,7 +518,7 @@ inline vec2::vec2(float_t x, float_t y) : x(x), y(y) { } inline __m128 vec2::load_xmm(const float_t data) { - __m128 _data = _mm_set_ss(data); + __m128 _data = _mm_load_ss(&data); return _mm_shuffle_ps(_data, _data, 0x50); } @@ -714,7 +754,7 @@ inline vec2 vec2::normalize_rcp(const vec2& left) { zt = _mm_mul_ps(xt, xt); zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt)); if (_mm_cvtss_f32(zt) != 0.0f) - zt = _mm_div_ss(_mm_set_ss(1.0f), zt); + zt = _mm_div_ss(vec4::load_xmm(1.0f), zt); return vec2::store_xmm(_mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0))); } @@ -925,7 +965,7 @@ inline bool operator !=(const vec3& left, const vec3& right) { } inline __m128 vec3::load_xmm(const float_t data) { - __m128 _data = _mm_set_ss(data); + __m128 _data = _mm_load_ss(&data); return _mm_shuffle_ps(_data, _data, 0x40); } @@ -1046,7 +1086,7 @@ inline vec3 vec3::normalize_rcp(const vec3& left) { zt = _mm_hadd_ps(zt, zt); zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt)); if (_mm_cvtss_f32(zt) != 0.0f) - zt = _mm_div_ss(_mm_set_ss(1.0f), zt); + zt = _mm_div_ss(vec4::load_xmm(1.0f), zt); return vec3::store_xmm(_mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0))); } @@ -1270,7 +1310,7 @@ inline bool operator !=(const vec4& left, const vec4& right) { } inline __m128 vec4::load_xmm(const float_t data) { - __m128 _data = _mm_set_ss(data); + __m128 _data = _mm_load_ss(&data); return _mm_shuffle_ps(_data, _data, 0); } @@ -1381,7 +1421,7 @@ inline vec4 vec4::normalize_rcp(const vec4& left) { zt = _mm_hadd_ps(zt, zt); zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt)); if (_mm_cvtss_f32(zt) != 0.0f) - zt = _mm_div_ss(_mm_set_ss(1.0f), zt); + zt = _mm_div_ss(vec4::load_xmm(1.0f), zt); return vec4::store_xmm(_mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0))); } @@ -1686,6 +1726,321 @@ inline vec4i vec4i::clamp(const vec4i& left, const int32_t min, const int32_t ma vec4i::load_xmm(min)), vec4i::load_xmm(max))); } +inline vec2d::vec2d() : x(), y() { + +} + +inline vec2d::vec2d(double_t value) : x(value), y(value) { + +} + +inline vec2d::vec2d(double_t x, double_t y) : x(x), y(y) { + +} + +inline __m128d vec2d::load_xmm(const double_t data) { + __m128d _data = _mm_load_sd(&data); + return _mm_shuffle_pd(_data, _data, 0); +} + +inline __m128d vec2d::load_xmm(const vec2d& data) { + return _mm_loadu_pd((const double_t*) & data); +} + +inline __m128d vec2d::load_xmm(const vec2d&& data) { + return _mm_loadu_pd((const double_t*) & data); +} + +inline vec2d vec2d::store_xmm(const __m128d& data) { + vec2d _data; + _mm_storeu_pd((double_t*) & _data, data); + return _data; +} + +inline vec2d vec2d::store_xmm(const __m128d&& data) { + vec2d _data; + _mm_storeu_pd((double_t*) & _data, data); + return _data; +} + +inline vec2d operator +(const vec2d& left, const vec2d& right) { + return vec2d::store_xmm(_mm_add_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator +(const vec2d& left, const double_t right) { + return vec2d::store_xmm(_mm_add_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator +(const double_t left, const vec2d& right) { + return vec2d::store_xmm(_mm_add_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline void operator +=(vec2d& left, const vec2d& right) { + left = vec2d::store_xmm(_mm_add_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline void operator +=(vec2d& left, const double_t right) { + left = vec2d::store_xmm(_mm_add_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator -(const vec2d& left, const vec2d& right) { + return vec2d::store_xmm(_mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator -(const vec2d& left, const double_t right) { + return vec2d::store_xmm(_mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator -(const double_t left, const vec2d& right) { + return vec2d::store_xmm(_mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline void operator -=(vec2d& left, const vec2d& right) { + left = vec2d::store_xmm(_mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline void operator -=(vec2d& left, const double_t right) { + left = vec2d::store_xmm(_mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator *(const vec2d& left, const vec2d& right) { + return vec2d::store_xmm(_mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator *(const vec2d& left, const double_t right) { + return vec2d::store_xmm(_mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator *(const double_t left, const vec2d& right) { + return vec2d::store_xmm(_mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline void operator *=(vec2d& left, const vec2d& right) { + left = vec2d::store_xmm(_mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline void operator *=(vec2d& left, const double_t right) { + left = vec2d::store_xmm(_mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator /(const vec2d& left, const vec2d& right) { + return vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator /(const vec2d& left, const double_t right) { + return vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator /(const double_t left, const vec2d& right) { + return vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline void operator /=(vec2d& left, const vec2d& right) { + left = vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline void operator /=(vec2d& left, const double_t right) { + left = vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator &(const vec2d& left, const vec2d& right) { + return vec2d::store_xmm(_mm_and_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator &(const vec2d& left, const double_t right) { + return vec2d::store_xmm(_mm_and_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator &(const double_t left, const vec2d& right) { + return vec2d::store_xmm(_mm_and_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline void operator &=(vec2d& left, const vec2d& right) { + left = vec2d::store_xmm(_mm_and_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline void operator &=(vec2d& left, const double_t right) { + left = vec2d::store_xmm(_mm_and_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator ^(const vec2d& left, const vec2d& right) { + return vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator ^(const vec2d& left, const double_t right) { + return vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator ^(const double_t left, const vec2d& right) { + return vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline void operator ^=(vec2d& left, const vec2d& right) { + left = vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline void operator ^=(vec2d& left, const double_t right) { + left = vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d operator -(const vec2d& left) { + return vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d_neg)); +} + +inline bool operator ==(const vec2d& left, const vec2d& right) { + return !memcmp(&left, &right, sizeof(vec2d)); +} + +inline bool operator !=(const vec2d& left, const vec2d& right) { + return !!memcmp(&left, &right, sizeof(vec2d)); +} + +inline double_t vec2d::angle(const vec2d& left, const vec2d& right) { + return acos(vec2d::dot(left, right) / (vec2d::length(left) * vec2d::length(right))); +} + +inline double_t vec2d::dot(const vec2d& left, const vec2d& right) { + __m128d zt; + zt = _mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)); + return _mm_cvtsd_f64(_mm_hadd_pd(zt, zt)); +} + +inline double_t vec2d::length(const vec2d& left) { + __m128d xt; + __m128d zt; + xt = vec2d::load_xmm(left); + zt = _mm_mul_pd(xt, xt); + return _mm_cvtsd_f64(_mm_sqrt_sd(_mm_hadd_pd(zt, zt), zt)); +} + +inline double_t vec2d::length_squared(const vec2d& left) { + __m128d xt; + __m128d zt; + xt = vec2d::load_xmm(left); + zt = _mm_mul_pd(xt, xt); + return _mm_cvtsd_f64(_mm_hadd_pd(zt, zt)); +} + +inline double_t vec2d::distance(const vec2d& left, const vec2d& right) { + __m128d zt; + zt = _mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)); + zt = _mm_mul_pd(zt, zt); + return _mm_cvtsd_f64(_mm_sqrt_sd(_mm_hadd_pd(zt, zt), zt)); +} + +inline double_t vec2d::distance_squared(const vec2d& left, const vec2d& right) { + __m128d zt; + zt = _mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)); + zt = _mm_mul_pd(zt, zt); + return _mm_cvtsd_f64(_mm_hadd_pd(zt, zt)); +} + +inline vec2d vec2d::abs(const vec2d& left) { + return vec2d::store_xmm(_mm_castsi128_pd(_mm_and_si128(_mm_castpd_si128(vec2d::load_xmm(left)), vec2i64_abs))); +} + +inline vec2d vec2d::lerp(const vec2d& left, const vec2d& right, const vec2d& blend) { + __m128d b1; + __m128d b2; + b1 = vec2d::load_xmm(blend); + b2 = _mm_sub_pd(vec2d::load_xmm(1.0), b1); + return vec2d::store_xmm(_mm_add_pd(_mm_mul_pd(vec2d::load_xmm(left), b2), + _mm_mul_pd(vec2d::load_xmm(right), b1))); +} + +inline vec2d vec2d::lerp(const vec2d& left, const vec2d& right, const double_t blend) { + __m128d b1; + __m128d b2; + b1 = vec2d::load_xmm(blend); + b2 = _mm_sub_pd(vec2d::load_xmm(1.0), b1); + return vec2d::store_xmm(_mm_add_pd(_mm_mul_pd(vec2d::load_xmm(left), b2), + _mm_mul_pd(vec2d::load_xmm(right), b1))); +} + +inline vec2d vec2d::normalize(const vec2d& left) { + __m128d xt; + __m128d zt; + xt = vec2d::load_xmm(left); + zt = _mm_mul_pd(xt, xt); + zt = _mm_sqrt_sd(_mm_hadd_pd(zt, zt), zt); + if (_mm_cvtsd_f64(zt) != 0.0f) + return vec2d::store_xmm(_mm_div_pd(xt, _mm_shuffle_pd(zt, zt, 0))); + return vec2d::store_xmm(xt); +} + +inline vec2d vec2d::normalize_rcp(const vec2d& left) { + __m128d xt; + __m128d zt; + xt = vec2d::load_xmm(left); + zt = _mm_mul_pd(xt, xt); + zt = _mm_sqrt_sd(_mm_hadd_pd(zt, zt), zt); + if (_mm_cvtsd_f64(zt) != 0.0f) + zt = _mm_div_sd(vec2d::load_xmm(1.0), zt); + return vec2d::store_xmm(_mm_mul_pd(xt, _mm_shuffle_pd(zt, zt, 0))); +} + +inline vec2d vec2d::rcp(const vec2d& left) { + return vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(1.0), vec2d::load_xmm(left))); +} + +inline vec2d vec2d::min(const vec2d& left, const vec2d& right) { + return vec2d::store_xmm(_mm_min_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d vec2d::max(const vec2d& left, const vec2d& right) { + return vec2d::store_xmm(_mm_max_pd(vec2d::load_xmm(left), vec2d::load_xmm(right))); +} + +inline vec2d vec2d::clamp(const vec2d& left, const vec2d& min, const vec2d& max) { + return vec2d::store_xmm(_mm_min_pd(_mm_max_pd(vec2d::load_xmm(left), + vec2d::load_xmm(min)), vec2d::load_xmm(max))); +} + +inline vec2d vec2d::clamp(const vec2d& left, const double_t min, const double_t max) { + return vec2d::store_xmm(_mm_min_pd(_mm_max_pd(vec2d::load_xmm(left), + vec2d::load_xmm(min)), vec2d::load_xmm(max))); +} + +inline vec2d vec2d::mult_min_max(const vec2d& left, const vec2d& min, const vec2d& max) { + __m128d xt; + __m128d yt; + __m128d zt; + xt = vec2d::load_xmm(left); + yt = _mm_and_pd(_mm_cmplt_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(-min)); + zt = _mm_and_pd(_mm_cmpge_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(max)); + return vec2d::store_xmm(_mm_mul_pd(xt, _mm_or_pd(yt, zt))); +} + +inline vec2d vec2d::mult_min_max(const vec2d& left, const double_t min, const double_t max) { + __m128d xt; + __m128d yt; + __m128d zt; + xt = vec2d::load_xmm(left); + yt = _mm_and_pd(_mm_cmplt_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(-min)); + zt = _mm_and_pd(_mm_cmpge_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(max)); + return vec2d::store_xmm(_mm_mul_pd(xt, _mm_or_pd(yt, zt))); +} + +inline vec2d vec2d::div_min_max(const vec2d& left, const vec2d& min, const vec2d& max) { + __m128d xt; + __m128d yt; + __m128d zt; + xt = vec2d::load_xmm(left); + yt = _mm_and_pd(_mm_cmplt_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(-min)); + zt = _mm_and_pd(_mm_cmpge_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(max)); + return vec2d::store_xmm(_mm_div_pd(xt, _mm_or_pd(yt, zt))); +} + +inline vec2d vec2d::div_min_max(const vec2d& left, const double_t min, const double_t max) { + __m128d xt; + __m128d yt; + __m128d zt; + xt = vec2d::load_xmm(left); + yt = _mm_and_pd(_mm_cmplt_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(-min)); + zt = _mm_and_pd(_mm_cmpge_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(max)); + return vec2d::store_xmm(_mm_div_pd(xt, _mm_or_pd(yt, zt))); +} + inline void vec2i8_to_vec2(const vec2i8& src, vec2& dst) { dst.x = (float_t)src.x; dst.y = (float_t)src.y;