diff --git a/src/KKdLib/KKdLib.vcxproj b/src/KKdLib/KKdLib.vcxproj
index bec7190..a02db4c 100644
--- a/src/KKdLib/KKdLib.vcxproj
+++ b/src/KKdLib/KKdLib.vcxproj
@@ -200,6 +200,7 @@
$(IntDir)/%(RelativeDir)
false
26812
+ true
Windows
diff --git a/src/KKdLib/default.hpp b/src/KKdLib/default.hpp
index c23e28f..ad90eea 100644
--- a/src/KKdLib/default.hpp
+++ b/src/KKdLib/default.hpp
@@ -113,7 +113,7 @@ inline double __CRTDECL actgh(double _X) {
template
inline T lerp_def(T x, T y, U blend) {
- return ((U)1 - blend) * x + blend * y;
+ return ((T)1 - (T)blend) * x + blend * y;
}
extern void* force_malloc(size_t size);
diff --git a/src/KKdLib/f2/enrs.cpp b/src/KKdLib/f2/enrs.cpp
index 31dacc5..75fc40c 100644
--- a/src/KKdLib/f2/enrs.cpp
+++ b/src/KKdLib/f2/enrs.cpp
@@ -42,6 +42,15 @@ void enrs_entry::append(enrs_sub_entry&& data) {
sub.push_back(data);
}
+enrs_entry& enrs_entry::operator=(const enrs_entry& ee) {
+ offset = ee.offset;
+ count = ee.count;
+ size = ee.size;
+ repeat_count = ee.repeat_count;
+ sub.assign(ee.sub.begin(), ee.sub.end());
+ return *this;
+}
+
enrs::enrs() {
}
@@ -198,6 +207,11 @@ End:
s.align_write(0x10);
}
+enrs& enrs::operator=(const enrs& e) {
+ vec.assign(e.vec.begin(), e.vec.end());
+ return *this;
+}
+
inline static bool enrs_length_get_size_type(uint32_t* length, size_t val) {
*length += val < 0x10 ? 1 : val < 0x1000 ? 2 : val < 0x10000000 ? 4 : 1;
return val >= 0x10000000;
diff --git a/src/KKdLib/f2/enrs.hpp b/src/KKdLib/f2/enrs.hpp
index aed551c..665c7d2 100644
--- a/src/KKdLib/f2/enrs.hpp
+++ b/src/KKdLib/f2/enrs.hpp
@@ -35,6 +35,8 @@ struct enrs_entry {
void append(uint32_t skip_bytes, uint32_t repeat_count, enrs_type type);
void append(enrs_sub_entry&& data);
+
+ enrs_entry& operator=(const enrs_entry& ee);
};
struct enrs {
@@ -47,4 +49,6 @@ struct enrs {
uint32_t length();
void read(stream& s);
void write(stream& s);
+
+ enrs& operator=(const enrs& e);
};
diff --git a/src/KKdLib/f2/header.hpp b/src/KKdLib/f2/header.hpp
index a955252..5674499 100644
--- a/src/KKdLib/f2/header.hpp
+++ b/src/KKdLib/f2/header.hpp
@@ -28,7 +28,7 @@ struct f2_header {
uint32_t section_size; // 0x14
uint32_t version; // 0x18
uint32_t unknown0; // 0x1C
- uint32_t murmurhash; // 0x20
+ uint32_t murmurhash; // 0x20
uint32_t unknown1[3]; // 0x24
uint32_t inner_signature; // 0x30
uint32_t unknown2[3]; // 0x34
diff --git a/src/KKdLib/f2/pof.cpp b/src/KKdLib/f2/pof.cpp
index 700e11c..e2087c6 100644
--- a/src/KKdLib/f2/pof.cpp
+++ b/src/KKdLib/f2/pof.cpp
@@ -28,6 +28,31 @@ void pof::add(stream& s, int64_t offset) {
vec.push_back(s.get_position() + offset);
}
+uint32_t pof::length() {
+ uint32_t l = 4;
+ size_t j = 0;
+ uint8_t bit_shift = (uint8_t)(shift_x ? 3 : 2);
+ size_t v = ((size_t)1 << bit_shift) - 1;
+
+ for (int64_t& i : vec) {
+ size_t o = i;
+ if (o & v)
+ break;
+ else if (&i != vec.data()) {
+ size_t k = o - j;
+ if (!k)
+ continue;
+ j = o;
+ o = k;
+ }
+ else
+ j = o;
+
+ if (pof_length_get_size(&l, o >> bit_shift))
+ break;
+ }
+ return l;
+}
void pof::read(stream& s) {
vec.clear();
@@ -96,30 +121,10 @@ void pof::write(stream& s) {
s.write_uint8_t(0);
}
-uint32_t pof::length() {
- uint32_t l = 4;
- size_t j = 0;
- uint8_t bit_shift = (uint8_t)(shift_x ? 3 : 2);
- size_t v = ((size_t)1 << bit_shift) - 1;
-
- for (int64_t& i : vec) {
- size_t o = i;
- if (o & v)
- break;
- else if (&i != vec.data()) {
- size_t k = o - j;
- if (!k)
- continue;
- j = o;
- o = k;
- }
- else
- j = o;
-
- if (pof_length_get_size(&l, o >> bit_shift))
- break;
- }
- return l;
+pof& pof::operator=(const pof& p) {
+ vec.assign(p.vec.begin(), p.vec.end());
+ shift_x = p.shift_x;
+ return *this;
}
inline void io_write_offset_pof_add(stream& s, int64_t val,
diff --git a/src/KKdLib/f2/pof.hpp b/src/KKdLib/f2/pof.hpp
index 2985ce5..51a7616 100644
--- a/src/KKdLib/f2/pof.hpp
+++ b/src/KKdLib/f2/pof.hpp
@@ -17,9 +17,11 @@ struct pof {
~pof();
void add(stream& s, int64_t offset);
+ uint32_t length();
void read(stream& s);
void write(stream& s);
- uint32_t length();
+
+ pof& operator=(const pof& p);
};
extern void io_write_offset_pof_add(stream& s, int64_t val,
diff --git a/src/KKdLib/f2/struct.cpp b/src/KKdLib/f2/struct.cpp
index 4727597..b28191f 100644
--- a/src/KKdLib/f2/struct.cpp
+++ b/src/KKdLib/f2/struct.cpp
@@ -147,15 +147,14 @@ static void f2_struct_get_length(f2_struct* s, bool shift_x) {
l += 0x20 + align_val(len, 0x10);
}
- if (has_sub_structs)
+ if (has_sub_structs) {
for (f2_struct& i : s->sub_structs) {
f2_struct_get_length(&i, shift_x);
l += i.header.data_size;
l += i.header.length;
}
-
- if (has_enrs || has_pof || has_sub_structs)
l += 0x20;
+ }
s->header.data_size = l;
}
@@ -210,11 +209,11 @@ static void f2_struct_write_inner(stream& s, f2_struct* st, uint32_t depth, bool
f2_struct_write_enrs(s, &st->enrs, use_depth ? depth + 1 : 0);
if (has_pof)
f2_struct_write_pof(s, &st->pof, use_depth ? depth + 1 : 0, shift_x);
- if (has_sub_structs)
+ if (has_sub_structs) {
for (f2_struct& i : st->sub_structs)
f2_struct_write_inner(s, &i, depth + 1, use_depth, shift_x);
- if (has_enrs || has_pof || has_sub_structs)
f2_header_write_end_of_container(s, use_depth ? depth + 1 : 0);
+ }
if (!depth)
f2_header_write_end_of_container(s, 0);
}
diff --git a/src/KKdLib/farc.cpp b/src/KKdLib/farc.cpp
index 690b3e8..7f1b741 100644
--- a/src/KKdLib/farc.cpp
+++ b/src/KKdLib/farc.cpp
@@ -619,8 +619,6 @@ static void farc_pack_files(farc* f, stream& s, farc_signature signature, farc_f
s.write_int32_t_reverse_endianness((int32_t)i.size, true);
s.write_int32_t_reverse_endianness(0x00, true);
}
- s.write_int32_t_reverse_endianness(
- (int32_t)((i.compressed ? FARC_GZIP : 0x00) | (i.encrypted ? FARC_AES : 0x00)), true);
}
break;
}
diff --git a/src/KKdLib/half_t.hpp b/src/KKdLib/half_t.hpp
index 40b7ffd..f8f5493 100644
--- a/src/KKdLib/half_t.hpp
+++ b/src/KKdLib/half_t.hpp
@@ -48,7 +48,7 @@ inline float_t half_to_float(half_t h) {
inline half_t float_to_half(float_t val) {
extern bool f16c;
if (f16c)
- return (half_t)_mm_cvtsi128_si32(_mm_cvtps_ph(_mm_set_ss(val), _MM_FROUND_CUR_DIRECTION));
+ return (half_t)_mm_cvtsi128_si32(_mm_cvtps_ph(_mm_load_ss(&val), _MM_FROUND_CUR_DIRECTION));
return float_to_half_convert(val);
}
@@ -62,6 +62,6 @@ inline double_t half_to_double(half_t h) {
inline half_t double_to_half(double_t val) {
extern bool f16c;
if (f16c)
- return (half_t)_mm_cvtsi128_si32(_mm_cvtps_ph(_mm_cvtpd_ps(_mm_set_sd(val)), _MM_FROUND_CUR_DIRECTION));
+ return (half_t)_mm_cvtsi128_si32(_mm_cvtps_ph(_mm_cvtpd_ps(_mm_load_sd(&val)), _MM_FROUND_CUR_DIRECTION));
return double_to_half_convert(val);
}
diff --git a/src/KKdLib/interpolation.cpp b/src/KKdLib/interpolation.cpp
index ae4276a..59d7fd9 100644
--- a/src/KKdLib/interpolation.cpp
+++ b/src/KKdLib/interpolation.cpp
@@ -28,6 +28,61 @@ void interpolate_chs_reverse_value(float_t* arr, size_t length,
t2 = h10.y * t1_t2.x - h10.x * t1_t2.y;
}
+void interpolate_chs_reverse_value(float_t* arr, size_t length, float_t& t1a, float_t& t2a,
+ float_t& t1b, float_t& t2b, float_t& t1c, float_t& t2c, size_t f1, size_t f2, size_t f) {
+ vec4 t = vec4(
+ (float_t)(int64_t)(f - f1 + 0),
+ (float_t)(int64_t)(f - f1 + 1),
+ (float_t)(int64_t)(f - f1 + 2),
+ (float_t)(int64_t)(f - f1 + 3)
+ ) / (float_t)(int64_t)(f2 - f1);
+ vec4 t_2 = t * t;
+ vec4 t_3 = t_2 * t;
+ vec4 t_23 = 3.0f * t_2;
+ vec4 t_32 = 2.0f * t_3;
+
+ vec4 h00 = t_32 - t_23 + 1.0f;
+ vec4 h01 = t_23 - t_32;
+ vec4 h10 = t_3 - 2.0f * t_2 + t;
+ vec4 h11 = t_3 - t_2;
+
+ vec4 t1_t2 = *(vec4*)&arr[f] - h00 * arr[f1] - h01 * arr[f2];
+ vec3 t_div = (*(vec3*)&t_2.x - *(vec3*)&t.x) * (*(vec3*)&t_2.y - *(vec3*)&t.y);
+ vec2 t1_t2a = *(vec2*)&t1_t2.x / t_div.x;
+ vec2 t1_t2b = *(vec2*)&t1_t2.y / t_div.y;
+ vec2 t1_t2c = *(vec2*)&t1_t2.z / t_div.z;
+
+ t1a = -h11.y * t1_t2a.x + h11.x * t1_t2a.y;
+ t2a = h10.y * t1_t2a.x - h10.x * t1_t2a.y;
+ t1b = -h11.z * t1_t2b.x + h11.y * t1_t2b.y;
+ t2b = h10.z * t1_t2b.x - h10.y * t1_t2b.y;
+ t1c = -h11.w * t1_t2c.x + h11.z * t1_t2c.y;
+ t2c = h10.w * t1_t2c.x - h10.z * t1_t2c.y;
+}
+
+void interpolate_chs_reverse_value(double_t* arr, size_t length,
+ double_t& t1, double_t& t2, size_t f1, size_t f2, size_t f) {
+ vec2d t = vec2d(
+ (double_t)(int64_t)(f - f1 + 0),
+ (double_t)(int64_t)(f - f1 + 1)
+ ) / (double_t)(int64_t)(f2 - f1);
+ vec2d t_2 = t * t;
+ vec2d t_3 = t_2 * t;
+ vec2d t_23 = 3.0 * t_2;
+ vec2d t_32 = 2.0 * t_3;
+
+ vec2d h00 = t_32 - t_23 + 1.0;
+ vec2d h01 = t_23 - t_32;
+ vec2d h10 = t_3 - 2.0 * t_2 + t;
+ vec2d h11 = t_3 - t_2;
+
+ vec2d t1_t2 = *(vec2d*)&arr[f] - h00 * arr[f1] - h01 * arr[f2];
+ t1_t2 /= (t_2.x - t.x) * (t_2.y - t.y);
+
+ t1 = -h11.y * t1_t2.x + h11.x * t1_t2.y;
+ t2 = h10.y * t1_t2.x - h10.x * t1_t2.y;
+}
+
void interpolate_chs_reverse(float_t* arr, size_t length,
float_t& t1, float_t& t2, size_t f1, size_t f2) {
t1 = 0.0f;
@@ -36,8 +91,48 @@ void interpolate_chs_reverse(float_t* arr, size_t length,
if (f2 - f1 - 2 < 1)
return;
- float_t _t1 = 0.0f;
- float_t _t2 = 0.0f;
+ double_t tt1 = 0.0;
+ double_t tt2 = 0.0;
+
+ size_t i = f1 + 1;
+ for (; i < f2 - 1 && i + 3 <= f2 - 1; i += 3) {
+ float_t t1a = 0.0f;
+ float_t t2a = 0.0f;
+ float_t t1b = 0.0f;
+ float_t t2b = 0.0f;
+ float_t t1c = 0.0f;
+ float_t t2c = 0.0f;
+ interpolate_chs_reverse_value(arr, length, t1a, t2a, t1b, t2b, t1c, t2c, f1, f2, i);
+ tt1 += t1a;
+ tt1 += t2a;
+ tt1 += t1b;
+ tt1 += t2b;
+ tt1 += t1c;
+ tt1 += t2c;
+ }
+
+ for (; i < f2 - 1; i++) {
+ float_t t1 = 0.0f;
+ float_t t2 = 0.0f;
+ interpolate_chs_reverse_value(arr, length, t1, t2, f1, f2, i);
+ tt1 += t1;
+ tt2 += t2;
+ }
+
+ t1 = (float_t)(tt1 / (double_t)(f2 - f1 - 2));
+ t2 = (float_t)(tt2 / (double_t)(f2 - f1 - 2));
+}
+
+void interpolate_chs_reverse(double_t* arr, size_t length,
+ double_t& t1, double_t& t2, size_t f1, size_t f2) {
+ t1 = 0.0;
+ t2 = 0.0;
+
+ if (f2 - f1 - 2 < 1)
+ return;
+
+ double_t _t1 = 0.0;
+ double_t _t2 = 0.0;
double_t tt1 = 0.0;
double_t tt2 = 0.0;
for (size_t i = f1 + 1; i < f2 - 1; i++) {
@@ -45,8 +140,8 @@ void interpolate_chs_reverse(float_t* arr, size_t length,
tt1 += _t1;
tt2 += _t2;
}
- t1 = (float_t)(tt1 / (double_t)(f2 - f1 - 2));
- t2 = (float_t)(tt2 / (double_t)(f2 - f1 - 2));
+ t1 = tt1 / (double_t)(f2 - f1 - 2);
+ t2 = tt2 / (double_t)(f2 - f1 - 2);
}
int32_t interpolate_chs_reverse_sequence(
@@ -122,7 +217,25 @@ int32_t interpolate_chs_reverse_sequence(
if (!fast) {
double_t t1_accum = 0.0;
double_t t2_accum = 0.0;
- for (size_t j = 1; j < i - 1; j++) {
+
+ size_t j = 1;
+ for (; j < i - 1 && j + 3 <= i - 1; j += 3) {
+ float_t t1a = 0.0f;
+ float_t t2a = 0.0f;
+ float_t t1b = 0.0f;
+ float_t t2b = 0.0f;
+ float_t t1c = 0.0f;
+ float_t t2c = 0.0f;
+ interpolate_chs_reverse_value(a, left_count, t1a, t2a, t1b, t2b, t1c, t2c, 0, i, j);
+ t1_accum += t1a;
+ t2_accum += t2a;
+ t1_accum += t1b;
+ t2_accum += t2b;
+ t1_accum += t1c;
+ t2_accum += t2c;
+ }
+
+ for (; j < i - 1; j++) {
float_t t1 = 0.0f;
float_t t2 = 0.0f;
interpolate_chs_reverse_value(a, left_count, t1, t2, 0, i, j);
@@ -210,6 +323,26 @@ int32_t interpolate_chs_reverse_sequence(
values.push_back({ (float_t)(int64_t)(count - 1), arr[count - 1], t2_old, 0.0f });
+ if (values.size() > 2) {
+ kft3* keys = values.data();
+ size_t length = values.size();
+ for (size_t i = 0; i < length - 3; i++)
+ if (*(uint32_t*)&keys[i + 0].value == *(uint32_t*)&keys[i + 1].value
+ && *(uint32_t*)&keys[i + 1].value == *(uint32_t*)&keys[i + 2].value
+ && *(uint32_t*)&keys[i + 0].tangent2 == 0
+ && *(uint32_t*)&keys[i + 1].tangent1 == 0
+ && *(uint32_t*)&keys[i + 1].tangent2 == 0
+ && *(uint32_t*)&keys[i + 2].tangent1 == 0) {
+ keys[i + 1].frame = keys[i + 2].frame;
+ keys[i + 1].tangent2 = keys[i + 2].tangent2;
+ values.erase(values.begin() + (i + 2));
+ keys = values.data();
+ length = values.size();
+ if (length < 3)
+ break;
+ }
+ }
+
kft3* keys = values.data();
size_t length = values.size();
for (size_t i = 0; i < count; i++) {
diff --git a/src/KKdLib/interpolation.hpp b/src/KKdLib/interpolation.hpp
index 58c5352..18f95c0 100644
--- a/src/KKdLib/interpolation.hpp
+++ b/src/KKdLib/interpolation.hpp
@@ -83,6 +83,45 @@ inline std::vector interpolate_linear(float_t p1, float_t p2, size_t f1
return arr;
}
+inline double_t interpolate_linear_value(const double_t p1, const double_t p2,
+ const double_t f1, const double_t f2, const double_t f) {
+ if (p1 == p2)
+ return p1;
+
+ double_t t = (f - f1) / (f2 - f1);
+ return (1.0 - t) * p1 + t * p2;
+}
+
+inline vec2d interpolate_linear_value(const vec2d p1, const vec2d p2,
+ const vec2d f1, const vec2d f2, const vec2d f) {
+ if (p1 == p2)
+ return p1;
+
+ __m128d _p1 = vec2d::load_xmm(p1);
+ __m128d _p2 = vec2d::load_xmm(p2);
+ __m128d _f1 = vec2d::load_xmm(f1);
+ __m128d _f2 = vec2d::load_xmm(f2);
+ __m128d _f = vec2d::load_xmm(f);
+
+ const __m128d _1 = vec2d::load_xmm(1.0);
+
+ __m128d t = _mm_div_pd(_mm_sub_pd(_f, _f1), _mm_sub_pd(_f2, _f1));
+ return vec2d::store_xmm(_mm_add_pd(_mm_mul_pd(_p1, _mm_sub_pd(_1, t)), _mm_mul_pd(_p2, t)));
+}
+
+inline std::vector interpolate_linear(double_t p1, double_t p2, size_t f1, size_t f2) {
+ size_t length = f2 - f1 + 1;
+ if (p1 == p2)
+ return std::vector(length, p1);
+
+ std::vector arr(length);
+ double_t* a = arr.data();
+ for (size_t i = 0, j = length; j; i++, j--, a++)
+ *a = interpolate_linear_value(p1, p2,
+ (double_t)f1, (double_t)f2, (double_t)(f1 + i));
+ return arr;
+}
+
inline float_t interpolate_chs_value(const float_t p1, const float_t p2,
const float_t t1, const float_t t2, const float_t f1, const float_t f2, const float_t f) {
if (p1 == p2 && fabsf(t1) == 0.0f && fabsf(t2) == 0.0f)
@@ -225,9 +264,85 @@ inline std::vector interpolate_chs(const float_t p1, const float_t p2,
return arr;
}
+inline double_t interpolate_chs_value(const double_t p1, const double_t p2,
+ const double_t t1, const double_t t2, const double_t f1, const double_t f2, const double_t f) {
+ if (p1 == p2 && fabs(t1) == 0.0 && fabs(t2) == 0.0)
+ return p1;
+
+ double_t df = f2 - f1;
+ double_t t = (f - f1) / df;
+ double_t t_2 = t * t;
+ double_t t_3 = t_2 * t;
+ double_t t_23 = 3.0f * t_2;
+ double_t t_32 = 2.0f * t_3;
+
+ double_t h00 = t_32 - t_23 + 1.0f;
+ double_t h01 = t_23 - t_32;
+ double_t h10 = t_3 - 2.0f * t_2 + t;
+ double_t h11 = t_3 - t_2;
+
+ return h00 * p1 + h01 * p2 + h10 * (t1 * df) + h11 * (t2 * df);
+}
+
+inline vec2d interpolate_chs_value(const vec2d p1, const vec2d p2,
+ const vec2d t1, const vec2d t2, const vec2d f1, const vec2d f2, const vec2d f) {
+ if (p1 == p2 && vec2d::abs(t1) == 0.0 && vec2d::abs(t2) == 0.0)
+ return p1;
+
+ __m128d _p1 = vec2d::load_xmm(p1);
+ __m128d _p2 = vec2d::load_xmm(p2);
+ __m128d _t1 = vec2d::load_xmm(t1);
+ __m128d _t2 = vec2d::load_xmm(t2);
+ __m128d _f1 = vec2d::load_xmm(f1);
+ __m128d _f2 = vec2d::load_xmm(f2);
+ __m128d _f = vec2d::load_xmm(f);
+
+ const __m128d _1 = vec2d::load_xmm(1.0);
+ const __m128d _2 = vec2d::load_xmm(2.0);
+ const __m128d _3 = vec2d::load_xmm(3.0);
+
+ __m128d df = _mm_sub_pd(_f2, _f1);
+ __m128d t = _mm_div_pd(_mm_sub_pd(_f, _f1), df);
+ __m128d t_2 = _mm_mul_pd(t, t);
+ __m128d t_3 = _mm_mul_pd(t_2, t);
+ __m128d t_23 = _mm_mul_pd(_3, t_2);
+ __m128d t_32 = _mm_mul_pd(_2, t_3);
+
+ __m128d h00 = _mm_add_pd(_mm_sub_pd(t_32, t_23), _1);
+ __m128d h01 = _mm_sub_pd(t_23, t_32);
+ __m128d h10 = _mm_add_pd(_mm_sub_pd(t_3, _mm_mul_pd(_2, t_2)), t);
+ __m128d h11 = _mm_sub_pd(t_3, t_2);
+
+ _p1 = _mm_mul_pd(h00, _p1);
+ _p2 = _mm_mul_pd(h01, _p2);
+ _t1 = _mm_mul_pd(h10, _mm_mul_pd(_t1, df));
+ _t2 = _mm_mul_pd(h11, _mm_mul_pd(_t2, df));
+ return vec2d::store_xmm(_mm_add_pd(_mm_add_pd(_p1, _p2), _mm_add_pd(_t1, _t2)));
+}
+
+inline std::vector interpolate_chs(const double_t p1, const double_t p2,
+ const double_t t1, const double_t t2, const size_t f1, const size_t f2) {
+ size_t length = f2 - f1 + 1;
+ if (p1 == p2 && fabs(t1) == 0.0 && fabs(t2) == 0.0)
+ return std::vector(length, p1);
+
+ std::vector arr(length);
+ double_t* a = arr.data();
+ for (size_t i = 0, j = length; j; i++, j--, a++)
+ *a = interpolate_chs_value(p1, p2, t1, t2,
+ (double_t)f1, (double_t)f2, (double_t)(f1 + i));
+ return arr;
+}
+
extern void interpolate_chs_reverse_value(float_t* arr, size_t length,
float_t& t1, float_t& t2, size_t f1, size_t f2, size_t f);
+extern void interpolate_chs_reverse_value(float_t* arr, size_t length, float_t& t1a, float_t& t2a,
+ float_t& t1b, float_t& t2b, float_t& t1c, float_t& t2c, size_t f1, size_t f2, size_t f);
+extern void interpolate_chs_reverse_value(double_t* arr, size_t length,
+ double_t& t1, double_t& t2, size_t f1, size_t f2, size_t f);
extern void interpolate_chs_reverse(float_t* arr, size_t length,
float_t& t1, float_t& t2, size_t f1, size_t f2);
+extern void interpolate_chs_reverse(double_t* arr, size_t length,
+ double_t& t1, double_t& t2, size_t f1, size_t f2);
extern int32_t interpolate_chs_reverse_sequence(
std::vector& values_src, std::vector& values, bool fast = false);
diff --git a/src/KKdLib/prj/math.hpp b/src/KKdLib/prj/math.hpp
index f45de9f..8bdc3c5 100644
--- a/src/KKdLib/prj/math.hpp
+++ b/src/KKdLib/prj/math.hpp
@@ -10,64 +10,64 @@
#include
namespace prj {
- inline int32_t extract_sign(float_t x) {
- return _mm_movemask_ps(_mm_set_ss(x)) & 0x01;
+ inline int32_t extract_sign(const float_t x) {
+ return _mm_movemask_ps(_mm_load_ss(&x)) & 0x01;
}
- inline float_t ceilf(float_t x) {
+ inline float_t ceilf(const float_t x) {
int32_t x_int = (int32_t)x;
if (x_int != 0x80000000 && (float_t)x_int != x)
- x = (float_t)(x_int + !extract_sign(x));
+ return (float_t)(x_int + !extract_sign(x));
return x;
}
- inline float_t floorf(float_t x) {
+ inline float_t floorf(const float_t x) {
int32_t x_int = (int32_t)x;
if (x_int != 0x80000000 && (float_t)x_int != x)
- x = (float_t)(x_int - extract_sign(x));
+ return (float_t)(x_int - extract_sign(x));
return x;
}
- inline float_t roundf(float_t x) {
+ inline float_t roundf(const float_t x) {
if (x >= 0.0f)
return floorf(x + 0.5f);
else
return ceilf(x - 0.5f);
}
- inline float_t truncf(float_t x) {
+ inline float_t truncf(const float_t x) {
if (x >= 0.0f)
return floorf(x);
else
return ceilf(x);
}
- inline int32_t extract_sign(double_t x) {
- return _mm_movemask_pd(_mm_set_sd(x)) & 0x01;
+ inline int32_t extract_sign(const double_t x) {
+ return _mm_movemask_pd(_mm_load_sd(&x)) & 0x01;
}
- inline double_t ceil(double_t x) {
+ inline double_t ceil(const double_t x) {
int64_t x_int = (int64_t)x;
if (x_int != 0x8000000000000000 && (float_t)x_int != x)
- x = (float_t)(x_int + !extract_sign(x));
+ return (float_t)(x_int + !extract_sign(x));
return x;
}
- inline double_t floor(double_t x) {
+ inline double_t floor(const double_t x) {
int64_t x_int = (int64_t)x;
if (x_int != 0x8000000000000000 && (float_t)x_int != x)
- x = (float_t)(x_int - extract_sign(x));
+ return (float_t)(x_int - extract_sign(x));
return x;
}
- inline double_t roundf(double_t x) {
+ inline double_t roundf(const double_t x) {
if (x >= 0.0f)
return floor(x + 0.5f);
else
return ceil(x - 0.5f);
}
- inline double_t truncf(double_t x) {
+ inline double_t truncf(const double_t x) {
if (x >= 0.0f)
return floor(x);
else
diff --git a/src/KKdLib/quat.hpp b/src/KKdLib/quat.hpp
index 09583de..24748ce 100644
--- a/src/KKdLib/quat.hpp
+++ b/src/KKdLib/quat.hpp
@@ -38,6 +38,7 @@ struct quat {
static quat lerp(const quat& left, const quat& right, const float_t blend);
static quat slerp(const quat& left, const quat& right, const float_t blend);
static quat normalize(const quat& left);
+ static quat normalize_rcp(const quat& left);
static quat rcp(const quat& left);
static quat min(const quat& min, const quat& max);
static quat max(const quat& min, const quat& max);
@@ -111,109 +112,59 @@ inline quat::quat(float_t m00, float_t m01, float_t m02, float_t m10,
}
inline quat operator +(const quat& left, const quat& right) {
- __m128 yt;
- quat z;
- *(quat*)&yt = right;
- _mm_storeu_ps((float*)&z, _mm_add_ps(_mm_loadu_ps((const float*)&left), yt));
- return z;
+ return quat::store_xmm(_mm_add_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator +(const quat& left, const float_t right) {
- __m128 yt;
- quat z;
- yt = _mm_set_ss(right);
- _mm_storeu_ps((float*)&z, _mm_add_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0)));
- return z;
+ return quat::store_xmm(_mm_add_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator -(const quat& left, const quat& right) {
- __m128 yt;
- quat z;
- *(quat*)&yt = right;
- _mm_storeu_ps((float*)&z, _mm_sub_ps(_mm_loadu_ps((const float*)&left), yt));
- return z;
+ return quat::store_xmm(_mm_sub_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator -(const quat& left, const float_t right) {
- __m128 yt;
- quat z;
- yt = _mm_set_ss(right);
- _mm_storeu_ps((float*)&z, _mm_sub_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0)));
- return z;
+ return quat::store_xmm(_mm_sub_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator *(const quat& left, const quat& right) {
- __m128 yt;
- quat z;
- *(quat*)&yt = right;
- _mm_storeu_ps((float*)&z, _mm_mul_ps(_mm_loadu_ps((const float*)&left), yt));
- return z;
+ return quat::store_xmm(_mm_mul_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator *(const quat& left, const float_t right) {
- __m128 yt;
- quat z;
- yt = _mm_set_ss(right);
- _mm_storeu_ps((float*)&z, _mm_mul_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0)));
- return z;
+ return quat::store_xmm(_mm_mul_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator /(const quat& left, const quat& right) {
- __m128 yt;
- quat z;
- *(quat*)&yt = right;
- _mm_storeu_ps((float*)&z, _mm_div_ps(_mm_loadu_ps((const float*)&left), yt));
- return z;
+ return quat::store_xmm(_mm_div_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator /(const quat& left, const float_t right) {
- __m128 yt;
- quat z;
- yt = _mm_set_ss(right);
- _mm_storeu_ps((float*)&z, _mm_div_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0)));
- return z;
+ return quat::store_xmm(_mm_div_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator &(const quat& left, const quat& right) {
- __m128 yt;
- quat z;
- *(quat*)&yt = right;
- _mm_storeu_ps((float*)&z, _mm_and_ps(_mm_loadu_ps((const float*)&left), yt));
- return z;
+ return quat::store_xmm(_mm_and_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator &(const quat& left, const float_t right) {
- __m128 yt;
- quat z;
- yt = _mm_set_ss(right);
- _mm_storeu_ps((float*)&z, _mm_and_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0)));
- return z;
+ return quat::store_xmm(_mm_and_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator ^(const quat& left, const quat& right) {
- __m128 yt;
- quat z;
- *(quat*)&yt = right;
- _mm_storeu_ps((float*)&z, _mm_xor_ps(_mm_loadu_ps((const float*)&left), yt));
- return z;
+ return quat::store_xmm(_mm_xor_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator ^(const quat& left, const float_t right) {
- __m128 yt;
- quat z;
- yt = _mm_set_ss(right);
- _mm_storeu_ps((float*)&z, _mm_xor_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0)));
- return z;
+ return quat::store_xmm(_mm_xor_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat operator -(const quat& left) {
- quat z;
- _mm_storeu_ps((float*)&z, _mm_xor_ps(_mm_loadu_ps((const float*)&left), vec4_neg));
- return z;
+ return quat::store_xmm(_mm_xor_ps(quat::load_xmm(left), vec4_neg));
}
inline __m128 quat::load_xmm(const float_t data) {
- __m128 _data = _mm_set_ss(data);
+ __m128 _data = _mm_load_ss(&data);
return _mm_shuffle_ps(_data, _data, 0);
}
@@ -270,7 +221,7 @@ inline quat quat::mul(const quat& in_q1, const quat& in_q2) {
inline float_t quat::dot(const quat& left, const quat& right) {
__m128 zt;
- zt = _mm_mul_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right)));
+ zt = _mm_mul_ps(quat::load_xmm(left), quat::load_xmm(right));
zt = _mm_hadd_ps(zt, zt);
return _mm_cvtss_f32(_mm_hadd_ps(zt, zt));
}
@@ -278,7 +229,7 @@ inline float_t quat::dot(const quat& left, const quat& right) {
inline float_t quat::length(const quat& left) {
__m128 xt;
__m128 zt;
- xt = _mm_loadu_ps((const float*)&left);
+ xt = quat::load_xmm(left);
zt = _mm_mul_ps(xt, xt);
zt = _mm_hadd_ps(zt, zt);
return _mm_cvtss_f32(_mm_sqrt_ss(_mm_hadd_ps(zt, zt)));
@@ -287,7 +238,7 @@ inline float_t quat::length(const quat& left) {
inline float_t quat::length_squared(const quat& left) {
__m128 xt;
__m128 zt;
- xt = _mm_loadu_ps((const float*)&left);
+ xt = quat::load_xmm(left);
zt = _mm_mul_ps(xt, xt);
zt = _mm_hadd_ps(zt, zt);
return _mm_cvtss_f32(_mm_hadd_ps(zt, zt));
@@ -295,7 +246,7 @@ inline float_t quat::length_squared(const quat& left) {
inline float_t quat::distance(const quat& left, const quat& right) {
__m128 zt;
- zt = _mm_sub_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right)));
+ zt = _mm_sub_ps(quat::load_xmm(left), quat::load_xmm(right));
zt = _mm_mul_ps(zt, zt);
zt = _mm_hadd_ps(zt, zt);
return _mm_cvtss_f32(_mm_sqrt_ss(_mm_hadd_ps(zt, zt)));
@@ -303,7 +254,7 @@ inline float_t quat::distance(const quat& left, const quat& right) {
inline float_t quat::distance_squared(const quat& left, const quat& right) {
__m128 zt;
- zt = _mm_sub_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right)));
+ zt = _mm_sub_ps(quat::load_xmm(left), quat::load_xmm(right));
zt = _mm_mul_ps(zt, zt);
zt = _mm_hadd_ps(zt, zt);
return _mm_cvtss_f32(_mm_hadd_ps(zt, zt));
@@ -352,64 +303,58 @@ inline quat quat::slerp(const quat& left, const quat& right, const float_t blend
inline quat quat::normalize(const quat& left) {
__m128 xt;
__m128 zt;
- quat z;
- xt = _mm_loadu_ps((const float*)&left);
+ xt = quat::load_xmm(left);
zt = _mm_mul_ps(xt, xt);
zt = _mm_hadd_ps(zt, zt);
zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt));
- if (zt.m128_f32[0] != 0.0f)
- zt.m128_f32[0] = 1.0f / zt.m128_f32[0];
- _mm_storeu_ps((float*)&z, _mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0)));
- return z;
+ if (_mm_cvtss_f32(zt) != 0.0f)
+ return quat::store_xmm(_mm_div_ps(xt, _mm_shuffle_ps(zt, zt, 0)));
+ return quat::store_xmm(xt);
+}
+
+inline quat quat::normalize_rcp(const quat& left) {
+ __m128 xt;
+ __m128 zt;
+ xt = quat::load_xmm(left);
+ zt = _mm_mul_ps(xt, xt);
+ zt = _mm_hadd_ps(zt, zt);
+ zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt));
+ if (_mm_cvtss_f32(zt) != 0.0f)
+ zt = _mm_div_ss(quat::load_xmm(1.0f), zt);
+ return quat::store_xmm(_mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0)));
}
inline quat quat::rcp(const quat& left) {
- quat z;
- _mm_storeu_ps((float*)&z, _mm_div_ps(_mm_loadu_ps((const float*)&(quat_identity)), _mm_loadu_ps((const float*)&left)));
- return z;
+ return quat::store_xmm(_mm_div_ps(quat::load_xmm(1.0f), quat::load_xmm(left)));
}
inline quat quat::min(const quat& left, const quat& right) {
- quat z;
- _mm_storeu_ps((float*)&z, _mm_min_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right))));
- return z;
+ return quat::store_xmm(_mm_min_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat quat::max(const quat& left, const quat& right) {
- quat z;
- _mm_storeu_ps((float*)&z, _mm_max_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right))));
- return z;
+ return quat::store_xmm(_mm_max_ps(quat::load_xmm(left), quat::load_xmm(right)));
}
inline quat quat::clamp(const quat& left, const quat& min, const quat& max) {
- quat w;
- _mm_storeu_ps((float*)&w, _mm_min_ps(_mm_max_ps(_mm_loadu_ps((const float*)&left),
- _mm_loadu_ps((const float*)&(min))), _mm_loadu_ps((const float*)&(max))));
- return w;
+ return quat::store_xmm(_mm_min_ps(_mm_max_ps(quat::load_xmm(left),
+ quat::load_xmm(min)), quat::load_xmm(max)));
}
inline quat quat::clamp(const quat& left, const float_t min, const float_t max) {
- __m128 yt;
- __m128 zt;
- quat w;
- yt = _mm_set_ss(min);
- zt = _mm_set_ss(max);
- _mm_storeu_ps((float*)&w, _mm_min_ps(_mm_max_ps(_mm_loadu_ps((const float*)&left),
- _mm_shuffle_ps(yt, yt, 0)), _mm_shuffle_ps(zt, zt, 0)));
- return w;
+ return quat::store_xmm(_mm_min_ps(_mm_max_ps(quat::load_xmm(left),
+ quat::load_xmm(min)), quat::load_xmm(max)));
}
inline quat quat::mult_min_max(const quat& left, const quat& min, const quat& max) {
__m128 xt;
__m128 yt;
__m128 wt;
- quat w;
- xt = _mm_loadu_ps((const float*)&left);
- yt = _mm_xor_ps(_mm_loadu_ps((const float*)&(min)), vec4_neg);
+ xt = quat::load_xmm(left);
+ yt = _mm_xor_ps(quat::load_xmm(min), vec4_neg);
wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))),
- _mm_and_ps(_mm_loadu_ps((const float*)&(max)), _mm_cmpge_ps(xt, vec4::load_xmm(0.0f))));
- _mm_storeu_ps((float*)&w, _mm_mul_ps(xt, wt));
- return w;
+ _mm_and_ps(quat::load_xmm(max), _mm_cmpge_ps(xt, vec4::load_xmm(0.0f))));
+ return quat::store_xmm(_mm_mul_ps(xt, wt));
}
inline quat quat::mult_min_max(const quat& left, const float_t min, const float_t max) {
@@ -417,30 +362,24 @@ inline quat quat::mult_min_max(const quat& left, const float_t min, const float_
__m128 yt;
__m128 zt;
__m128 wt;
- quat w;
- xt = _mm_loadu_ps((const float*)&left);
- yt = _mm_set_ss(min);
- yt = _mm_shuffle_ps(yt, yt, 0);
- zt = _mm_set_ss(max);
- zt = _mm_shuffle_ps(zt, zt, 0);
+ xt = quat::load_xmm(left);
+ yt = quat::load_xmm(min);
+ zt = quat::load_xmm(max);
yt = _mm_xor_ps(yt, vec4_neg);
wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))),
_mm_and_ps(zt, _mm_cmpge_ps(xt, vec4::load_xmm(0.0f))));
- _mm_storeu_ps((float*)&w, _mm_mul_ps(xt, wt));
- return w;
+ return quat::store_xmm(_mm_mul_ps(xt, wt));
}
inline quat quat::div_min_max(const quat& left, const quat& min, const quat& max) {
__m128 xt;
__m128 yt;
__m128 wt;
- quat w;
- xt = _mm_loadu_ps((const float*)&left);
- yt = _mm_xor_ps(_mm_loadu_ps((const float*)&(min)), vec4_neg);
+ xt = quat::load_xmm(left);
+ yt = _mm_xor_ps(quat::load_xmm(min), vec4_neg);
wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))),
- _mm_and_ps(_mm_loadu_ps((const float*)&(max)), _mm_cmpge_ps(xt, vec4::load_xmm(0.0f))));
- _mm_storeu_ps((float*)&w, _mm_div_ps(xt, wt));
- return w;
+ _mm_and_ps(quat::load_xmm(max), _mm_cmpge_ps(xt, vec4::load_xmm(0.0f))));
+ return quat::store_xmm(_mm_div_ps(xt, wt));
}
inline quat quat::div_min_max(const quat& left, const float_t min, const float_t max) {
@@ -448,15 +387,11 @@ inline quat quat::div_min_max(const quat& left, const float_t min, const float_t
__m128 yt;
__m128 zt;
__m128 wt;
- quat w;
- xt = _mm_loadu_ps((const float*)&left);
- yt = _mm_set_ss(min);
- yt = _mm_shuffle_ps(yt, yt, 0);
- zt = _mm_set_ss(max);
- zt = _mm_shuffle_ps(zt, zt, 0);
+ xt = quat::load_xmm(left);
+ yt = quat::load_xmm(min);
+ zt = quat::load_xmm(max);
yt = _mm_xor_ps(yt, vec4_neg);
wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))),
_mm_and_ps(zt, _mm_cmpge_ps(xt, vec4::load_xmm(0.0f))));
- _mm_storeu_ps((float*)&w, _mm_div_ps(xt, wt));
- return w;
+ return quat::store_xmm(_mm_div_ps(xt, wt));
}
diff --git a/src/KKdLib/vec.cpp b/src/KKdLib/vec.cpp
index f499c76..375c958 100644
--- a/src/KKdLib/vec.cpp
+++ b/src/KKdLib/vec.cpp
@@ -9,6 +9,8 @@ const __m128 vec2_neg = { -0.0f, -0.0f, 0.0f, 0.0f };
const __m128 vec3_neg = { -0.0f, -0.0f, -0.0f, 0.0f };
const __m128 vec4_neg = { -0.0f, -0.0f, -0.0f, -0.0f };
+extern const __m128d vec2d_neg = { -0.0, -0.0f };
+
const __m128i vec2i_abs = {
(char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F,
(char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F,
@@ -29,3 +31,8 @@ const __m128i vec4i_abs = {
(char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F,
(char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F,
};
+
+extern const __m128i vec2i64_abs = {
+ (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F,
+ (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F,
+};
diff --git a/src/KKdLib/vec.hpp b/src/KKdLib/vec.hpp
index 3d68ce4..f605087 100644
--- a/src/KKdLib/vec.hpp
+++ b/src/KKdLib/vec.hpp
@@ -457,14 +457,54 @@ struct vec4i {
static vec4i clamp(const vec4i& left, const int32_t min, const int32_t max);
};
+struct vec2d {
+ double_t x;
+ double_t y;
+
+ vec2d();
+ vec2d(double_t value);
+ vec2d(double_t x, double_t y);
+
+ static __m128d load_xmm(const double_t data);
+ static __m128d load_xmm(const vec2d& data);
+ static __m128d load_xmm(const vec2d&& data);
+ static vec2d store_xmm(const __m128d& data);
+ static vec2d store_xmm(const __m128d&& data);
+
+ static double_t angle(const vec2d& left, const vec2d& right);
+ static double_t dot(const vec2d& left, const vec2d& right);
+ static double_t length(const vec2d& left);
+ static double_t length_squared(const vec2d& left);
+ static double_t distance(const vec2d& left, const vec2d& right);
+ static double_t distance_squared(const vec2d& left, const vec2d& right);
+ static vec2d abs(const vec2d& left);
+ static vec2d lerp(const vec2d& left, const vec2d& right, const vec2d& blend);
+ static vec2d lerp(const vec2d& left, const vec2d& right, const double_t blend);
+ static vec2d normalize(const vec2d& left);
+ static vec2d normalize_rcp(const vec2d& left);
+ static vec2d rcp(const vec2d& left);
+ static vec2d min(const vec2d& left, const vec2d& right);
+ static vec2d max(const vec2d& left, const vec2d& right);
+ static vec2d clamp(const vec2d& left, const vec2d& min, const vec2d& max);
+ static vec2d clamp(const vec2d& left, const double_t min, const double_t max);
+ static vec2d mult_min_max(const vec2d& left, const vec2d& min, const vec2d& max);
+ static vec2d mult_min_max(const vec2d& left, const double_t min, const double_t max);
+ static vec2d div_min_max(const vec2d& left, const vec2d& min, const vec2d& max);
+ static vec2d div_min_max(const vec2d& left, const double_t min, const double_t max);
+};
+
extern const __m128 vec2_neg;
extern const __m128 vec3_neg;
extern const __m128 vec4_neg;
+extern const __m128d vec2d_neg;
+
extern const __m128i vec2i_abs;
extern const __m128i vec3i_abs;
extern const __m128i vec4i_abs;
+extern const __m128i vec2i64_abs;
+
inline vec2::vec2() : x(), y() {
}
@@ -478,7 +518,7 @@ inline vec2::vec2(float_t x, float_t y) : x(x), y(y) {
}
inline __m128 vec2::load_xmm(const float_t data) {
- __m128 _data = _mm_set_ss(data);
+ __m128 _data = _mm_load_ss(&data);
return _mm_shuffle_ps(_data, _data, 0x50);
}
@@ -714,7 +754,7 @@ inline vec2 vec2::normalize_rcp(const vec2& left) {
zt = _mm_mul_ps(xt, xt);
zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt));
if (_mm_cvtss_f32(zt) != 0.0f)
- zt = _mm_div_ss(_mm_set_ss(1.0f), zt);
+ zt = _mm_div_ss(vec4::load_xmm(1.0f), zt);
return vec2::store_xmm(_mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0)));
}
@@ -925,7 +965,7 @@ inline bool operator !=(const vec3& left, const vec3& right) {
}
inline __m128 vec3::load_xmm(const float_t data) {
- __m128 _data = _mm_set_ss(data);
+ __m128 _data = _mm_load_ss(&data);
return _mm_shuffle_ps(_data, _data, 0x40);
}
@@ -1046,7 +1086,7 @@ inline vec3 vec3::normalize_rcp(const vec3& left) {
zt = _mm_hadd_ps(zt, zt);
zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt));
if (_mm_cvtss_f32(zt) != 0.0f)
- zt = _mm_div_ss(_mm_set_ss(1.0f), zt);
+ zt = _mm_div_ss(vec4::load_xmm(1.0f), zt);
return vec3::store_xmm(_mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0)));
}
@@ -1270,7 +1310,7 @@ inline bool operator !=(const vec4& left, const vec4& right) {
}
inline __m128 vec4::load_xmm(const float_t data) {
- __m128 _data = _mm_set_ss(data);
+ __m128 _data = _mm_load_ss(&data);
return _mm_shuffle_ps(_data, _data, 0);
}
@@ -1381,7 +1421,7 @@ inline vec4 vec4::normalize_rcp(const vec4& left) {
zt = _mm_hadd_ps(zt, zt);
zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt));
if (_mm_cvtss_f32(zt) != 0.0f)
- zt = _mm_div_ss(_mm_set_ss(1.0f), zt);
+ zt = _mm_div_ss(vec4::load_xmm(1.0f), zt);
return vec4::store_xmm(_mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0)));
}
@@ -1686,6 +1726,321 @@ inline vec4i vec4i::clamp(const vec4i& left, const int32_t min, const int32_t ma
vec4i::load_xmm(min)), vec4i::load_xmm(max)));
}
+inline vec2d::vec2d() : x(), y() {
+
+}
+
+inline vec2d::vec2d(double_t value) : x(value), y(value) {
+
+}
+
+inline vec2d::vec2d(double_t x, double_t y) : x(x), y(y) {
+
+}
+
+inline __m128d vec2d::load_xmm(const double_t data) {
+ __m128d _data = _mm_load_sd(&data);
+ return _mm_shuffle_pd(_data, _data, 0);
+}
+
+inline __m128d vec2d::load_xmm(const vec2d& data) {
+ return _mm_loadu_pd((const double_t*) & data);
+}
+
+inline __m128d vec2d::load_xmm(const vec2d&& data) {
+ return _mm_loadu_pd((const double_t*) & data);
+}
+
+inline vec2d vec2d::store_xmm(const __m128d& data) {
+ vec2d _data;
+ _mm_storeu_pd((double_t*) & _data, data);
+ return _data;
+}
+
+inline vec2d vec2d::store_xmm(const __m128d&& data) {
+ vec2d _data;
+ _mm_storeu_pd((double_t*) & _data, data);
+ return _data;
+}
+
+inline vec2d operator +(const vec2d& left, const vec2d& right) {
+ return vec2d::store_xmm(_mm_add_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator +(const vec2d& left, const double_t right) {
+ return vec2d::store_xmm(_mm_add_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator +(const double_t left, const vec2d& right) {
+ return vec2d::store_xmm(_mm_add_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline void operator +=(vec2d& left, const vec2d& right) {
+ left = vec2d::store_xmm(_mm_add_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline void operator +=(vec2d& left, const double_t right) {
+ left = vec2d::store_xmm(_mm_add_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator -(const vec2d& left, const vec2d& right) {
+ return vec2d::store_xmm(_mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator -(const vec2d& left, const double_t right) {
+ return vec2d::store_xmm(_mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator -(const double_t left, const vec2d& right) {
+ return vec2d::store_xmm(_mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline void operator -=(vec2d& left, const vec2d& right) {
+ left = vec2d::store_xmm(_mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline void operator -=(vec2d& left, const double_t right) {
+ left = vec2d::store_xmm(_mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator *(const vec2d& left, const vec2d& right) {
+ return vec2d::store_xmm(_mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator *(const vec2d& left, const double_t right) {
+ return vec2d::store_xmm(_mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator *(const double_t left, const vec2d& right) {
+ return vec2d::store_xmm(_mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline void operator *=(vec2d& left, const vec2d& right) {
+ left = vec2d::store_xmm(_mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline void operator *=(vec2d& left, const double_t right) {
+ left = vec2d::store_xmm(_mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator /(const vec2d& left, const vec2d& right) {
+ return vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator /(const vec2d& left, const double_t right) {
+ return vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator /(const double_t left, const vec2d& right) {
+ return vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline void operator /=(vec2d& left, const vec2d& right) {
+ left = vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline void operator /=(vec2d& left, const double_t right) {
+ left = vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator &(const vec2d& left, const vec2d& right) {
+ return vec2d::store_xmm(_mm_and_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator &(const vec2d& left, const double_t right) {
+ return vec2d::store_xmm(_mm_and_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator &(const double_t left, const vec2d& right) {
+ return vec2d::store_xmm(_mm_and_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline void operator &=(vec2d& left, const vec2d& right) {
+ left = vec2d::store_xmm(_mm_and_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline void operator &=(vec2d& left, const double_t right) {
+ left = vec2d::store_xmm(_mm_and_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator ^(const vec2d& left, const vec2d& right) {
+ return vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator ^(const vec2d& left, const double_t right) {
+ return vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator ^(const double_t left, const vec2d& right) {
+ return vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline void operator ^=(vec2d& left, const vec2d& right) {
+ left = vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline void operator ^=(vec2d& left, const double_t right) {
+ left = vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d operator -(const vec2d& left) {
+ return vec2d::store_xmm(_mm_xor_pd(vec2d::load_xmm(left), vec2d_neg));
+}
+
+inline bool operator ==(const vec2d& left, const vec2d& right) {
+ return !memcmp(&left, &right, sizeof(vec2d));
+}
+
+inline bool operator !=(const vec2d& left, const vec2d& right) {
+ return !!memcmp(&left, &right, sizeof(vec2d));
+}
+
+inline double_t vec2d::angle(const vec2d& left, const vec2d& right) {
+ return acos(vec2d::dot(left, right) / (vec2d::length(left) * vec2d::length(right)));
+}
+
+inline double_t vec2d::dot(const vec2d& left, const vec2d& right) {
+ __m128d zt;
+ zt = _mm_mul_pd(vec2d::load_xmm(left), vec2d::load_xmm(right));
+ return _mm_cvtsd_f64(_mm_hadd_pd(zt, zt));
+}
+
+inline double_t vec2d::length(const vec2d& left) {
+ __m128d xt;
+ __m128d zt;
+ xt = vec2d::load_xmm(left);
+ zt = _mm_mul_pd(xt, xt);
+ return _mm_cvtsd_f64(_mm_sqrt_sd(_mm_hadd_pd(zt, zt), zt));
+}
+
+inline double_t vec2d::length_squared(const vec2d& left) {
+ __m128d xt;
+ __m128d zt;
+ xt = vec2d::load_xmm(left);
+ zt = _mm_mul_pd(xt, xt);
+ return _mm_cvtsd_f64(_mm_hadd_pd(zt, zt));
+}
+
+inline double_t vec2d::distance(const vec2d& left, const vec2d& right) {
+ __m128d zt;
+ zt = _mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right));
+ zt = _mm_mul_pd(zt, zt);
+ return _mm_cvtsd_f64(_mm_sqrt_sd(_mm_hadd_pd(zt, zt), zt));
+}
+
+inline double_t vec2d::distance_squared(const vec2d& left, const vec2d& right) {
+ __m128d zt;
+ zt = _mm_sub_pd(vec2d::load_xmm(left), vec2d::load_xmm(right));
+ zt = _mm_mul_pd(zt, zt);
+ return _mm_cvtsd_f64(_mm_hadd_pd(zt, zt));
+}
+
+inline vec2d vec2d::abs(const vec2d& left) {
+ return vec2d::store_xmm(_mm_castsi128_pd(_mm_and_si128(_mm_castpd_si128(vec2d::load_xmm(left)), vec2i64_abs)));
+}
+
+inline vec2d vec2d::lerp(const vec2d& left, const vec2d& right, const vec2d& blend) {
+ __m128d b1;
+ __m128d b2;
+ b1 = vec2d::load_xmm(blend);
+ b2 = _mm_sub_pd(vec2d::load_xmm(1.0), b1);
+ return vec2d::store_xmm(_mm_add_pd(_mm_mul_pd(vec2d::load_xmm(left), b2),
+ _mm_mul_pd(vec2d::load_xmm(right), b1)));
+}
+
+inline vec2d vec2d::lerp(const vec2d& left, const vec2d& right, const double_t blend) {
+ __m128d b1;
+ __m128d b2;
+ b1 = vec2d::load_xmm(blend);
+ b2 = _mm_sub_pd(vec2d::load_xmm(1.0), b1);
+ return vec2d::store_xmm(_mm_add_pd(_mm_mul_pd(vec2d::load_xmm(left), b2),
+ _mm_mul_pd(vec2d::load_xmm(right), b1)));
+}
+
+inline vec2d vec2d::normalize(const vec2d& left) {
+ __m128d xt;
+ __m128d zt;
+ xt = vec2d::load_xmm(left);
+ zt = _mm_mul_pd(xt, xt);
+ zt = _mm_sqrt_sd(_mm_hadd_pd(zt, zt), zt);
+ if (_mm_cvtsd_f64(zt) != 0.0f)
+ return vec2d::store_xmm(_mm_div_pd(xt, _mm_shuffle_pd(zt, zt, 0)));
+ return vec2d::store_xmm(xt);
+}
+
+inline vec2d vec2d::normalize_rcp(const vec2d& left) {
+ __m128d xt;
+ __m128d zt;
+ xt = vec2d::load_xmm(left);
+ zt = _mm_mul_pd(xt, xt);
+ zt = _mm_sqrt_sd(_mm_hadd_pd(zt, zt), zt);
+ if (_mm_cvtsd_f64(zt) != 0.0f)
+ zt = _mm_div_sd(vec2d::load_xmm(1.0), zt);
+ return vec2d::store_xmm(_mm_mul_pd(xt, _mm_shuffle_pd(zt, zt, 0)));
+}
+
+inline vec2d vec2d::rcp(const vec2d& left) {
+ return vec2d::store_xmm(_mm_div_pd(vec2d::load_xmm(1.0), vec2d::load_xmm(left)));
+}
+
+inline vec2d vec2d::min(const vec2d& left, const vec2d& right) {
+ return vec2d::store_xmm(_mm_min_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d vec2d::max(const vec2d& left, const vec2d& right) {
+ return vec2d::store_xmm(_mm_max_pd(vec2d::load_xmm(left), vec2d::load_xmm(right)));
+}
+
+inline vec2d vec2d::clamp(const vec2d& left, const vec2d& min, const vec2d& max) {
+ return vec2d::store_xmm(_mm_min_pd(_mm_max_pd(vec2d::load_xmm(left),
+ vec2d::load_xmm(min)), vec2d::load_xmm(max)));
+}
+
+inline vec2d vec2d::clamp(const vec2d& left, const double_t min, const double_t max) {
+ return vec2d::store_xmm(_mm_min_pd(_mm_max_pd(vec2d::load_xmm(left),
+ vec2d::load_xmm(min)), vec2d::load_xmm(max)));
+}
+
+inline vec2d vec2d::mult_min_max(const vec2d& left, const vec2d& min, const vec2d& max) {
+ __m128d xt;
+ __m128d yt;
+ __m128d zt;
+ xt = vec2d::load_xmm(left);
+ yt = _mm_and_pd(_mm_cmplt_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(-min));
+ zt = _mm_and_pd(_mm_cmpge_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(max));
+ return vec2d::store_xmm(_mm_mul_pd(xt, _mm_or_pd(yt, zt)));
+}
+
+inline vec2d vec2d::mult_min_max(const vec2d& left, const double_t min, const double_t max) {
+ __m128d xt;
+ __m128d yt;
+ __m128d zt;
+ xt = vec2d::load_xmm(left);
+ yt = _mm_and_pd(_mm_cmplt_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(-min));
+ zt = _mm_and_pd(_mm_cmpge_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(max));
+ return vec2d::store_xmm(_mm_mul_pd(xt, _mm_or_pd(yt, zt)));
+}
+
+inline vec2d vec2d::div_min_max(const vec2d& left, const vec2d& min, const vec2d& max) {
+ __m128d xt;
+ __m128d yt;
+ __m128d zt;
+ xt = vec2d::load_xmm(left);
+ yt = _mm_and_pd(_mm_cmplt_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(-min));
+ zt = _mm_and_pd(_mm_cmpge_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(max));
+ return vec2d::store_xmm(_mm_div_pd(xt, _mm_or_pd(yt, zt)));
+}
+
+inline vec2d vec2d::div_min_max(const vec2d& left, const double_t min, const double_t max) {
+ __m128d xt;
+ __m128d yt;
+ __m128d zt;
+ xt = vec2d::load_xmm(left);
+ yt = _mm_and_pd(_mm_cmplt_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(-min));
+ zt = _mm_and_pd(_mm_cmpge_pd(xt, vec2d::load_xmm(0.0)), vec2d::load_xmm(max));
+ return vec2d::store_xmm(_mm_div_pd(xt, _mm_or_pd(yt, zt)));
+}
+
inline void vec2i8_to_vec2(const vec2i8& src, vec2& dst) {
dst.x = (float_t)src.x;
dst.y = (float_t)src.y;