From d9027ce341e7a5ffe9865a0c361830251c63bde3 Mon Sep 17 00:00:00 2001 From: korenkonder Date: Tue, 19 Dec 2023 03:04:58 +0300 Subject: [PATCH] KKdLib --- .gitignore | 1 - DivaGL.sln | 18 +- src/DivaGL/auth_3d.cpp | 5 +- src/DivaGL/file_handler.cpp | 3 - src/DivaGL/light_param/fog.hpp | 15 +- src/DivaGL/light_param/light.hpp | 41 +- src/DivaGL/object.cpp | 22 + src/DivaGL/render.hpp | 8 +- src/DivaGL/resolution_mode.hpp | 59 +- src/DivaGL/sprite.hpp | 20 + src/DivaGL/texture.hpp | 1 - src/KKdLib/KKdLib.rc | 46 + src/KKdLib/KKdLib.vcxproj | 279 ++++ src/KKdLib/KKdLib.vcxproj.user | 22 + src/KKdLib/aes.cpp | 1346 ++++++++++++++++ src/KKdLib/aes.hpp | 90 ++ src/KKdLib/default.cpp | 345 ++++ src/KKdLib/default.hpp | 543 +++++++ src/KKdLib/deflate.cpp | 187 +++ src/KKdLib/deflate.hpp | 23 + src/KKdLib/divafile.cpp | 155 ++ src/KKdLib/divafile.hpp | 19 + src/KKdLib/f2/enrs.cpp | 286 ++++ src/KKdLib/f2/enrs.hpp | 50 + src/KKdLib/f2/header.cpp | 37 + src/KKdLib/f2/header.hpp | 39 + src/KKdLib/f2/pof.cpp | 206 +++ src/KKdLib/f2/pof.hpp | 29 + src/KKdLib/f2/struct.cpp | 245 +++ src/KKdLib/f2/struct.hpp | 34 + src/KKdLib/farc.cpp | 950 +++++++++++ src/KKdLib/farc.hpp | 84 + src/KKdLib/half_t.cpp | 82 + src/KKdLib/half_t.hpp | 67 + src/KKdLib/hash.cpp | 225 +++ src/KKdLib/hash.hpp | 178 +++ src/KKdLib/interpolation.cpp | 248 +++ src/KKdLib/interpolation.hpp | 233 +++ src/KKdLib/io/file_stream.cpp | 176 +++ src/KKdLib/io/file_stream.hpp | 49 + src/KKdLib/io/memory_stream.cpp | 255 +++ src/KKdLib/io/memory_stream.hpp | 55 + src/KKdLib/io/path.cpp | 395 +++++ src/KKdLib/io/path.hpp | 27 + src/KKdLib/io/stream.cpp | 716 +++++++++ src/KKdLib/io/stream.hpp | 157 ++ src/KKdLib/kf.cpp | 81 + src/KKdLib/kf.hpp | 39 + src/KKdLib/mat.cpp | 2008 +++++++++++++++++++++++ src/KKdLib/mat.hpp | 345 ++++ src/KKdLib/msgpack.cpp | 540 +++++++ src/KKdLib/msgpack.hpp | 277 ++++ src/KKdLib/prj/algorithm.hpp | 56 + src/KKdLib/prj/math.hpp | 43 + src/KKdLib/prj/shared_ptr.hpp | 284 ++++ src/KKdLib/prj/stack_allocator.cpp | 104 ++ src/KKdLib/prj/stack_allocator.hpp | 85 + src/KKdLib/prj/time.cpp | 39 + src/KKdLib/prj/time.hpp | 27 + src/KKdLib/prj/vector_pair.hpp | 144 ++ src/KKdLib/prj/vector_pair_combine.hpp | 262 +++ src/KKdLib/quat.cpp | 10 + src/KKdLib/quat.hpp | 462 ++++++ src/KKdLib/str_utils.cpp | 690 ++++++++ src/KKdLib/str_utils.hpp | 59 + src/KKdLib/time.cpp | 39 + src/KKdLib/time.hpp | 19 + src/KKdLib/timer.cpp | 110 ++ src/KKdLib/timer.hpp | 37 + src/KKdLib/txp.cpp | 429 +++++ src/KKdLib/txp.hpp | 66 + src/KKdLib/types.hpp | 32 + src/KKdLib/vec.cpp | 31 + src/KKdLib/vec.hpp | 2021 ++++++++++++++++++++++++ src/KKdLib/waitable_timer.hpp | 57 + 75 files changed, 16446 insertions(+), 21 deletions(-) create mode 100644 src/KKdLib/KKdLib.rc create mode 100644 src/KKdLib/KKdLib.vcxproj create mode 100644 src/KKdLib/KKdLib.vcxproj.user create mode 100644 src/KKdLib/aes.cpp create mode 100644 src/KKdLib/aes.hpp create mode 100644 src/KKdLib/default.cpp create mode 100644 src/KKdLib/default.hpp create mode 100644 src/KKdLib/deflate.cpp create mode 100644 src/KKdLib/deflate.hpp create mode 100644 src/KKdLib/divafile.cpp create mode 100644 src/KKdLib/divafile.hpp create mode 100644 src/KKdLib/f2/enrs.cpp create mode 100644 src/KKdLib/f2/enrs.hpp create mode 100644 src/KKdLib/f2/header.cpp create mode 100644 src/KKdLib/f2/header.hpp create mode 100644 src/KKdLib/f2/pof.cpp create mode 100644 src/KKdLib/f2/pof.hpp create mode 100644 src/KKdLib/f2/struct.cpp create mode 100644 src/KKdLib/f2/struct.hpp create mode 100644 src/KKdLib/farc.cpp create mode 100644 src/KKdLib/farc.hpp create mode 100644 src/KKdLib/half_t.cpp create mode 100644 src/KKdLib/half_t.hpp create mode 100644 src/KKdLib/hash.cpp create mode 100644 src/KKdLib/hash.hpp create mode 100644 src/KKdLib/interpolation.cpp create mode 100644 src/KKdLib/interpolation.hpp create mode 100644 src/KKdLib/io/file_stream.cpp create mode 100644 src/KKdLib/io/file_stream.hpp create mode 100644 src/KKdLib/io/memory_stream.cpp create mode 100644 src/KKdLib/io/memory_stream.hpp create mode 100644 src/KKdLib/io/path.cpp create mode 100644 src/KKdLib/io/path.hpp create mode 100644 src/KKdLib/io/stream.cpp create mode 100644 src/KKdLib/io/stream.hpp create mode 100644 src/KKdLib/kf.cpp create mode 100644 src/KKdLib/kf.hpp create mode 100644 src/KKdLib/mat.cpp create mode 100644 src/KKdLib/mat.hpp create mode 100644 src/KKdLib/msgpack.cpp create mode 100644 src/KKdLib/msgpack.hpp create mode 100644 src/KKdLib/prj/algorithm.hpp create mode 100644 src/KKdLib/prj/math.hpp create mode 100644 src/KKdLib/prj/shared_ptr.hpp create mode 100644 src/KKdLib/prj/stack_allocator.cpp create mode 100644 src/KKdLib/prj/stack_allocator.hpp create mode 100644 src/KKdLib/prj/time.cpp create mode 100644 src/KKdLib/prj/time.hpp create mode 100644 src/KKdLib/prj/vector_pair.hpp create mode 100644 src/KKdLib/prj/vector_pair_combine.hpp create mode 100644 src/KKdLib/quat.cpp create mode 100644 src/KKdLib/quat.hpp create mode 100644 src/KKdLib/str_utils.cpp create mode 100644 src/KKdLib/str_utils.hpp create mode 100644 src/KKdLib/time.cpp create mode 100644 src/KKdLib/time.hpp create mode 100644 src/KKdLib/timer.cpp create mode 100644 src/KKdLib/timer.hpp create mode 100644 src/KKdLib/txp.cpp create mode 100644 src/KKdLib/txp.hpp create mode 100644 src/KKdLib/types.hpp create mode 100644 src/KKdLib/vec.cpp create mode 100644 src/KKdLib/vec.hpp create mode 100644 src/KKdLib/waitable_timer.hpp diff --git a/.gitignore b/.gitignore index 3517e98..89beb26 100644 --- a/.gitignore +++ b/.gitignore @@ -2,7 +2,6 @@ bin lib obj -src/KKdLib *.aps *.cs !extern/lib diff --git a/DivaGL.sln b/DivaGL.sln index 3425cc9..e37d03e 100644 --- a/DivaGL.sln +++ b/DivaGL.sln @@ -4,7 +4,7 @@ VisualStudioVersion = 16.0.30320.27 MinimumVisualStudioVersion = 10.0.40219.1 Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "DivaGL", "src\DivaGL\DivaGL.vcxproj", "{4D949362-9095-4113-9423-7E0492618F16}" EndProject -Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "KKdLib", "src\KKdLib\KKdLib.vcxproj", "{4C9636B9-E5B6-41E5-9AFC-5B33CC9633DE}" +Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "KKdLib", "src\KKdLib\KKdLib.vcxproj", "{65707AD3-568A-4FE9-A75E-BDBF3D5EFC88}" EndProject Global GlobalSection(SolutionConfigurationPlatforms) = preSolution @@ -22,14 +22,14 @@ Global {4D949362-9095-4113-9423-7E0492618F16}.Release|x64.Build.0 = Release|x64 {4D949362-9095-4113-9423-7E0492618F16}.ReleaseDLL|x64.ActiveCfg = ReleaseDLL|x64 {4D949362-9095-4113-9423-7E0492618F16}.ReleaseDLL|x64.Build.0 = ReleaseDLL|x64 - {4C9636B9-E5B6-41E5-9AFC-5B33CC9633DE}.Debug|x64.ActiveCfg = DebugOpt|x64 - {4C9636B9-E5B6-41E5-9AFC-5B33CC9633DE}.Debug|x64.Build.0 = DebugOpt|x64 - {4C9636B9-E5B6-41E5-9AFC-5B33CC9633DE}.DebugDLL|x64.ActiveCfg = DebugOpt|x64 - {4C9636B9-E5B6-41E5-9AFC-5B33CC9633DE}.DebugDLL|x64.Build.0 = DebugOpt|x64 - {4C9636B9-E5B6-41E5-9AFC-5B33CC9633DE}.Release|x64.ActiveCfg = Release|x64 - {4C9636B9-E5B6-41E5-9AFC-5B33CC9633DE}.Release|x64.Build.0 = Release|x64 - {4C9636B9-E5B6-41E5-9AFC-5B33CC9633DE}.ReleaseDLL|x64.ActiveCfg = Release|x64 - {4C9636B9-E5B6-41E5-9AFC-5B33CC9633DE}.ReleaseDLL|x64.Build.0 = Release|x64 + {65707AD3-568A-4FE9-A75E-BDBF3D5EFC88}.Debug|x64.ActiveCfg = DebugOpt|x64 + {65707AD3-568A-4FE9-A75E-BDBF3D5EFC88}.Debug|x64.Build.0 = DebugOpt|x64 + {65707AD3-568A-4FE9-A75E-BDBF3D5EFC88}.DebugDLL|x64.ActiveCfg = DebugOpt|x64 + {65707AD3-568A-4FE9-A75E-BDBF3D5EFC88}.DebugDLL|x64.Build.0 = DebugOpt|x64 + {65707AD3-568A-4FE9-A75E-BDBF3D5EFC88}.Release|x64.ActiveCfg = Release|x64 + {65707AD3-568A-4FE9-A75E-BDBF3D5EFC88}.Release|x64.Build.0 = Release|x64 + {65707AD3-568A-4FE9-A75E-BDBF3D5EFC88}.ReleaseDLL|x64.ActiveCfg = Release|x64 + {65707AD3-568A-4FE9-A75E-BDBF3D5EFC88}.ReleaseDLL|x64.Build.0 = Release|x64 EndGlobalSection GlobalSection(SolutionProperties) = preSolution HideSolutionNode = FALSE diff --git a/src/DivaGL/auth_3d.cpp b/src/DivaGL/auth_3d.cpp index 7b5667c..2435596 100644 --- a/src/DivaGL/auth_3d.cpp +++ b/src/DivaGL/auth_3d.cpp @@ -4,7 +4,6 @@ */ #include "auth_3d.hpp" -#include "../KKdLib/database/item_table.hpp" #include "light_param/fog.hpp" #include "mdl/disp_manager.hpp" #include "rob/rob.hpp" @@ -447,8 +446,8 @@ struct auth_3d { object_info object_info; mat4* bone_mats; bool shadow; - chara_index src_chara; - chara_index dst_chara; + int32_t src_chara; // chara_index + int32_t dst_chara; // chara_index int32_t pos; int64_t field_A8; void* frame_rate; // FrameRateControl diff --git a/src/DivaGL/file_handler.cpp b/src/DivaGL/file_handler.cpp index 2a201f4..55f76a4 100644 --- a/src/DivaGL/file_handler.cpp +++ b/src/DivaGL/file_handler.cpp @@ -4,10 +4,7 @@ */ #include -#include "../KKdLib/io/file_stream.hpp" -#include "../KKdLib/hash.hpp" #include "file_handler.hpp" -#include void p_file_handler::call_free_callback() { static void(FASTCALL * p_file_handler__call_free_callback)(p_file_handler * pfhndl) diff --git a/src/DivaGL/light_param/fog.hpp b/src/DivaGL/light_param/fog.hpp index f40d80a..182df93 100644 --- a/src/DivaGL/light_param/fog.hpp +++ b/src/DivaGL/light_param/fog.hpp @@ -6,9 +6,22 @@ #pragma once #include "../../KKdLib/default.hpp" -#include "../../KKdLib/light_param/fog.hpp" #include "../../KKdLib/vec.hpp" +enum fog_id { + FOG_DEPTH = 0x00, + FOG_HEIGHT = 0x01, + FOG_BUMP = 0x02, + FOG_MAX = 0x03, +}; + +enum fog_type { + FOG_NONE = 0x00, + FOG_LINEAR = 0x01, + FOG_EXP = 0x02, + FOG_EXP2 = 0x03, +}; + struct fog { fog_type type; float_t density; diff --git a/src/DivaGL/light_param/light.hpp b/src/DivaGL/light_param/light.hpp index 37dbfa8..8a9f6db 100644 --- a/src/DivaGL/light_param/light.hpp +++ b/src/DivaGL/light_param/light.hpp @@ -6,10 +6,49 @@ #pragma once #include "../../KKdLib/default.hpp" -#include "../../KKdLib/light_param/light.hpp" #include "../../KKdLib/mat.hpp" #include "../../KKdLib/vec.hpp" +enum light_id { + LIGHT_CHARA = 0x00, + LIGHT_STAGE = 0x01, + LIGHT_SUN = 0x02, + LIGHT_REFLECT = 0x03, + LIGHT_SHADOW = 0x04, + LIGHT_CHARA_COLOR = 0x05, + LIGHT_TONE_CURVE = 0x06, + LIGHT_PROJECTION = 0x07, + LIGHT_MAX = 0x08, +}; + +enum light_set_id { + LIGHT_SET_MAIN = 0x00, + LIGHT_SET_MAX = 0x01, +}; + +enum light_type { + LIGHT_OFF = 0x00, + LIGHT_PARALLEL = 0x01, + LIGHT_POINT = 0x02, + LIGHT_SPOT = 0x03, +}; + +struct light_attenuation { + float_t constant; + float_t linear; + float_t quadratic; +}; + +struct light_clip_plane { + bool data[4]; +}; + +struct light_tone_curve { + float_t start_point; + float_t end_point; + float_t coefficient; +}; + struct light_data { light_type type; vec4 ambient; diff --git a/src/DivaGL/object.cpp b/src/DivaGL/object.cpp index 3c6683f..beceb5a 100644 --- a/src/DivaGL/object.cpp +++ b/src/DivaGL/object.cpp @@ -31,6 +31,28 @@ static void obj_vertex_add_bone_weight(vec4& bone_weight, vec4i16& bone_index, i static void obj_vertex_validate_bone_data(vec4& bone_weight, vec4i16& bone_index); static uint32_t obj_vertex_format_get_vertex_size(obj_vertex_format format); +obj_material_shader_lighting_type obj_material_shader_attrib::get_lighting_type() const { + if (!m.is_lgt_diffuse && !m.is_lgt_specular) + return OBJ_MATERIAL_SHADER_LIGHTING_CONSTANT; + else if (!m.is_lgt_specular) + return OBJ_MATERIAL_SHADER_LIGHTING_LAMBERT; + else + return OBJ_MATERIAL_SHADER_LIGHTING_PHONG; +} + +int32_t obj_texture_attrib::get_blend() const { + switch (m.blend) { + case 4: + return 2; + case 6: + return 1; + case 16: + return 3; + default: + return 0; + } +} + void obj_mesh_vertex_buffer::cycle_index() { if (++index >= count) index = 0; diff --git a/src/DivaGL/render.hpp b/src/DivaGL/render.hpp index 8b9124b..e1a38b5 100644 --- a/src/DivaGL/render.hpp +++ b/src/DivaGL/render.hpp @@ -6,13 +6,19 @@ #pragma once #include "../KKdLib/default.hpp" -#include "../KKdLib/light_param/glow.hpp" #include "../KKdLib/vec.hpp" #include "renderer/dof.hpp" #include "renderer/transparency.hpp" #include "camera.hpp" #include "gl_uniform_buffer.hpp" +enum tone_map_method { + TONE_MAP_YCC_EXPONENT = 0, + TONE_MAP_RGB_LINEAR = 1, + TONE_MAP_RGB_LINEAR2 = 2, + TONE_MAP_MAX = 3, +}; + namespace rndr { struct Render { enum MagFilterType { diff --git a/src/DivaGL/resolution_mode.hpp b/src/DivaGL/resolution_mode.hpp index 850c5ea..e2f66af 100644 --- a/src/DivaGL/resolution_mode.hpp +++ b/src/DivaGL/resolution_mode.hpp @@ -6,9 +6,66 @@ #pragma once #include "../KKdLib/default.hpp" -#include "../KKdLib/spr.hpp" #include "../KKdLib/vec.hpp" +enum resolution_mode { + RESOLUTION_MODE_QVGA = 0x00, + RESOLUTION_MODE_VGA = 0x01, + RESOLUTION_MODE_SVGA = 0x02, + RESOLUTION_MODE_XGA = 0x03, + RESOLUTION_MODE_SXGA = 0x04, + RESOLUTION_MODE_SXGAPlus = 0x05, + RESOLUTION_MODE_UXGA = 0x06, + RESOLUTION_MODE_WVGA = 0x07, + RESOLUTION_MODE_WSVGA = 0x08, + RESOLUTION_MODE_WXGA = 0x09, + RESOLUTION_MODE_FWXGA = 0x0A, + RESOLUTION_MODE_WUXGA = 0x0B, + RESOLUTION_MODE_WQXGA = 0x0C, + RESOLUTION_MODE_HD = 0x0D, + RESOLUTION_MODE_FHD = 0x0E, + RESOLUTION_MODE_QHD = 0x0F, + RESOLUTION_MODE_WQVGA = 0x10, + RESOLUTION_MODE_qHD = 0x11, + RESOLUTION_MODE_MAX = 0x12, + + // MM+ + /*RESOLUTION_MODE_QVGA = 0x00, + RESOLUTION_MODE_VGA = 0x01, + RESOLUTION_MODE_SVGA = 0x02, + RESOLUTION_MODE_XGA = 0x03, + RESOLUTION_MODE_SXGA = 0x04, + RESOLUTION_MODE_SXGAPlus = 0x05, + RESOLUTION_MODE_UXGA = 0x06, + RESOLUTION_MODE_WVGA = 0x07, + RESOLUTION_MODE_WSVGA = 0x08, + RESOLUTION_MODE_WXGA = 0x09, + RESOLUTION_MODE_FWXGA = 0x0A, + RESOLUTION_MODE_WUXGA = 0x0B, + RESOLUTION_MODE_WQXGA = 0x0C, + RESOLUTION_MODE_HD = 0x0D, + RESOLUTION_MODE_FHD = 0x0E, + RESOLUTION_MODE_UHD = 0x0F, + RESOLUTION_MODE_3KatUHD = 0x10, + RESOLUTION_MODE_3K = 0x11, + RESOLUTION_MODE_QHD = 0x12, + RESOLUTION_MODE_WQVGA = 0x13, + RESOLUTION_MODE_qHD = 0x14, + RESOLUTION_MODE_XGAPlus = 0x15, + RESOLUTION_MODE_1176x664 = 0x16, + RESOLUTION_MODE_1200x960 = 0x17, + RESOLUTION_MODE_WXGA1280x900 = 0x18, + RESOLUTION_MODE_SXGAMinus = 0x19, + RESOLUTION_MODE_FWXGA1366x768 = 0x1A, + RESOLUTION_MODE_WXGAPlus = 0x1B, + RESOLUTION_MODE_HDPlus = 0x1C, + RESOLUTION_MODE_WSXGA = 0x1D, + RESOLUTION_MODE_WSXGAPlus = 0x1E, + RESOLUTION_MODE_1920x1440 = 0x1F, + RESOLUTION_MODE_QWXGA = 0x20, + RESOLUTION_MODE_MAX = 0x21,*/ +}; + struct resolution_table_struct { int32_t width_full; int32_t height_full; diff --git a/src/DivaGL/sprite.hpp b/src/DivaGL/sprite.hpp index 666e110..07101be 100644 --- a/src/DivaGL/sprite.hpp +++ b/src/DivaGL/sprite.hpp @@ -35,6 +35,26 @@ struct SpriteHeaderFile { uint32_t sprdata_offset; }; +namespace spr { + struct SprInfo { + uint32_t texid; + int32_t rotate; + float_t su; + float_t sv; + float_t eu; + float_t ev; + float_t px; + float_t py; + float_t width; + float_t height; + }; +}; + +struct SpriteData { + uint32_t attr; + resolution_mode resolution_mode; +}; + struct SpriteHeader { uint32_t flag; uint32_t texofs; diff --git a/src/DivaGL/texture.hpp b/src/DivaGL/texture.hpp index 23328a7..09cb009 100644 --- a/src/DivaGL/texture.hpp +++ b/src/DivaGL/texture.hpp @@ -6,7 +6,6 @@ #pragma once #include "../KKdLib/default.hpp" -#include "../KKdLib/image.hpp" #include "../KKdLib/txp.hpp" #include "wrap.hpp" diff --git a/src/KKdLib/KKdLib.rc b/src/KKdLib/KKdLib.rc new file mode 100644 index 0000000..de07c4d --- /dev/null +++ b/src/KKdLib/KKdLib.rc @@ -0,0 +1,46 @@ +#define IDR_VERSION2 101 + +#ifdef APSTUDIO_INVOKED +#ifndef APSTUDIO_READONLY_SYMBOLS +#define _APS_NEXT_RESOURCE_VALUE 102 +#define _APS_NEXT_COMMAND_VALUE 40001 +#define _APS_NEXT_CONTROL_VALUE 1000 +#define _APS_NEXT_SYMED_VALUE 101 +#endif +#endif + +#define APSTUDIO_READONLY_SYMBOLS +#include "winres.h" +#undef APSTUDIO_READONLY_SYMBOLS + +VS_VERSION_INFO VERSIONINFO + FILEVERSION 0,7,1,0 + PRODUCTVERSION 0,7,1,0 + FILEFLAGSMASK 0x3fL +#ifdef _DEBUG + FILEFLAGS 0x1L +#else + FILEFLAGS 0x0L +#endif + FILEOS 0x40004L + FILETYPE 0x0L + FILESUBTYPE 0x0L +BEGIN + BLOCK "StringFileInfo" + BEGIN + BLOCK "000904b0" + BEGIN + VALUE "FileDescription", "KKdLib" + VALUE "FileVersion", "0.7.1.0" + VALUE "InternalName", "KKdLib" + VALUE "LegalCopyright", "korenkonder (C) 2017-2023" + VALUE "OriginalFilename", "KKdLib" + VALUE "ProductName", "KKdLib" + VALUE "ProductVersion", "0.7.1.0" + END + END + BLOCK "VarFileInfo" + BEGIN + VALUE "Translation", 0x9, 1200 + END +END diff --git a/src/KKdLib/KKdLib.vcxproj b/src/KKdLib/KKdLib.vcxproj new file mode 100644 index 0000000..c22bf62 --- /dev/null +++ b/src/KKdLib/KKdLib.vcxproj @@ -0,0 +1,279 @@ + + + + + Debug + x64 + + + DebugOpt + x64 + + + ReleaseLTCG + x64 + + + Release + x64 + + + + 16.0 + {65707ad3-568a-4fe9-a75e-bdbf3d5efc88} + Win32Proj + 10.0 + + + + StaticLibrary + true + v142 + Unicode + + + StaticLibrary + false + v142 + Unicode + + + StaticLibrary + false + v142 + Unicode + + + StaticLibrary + false + v142 + Unicode + true + + + + + + + + + + + + + + + + + true + KKdLib + $(SolutionDir)lib\debug\ + $(SolutionDir)obj\debug\KKdLib\ + + + false + KKdLib + $(SolutionDir)lib\debugopt\ + $(SolutionDir)obj\debugopt\KKdLib\ + + + false + KKdLib + $(SolutionDir)lib\release\ + $(SolutionDir)obj\release\KKdLib\ + + + false + KKdLib + $(SolutionDir)lib\releaseltcg\ + $(SolutionDir)obj\releaseltcg\KKdLib\ + + + + NotUsing + Level3 + Disabled + WIN32;DEBUG;%(PreprocessorDefinitions) + true + NotSet + EditAndContinue + false + OnlyExplicitInline + false + FastCall + $(SolutionDir)extern\src + true + CompileAsCpp + EnableFastChecks + MultiThreadedDebug + $(IntDir)/%(RelativeDir) + 26812 + + + Windows + true + %(AdditionalDependencies) + + + + + NotUsing + Level3 + Full + WIN32;DEBUG;%(PreprocessorDefinitions) + true + NotSet + ProgramDatabase + false + Speed + OnlyExplicitInline + FastCall + true + $(SolutionDir)extern\src + CompileAsCpp + MultiThreaded + $(IntDir)/%(RelativeDir) + false + 26812 + + + Windows + true + %(AdditionalDependencies) + + + false + + + + + NotUsing + Level3 + MaxSpeed + WIN32;%(PreprocessorDefinitions) + true + NotSet + ProgramDatabase + false + Speed + OnlyExplicitInline + FastCall + true + $(SolutionDir)extern\src + CompileAsCpp + MultiThreaded + $(IntDir)/%(RelativeDir) + false + 26812 + + + Windows + true + %(AdditionalDependencies) + + + true + + + + + NotUsing + Level3 + MaxSpeed + WIN32;%(PreprocessorDefinitions) + true + NotSet + ProgramDatabase + false + Speed + OnlyExplicitInline + FastCall + true + $(SolutionDir)extern\src + CompileAsCpp + true + /Gw %(AdditionalOptions) + MultiThreaded + $(IntDir)/%(RelativeDir) + false + 26812 + + + Windows + true + %(AdditionalDependencies) + + + true + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + \ No newline at end of file diff --git a/src/KKdLib/KKdLib.vcxproj.user b/src/KKdLib/KKdLib.vcxproj.user new file mode 100644 index 0000000..4758d5a --- /dev/null +++ b/src/KKdLib/KKdLib.vcxproj.user @@ -0,0 +1,22 @@ + + + + true + + + NativeOnly + WindowsLocalDebugger + + + NativeOnly + WindowsLocalDebugger + + + NativeOnly + WindowsLocalDebugger + + + NativeOnly + WindowsLocalDebugger + + \ No newline at end of file diff --git a/src/KKdLib/aes.cpp b/src/KKdLib/aes.cpp new file mode 100644 index 0000000..ceec68e --- /dev/null +++ b/src/KKdLib/aes.cpp @@ -0,0 +1,1346 @@ +/* + Original: https://github.com/kokke/tiny-AES-c +*/ + +/* + +This is an implementation of the AES algorithm, specifically ECB, CTR and CBC mode. +Block size can be chosen in aes.h - available choices are AES128, AES192, AES256. + +The implementation is verified against the test vectors in: + National Institute of Standards and Technology Special Publication 800-38A 2001 ED + +ECB-AES128 +---------- + + plain-text: + 6bc1bee22e409f96e93d7e117393172a + ae2d8a571e03ac9c9eb76fac45af8e51 + 30c81c46a35ce411e5fbc1191a0a52ef + f69f2445df4f9b17ad2b417be66c3710 + + key: + 2b7e151628aed2a6abf7158809cf4f3c + + resulting cipher + 3ad77bb40d7a3660a89ecaf32466ef97 + f5d3d58503b9699de785895a96fdbaaf + 43b1cd7f598ece23881b00e3ed030688 + 7b0c785e27e8ad3f8223207104725dd4 + + +NOTE: String length must be evenly divisible by 16byte (str_len % 16 == 0) + You should pad the end of the string with zeros if this is not the case. + For AES192/256 the key size is proportionally larger. + +*/ + +/*****************************************************************************/ +/* Includes: */ +/*****************************************************************************/ +#include "aes.hpp" + +/*****************************************************************************/ +/* Defines: */ +/*****************************************************************************/ +// The number of columns comprising a state in AES. This is a constant in AES. Value=4 +#define Nb 4 + +#define Nk128 4 // The number of 32 bit words in a key. +#define Nr128 10 // The number of rounds in AES Cipher. +#define Nk192 6 +#define Nr192 12 +#define Nk256 8 +#define Nr256 14 + +// jcallan@github points out that declaring Multiply as a function +// reduces code size considerably with the Keil ARM compiler. +// See this link for more information: https://github.com/kokke/tiny-AES-C/pull/3 +#ifndef MULTIPLY_AS_A_FUNCTION + #define MULTIPLY_AS_A_FUNCTION 0 +#endif + +/*****************************************************************************/ +/* Private variables: */ +/*****************************************************************************/ +// state - array holding the intermediate results during decryption. +typedef uint8_t state_t[4][4]; + +// The lookup-tables are marked const so they can be placed in read-only storage instead of RAM +// The numbers below can be computed dynamically trading ROM for RAM - +// This can be useful in (embedded) bootloader applications, where ROM is often limited. +static const uint8_t sbox[256] = { + //0 1 2 3 4 5 6 7 8 9 A B C D E F + 0x63, 0x7c, 0x77, 0x7b, 0xf2, 0x6b, 0x6f, 0xc5, 0x30, 0x01, 0x67, 0x2b, 0xfe, 0xd7, 0xab, 0x76, + 0xca, 0x82, 0xc9, 0x7d, 0xfa, 0x59, 0x47, 0xf0, 0xad, 0xd4, 0xa2, 0xaf, 0x9c, 0xa4, 0x72, 0xc0, + 0xb7, 0xfd, 0x93, 0x26, 0x36, 0x3f, 0xf7, 0xcc, 0x34, 0xa5, 0xe5, 0xf1, 0x71, 0xd8, 0x31, 0x15, + 0x04, 0xc7, 0x23, 0xc3, 0x18, 0x96, 0x05, 0x9a, 0x07, 0x12, 0x80, 0xe2, 0xeb, 0x27, 0xb2, 0x75, + 0x09, 0x83, 0x2c, 0x1a, 0x1b, 0x6e, 0x5a, 0xa0, 0x52, 0x3b, 0xd6, 0xb3, 0x29, 0xe3, 0x2f, 0x84, + 0x53, 0xd1, 0x00, 0xed, 0x20, 0xfc, 0xb1, 0x5b, 0x6a, 0xcb, 0xbe, 0x39, 0x4a, 0x4c, 0x58, 0xcf, + 0xd0, 0xef, 0xaa, 0xfb, 0x43, 0x4d, 0x33, 0x85, 0x45, 0xf9, 0x02, 0x7f, 0x50, 0x3c, 0x9f, 0xa8, + 0x51, 0xa3, 0x40, 0x8f, 0x92, 0x9d, 0x38, 0xf5, 0xbc, 0xb6, 0xda, 0x21, 0x10, 0xff, 0xf3, 0xd2, + 0xcd, 0x0c, 0x13, 0xec, 0x5f, 0x97, 0x44, 0x17, 0xc4, 0xa7, 0x7e, 0x3d, 0x64, 0x5d, 0x19, 0x73, + 0x60, 0x81, 0x4f, 0xdc, 0x22, 0x2a, 0x90, 0x88, 0x46, 0xee, 0xb8, 0x14, 0xde, 0x5e, 0x0b, 0xdb, + 0xe0, 0x32, 0x3a, 0x0a, 0x49, 0x06, 0x24, 0x5c, 0xc2, 0xd3, 0xac, 0x62, 0x91, 0x95, 0xe4, 0x79, + 0xe7, 0xc8, 0x37, 0x6d, 0x8d, 0xd5, 0x4e, 0xa9, 0x6c, 0x56, 0xf4, 0xea, 0x65, 0x7a, 0xae, 0x08, + 0xba, 0x78, 0x25, 0x2e, 0x1c, 0xa6, 0xb4, 0xc6, 0xe8, 0xdd, 0x74, 0x1f, 0x4b, 0xbd, 0x8b, 0x8a, + 0x70, 0x3e, 0xb5, 0x66, 0x48, 0x03, 0xf6, 0x0e, 0x61, 0x35, 0x57, 0xb9, 0x86, 0xc1, 0x1d, 0x9e, + 0xe1, 0xf8, 0x98, 0x11, 0x69, 0xd9, 0x8e, 0x94, 0x9b, 0x1e, 0x87, 0xe9, 0xce, 0x55, 0x28, 0xdf, + 0x8c, 0xa1, 0x89, 0x0d, 0xbf, 0xe6, 0x42, 0x68, 0x41, 0x99, 0x2d, 0x0f, 0xb0, 0x54, 0xbb, 0x16 +}; + +static const uint8_t rsbox[256] = { + 0x52, 0x09, 0x6a, 0xd5, 0x30, 0x36, 0xa5, 0x38, 0xbf, 0x40, 0xa3, 0x9e, 0x81, 0xf3, 0xd7, 0xfb, + 0x7c, 0xe3, 0x39, 0x82, 0x9b, 0x2f, 0xff, 0x87, 0x34, 0x8e, 0x43, 0x44, 0xc4, 0xde, 0xe9, 0xcb, + 0x54, 0x7b, 0x94, 0x32, 0xa6, 0xc2, 0x23, 0x3d, 0xee, 0x4c, 0x95, 0x0b, 0x42, 0xfa, 0xc3, 0x4e, + 0x08, 0x2e, 0xa1, 0x66, 0x28, 0xd9, 0x24, 0xb2, 0x76, 0x5b, 0xa2, 0x49, 0x6d, 0x8b, 0xd1, 0x25, + 0x72, 0xf8, 0xf6, 0x64, 0x86, 0x68, 0x98, 0x16, 0xd4, 0xa4, 0x5c, 0xcc, 0x5d, 0x65, 0xb6, 0x92, + 0x6c, 0x70, 0x48, 0x50, 0xfd, 0xed, 0xb9, 0xda, 0x5e, 0x15, 0x46, 0x57, 0xa7, 0x8d, 0x9d, 0x84, + 0x90, 0xd8, 0xab, 0x00, 0x8c, 0xbc, 0xd3, 0x0a, 0xf7, 0xe4, 0x58, 0x05, 0xb8, 0xb3, 0x45, 0x06, + 0xd0, 0x2c, 0x1e, 0x8f, 0xca, 0x3f, 0x0f, 0x02, 0xc1, 0xaf, 0xbd, 0x03, 0x01, 0x13, 0x8a, 0x6b, + 0x3a, 0x91, 0x11, 0x41, 0x4f, 0x67, 0xdc, 0xea, 0x97, 0xf2, 0xcf, 0xce, 0xf0, 0xb4, 0xe6, 0x73, + 0x96, 0xac, 0x74, 0x22, 0xe7, 0xad, 0x35, 0x85, 0xe2, 0xf9, 0x37, 0xe8, 0x1c, 0x75, 0xdf, 0x6e, + 0x47, 0xf1, 0x1a, 0x71, 0x1d, 0x29, 0xc5, 0x89, 0x6f, 0xb7, 0x62, 0x0e, 0xaa, 0x18, 0xbe, 0x1b, + 0xfc, 0x56, 0x3e, 0x4b, 0xc6, 0xd2, 0x79, 0x20, 0x9a, 0xdb, 0xc0, 0xfe, 0x78, 0xcd, 0x5a, 0xf4, + 0x1f, 0xdd, 0xa8, 0x33, 0x88, 0x07, 0xc7, 0x31, 0xb1, 0x12, 0x10, 0x59, 0x27, 0x80, 0xec, 0x5f, + 0x60, 0x51, 0x7f, 0xa9, 0x19, 0xb5, 0x4a, 0x0d, 0x2d, 0xe5, 0x7a, 0x9f, 0x93, 0xc9, 0x9c, 0xef, + 0xa0, 0xe0, 0x3b, 0x4d, 0xae, 0x2a, 0xf5, 0xb0, 0xc8, 0xeb, 0xbb, 0x3c, 0x83, 0x53, 0x99, 0x61, + 0x17, 0x2b, 0x04, 0x7e, 0xba, 0x77, 0xd6, 0x26, 0xe1, 0x69, 0x14, 0x63, 0x55, 0x21, 0x0c, 0x7d +}; + +// The round constant word array, Rcon[i], contains the values given by +// x to the power (i-1) being powers of x (x is denoted as {02}) in the field GF(2^8) +static const uint8_t Rcon[11] = { + 0x8d, 0x01, 0x02, 0x04, 0x08, 0x10, 0x20, 0x40, 0x80, 0x1b, 0x36 +}; + +/* + * Jordan Goulder points out in PR #12 (https://github.com/kokke/tiny-AES-C/pull/12), + * that you can remove most of the elements in the Rcon array, because they are unused. + * + * From Wikipedia's article on the Rijndael key schedule @ https://en.wikipedia.org/wiki/Rijndael_key_schedule#Rcon + * + * "Only the first some of these constants are actually used - up to rcon[10] for AES-128 (as 11 round keys are needed), + * up to rcon[8] for AES-192, up to rcon[7] for AES-256. rcon[0] is not used in AES algorithm." + */ + +extern bool aes_ni; + +// This function produces Nb(Nr+1) round keys. The round keys are used in each round to decrypt the states. +static void key_expansion_aes128(uint8_t* RoundKey, const uint8_t* Key) { + unsigned i, j, k; + uint8_t tempa[4]; // Used for the column/row operations + + // The first round key is the key itself. + for (i = 0; i < Nk128; ++i) { + RoundKey[(i * 4) + 0] = Key[(i * 4) + 0]; + RoundKey[(i * 4) + 1] = Key[(i * 4) + 1]; + RoundKey[(i * 4) + 2] = Key[(i * 4) + 2]; + RoundKey[(i * 4) + 3] = Key[(i * 4) + 3]; + } + + // All other round keys are found from the previous round keys. + for (i = Nk128; i < Nb * (Nr128 + 1); ++i) { + { + k = (i - 1) * 4; + tempa[0] = RoundKey[k + 0]; + tempa[1] = RoundKey[k + 1]; + tempa[2] = RoundKey[k + 2]; + tempa[3] = RoundKey[k + 3]; + } + + if (i % Nk128 == 0) { + // This function shifts the 4 bytes in a word to the left once. + // [a0,a1,a2,a3] becomes [a1,a2,a3,a0] + + // Function RotWord() + { + const uint8_t u8tmp = tempa[0]; + tempa[0] = tempa[1]; + tempa[1] = tempa[2]; + tempa[2] = tempa[3]; + tempa[3] = u8tmp; + } + + // SubWord() is a function that takes a four-byte input word and + // applies the S-box to each of the four bytes to produce an output word. + + // Function Subword() + { + tempa[0] = sbox[tempa[0]]; + tempa[1] = sbox[tempa[1]]; + tempa[2] = sbox[tempa[2]]; + tempa[3] = sbox[tempa[3]]; + } + + tempa[0] = tempa[0] ^ Rcon[i / Nk128]; + } + j = i * 4; k = (i - Nk128) * 4; + RoundKey[j + 0] = RoundKey[k + 0] ^ tempa[0]; + RoundKey[j + 1] = RoundKey[k + 1] ^ tempa[1]; + RoundKey[j + 2] = RoundKey[k + 2] ^ tempa[2]; + RoundKey[j + 3] = RoundKey[k + 3] ^ tempa[3]; + } +} + +static void key_expansion_aes192(uint8_t* RoundKey, const uint8_t* Key) { + unsigned i, j, k; + uint8_t tempa[4]; // Used for the column/row operations + + // The first round key is the key itself. + for (i = 0; i < Nk192; ++i) { + RoundKey[(i * 4) + 0] = Key[(i * 4) + 0]; + RoundKey[(i * 4) + 1] = Key[(i * 4) + 1]; + RoundKey[(i * 4) + 2] = Key[(i * 4) + 2]; + RoundKey[(i * 4) + 3] = Key[(i * 4) + 3]; + } + + // All other round keys are found from the previous round keys. + for (i = Nk192; i < Nb * (Nr192 + 1); ++i) { + { + k = (i - 1) * 4; + tempa[0] = RoundKey[k + 0]; + tempa[1] = RoundKey[k + 1]; + tempa[2] = RoundKey[k + 2]; + tempa[3] = RoundKey[k + 3]; + } + + if (i % Nk192 == 0) { + // This function shifts the 4 bytes in a word to the left once. + // [a0,a1,a2,a3] becomes [a1,a2,a3,a0] + + // Function RotWord() + { + const uint8_t u8tmp = tempa[0]; + tempa[0] = tempa[1]; + tempa[1] = tempa[2]; + tempa[2] = tempa[3]; + tempa[3] = u8tmp; + } + + // SubWord() is a function that takes a four-byte input word and + // applies the S-box to each of the four bytes to produce an output word. + + // Function Subword() + { + tempa[0] = sbox[tempa[0]]; + tempa[1] = sbox[tempa[1]]; + tempa[2] = sbox[tempa[2]]; + tempa[3] = sbox[tempa[3]]; + } + + tempa[0] = tempa[0] ^ Rcon[i / Nk192]; + } + j = i * 4; k = (i - Nk192) * 4; + RoundKey[j + 0] = RoundKey[k + 0] ^ tempa[0]; + RoundKey[j + 1] = RoundKey[k + 1] ^ tempa[1]; + RoundKey[j + 2] = RoundKey[k + 2] ^ tempa[2]; + RoundKey[j + 3] = RoundKey[k + 3] ^ tempa[3]; + } +} + +static void key_expansion_aes256(uint8_t* RoundKey, const uint8_t* Key) { + unsigned i, j, k; + uint8_t tempa[4]; // Used for the column/row operations + + // The first round key is the key itself. + for (i = 0; i < Nk256; ++i) { + RoundKey[(i * 4) + 0] = Key[(i * 4) + 0]; + RoundKey[(i * 4) + 1] = Key[(i * 4) + 1]; + RoundKey[(i * 4) + 2] = Key[(i * 4) + 2]; + RoundKey[(i * 4) + 3] = Key[(i * 4) + 3]; + } + + // All other round keys are found from the previous round keys. + for (i = Nk256; i < Nb * (Nr256 + 1); ++i) { + { + k = (i - 1) * 4; + tempa[0] = RoundKey[k + 0]; + tempa[1] = RoundKey[k + 1]; + tempa[2] = RoundKey[k + 2]; + tempa[3] = RoundKey[k + 3]; + } + + if (i % Nk256 == 0) { + // This function shifts the 4 bytes in a word to the left once. + // [a0,a1,a2,a3] becomes [a1,a2,a3,a0] + + // Function RotWord() + { + const uint8_t u8tmp = tempa[0]; + tempa[0] = tempa[1]; + tempa[1] = tempa[2]; + tempa[2] = tempa[3]; + tempa[3] = u8tmp; + } + + // SubWord() is a function that takes a four-byte input word and + // applies the S-box to each of the four bytes to produce an output word. + + // Function Subword() + { + tempa[0] = sbox[tempa[0]]; + tempa[1] = sbox[tempa[1]]; + tempa[2] = sbox[tempa[2]]; + tempa[3] = sbox[tempa[3]]; + } + + tempa[0] = tempa[0] ^ Rcon[i / Nk256]; + } + if (i % Nk256 == 4) { + // Function Subword() + { + tempa[0] = sbox[tempa[0]]; + tempa[1] = sbox[tempa[1]]; + tempa[2] = sbox[tempa[2]]; + tempa[3] = sbox[tempa[3]]; + } + } + j = i * 4; k = (i - Nk256) * 4; + RoundKey[j + 0] = RoundKey[k + 0] ^ tempa[0]; + RoundKey[j + 1] = RoundKey[k + 1] ^ tempa[1]; + RoundKey[j + 2] = RoundKey[k + 2] ^ tempa[2]; + RoundKey[j + 3] = RoundKey[k + 3] ^ tempa[3]; + } +} + +inline static __m128i key_expansion_aes128_ni_assist(__m128i temp1, __m128i temp2) { + __m128i temp3; + temp2 = _mm_shuffle_epi32(temp2, 0xFF); + temp3 = _mm_slli_si128(temp1, 0x04); + temp1 = _mm_xor_si128(temp1, temp3); + temp3 = _mm_slli_si128(temp3, 0x04); + temp1 = _mm_xor_si128(temp1, temp3); + temp3 = _mm_slli_si128(temp3, 0x04); + temp1 = _mm_xor_si128(temp1, temp3); + temp1 = _mm_xor_si128(temp1, temp2); + return temp1; +} + +static void key_expansion_aes128_ni(__m128i* RoundKey, const uint8_t* Key) { + __m128i temp1, temp2; + temp1 = _mm_loadu_si128((__m128i*)&Key[0]); + RoundKey[0] = temp1; + temp2 = _mm_aeskeygenassist_si128(temp1, 0x01); + temp1 = key_expansion_aes128_ni_assist(temp1, temp2); + RoundKey[1] = temp1; + temp2 = _mm_aeskeygenassist_si128(temp1, 0x02); + temp1 = key_expansion_aes128_ni_assist(temp1, temp2); + RoundKey[2] = temp1; + temp2 = _mm_aeskeygenassist_si128(temp1, 0x04); + temp1 = key_expansion_aes128_ni_assist(temp1, temp2); + RoundKey[3] = temp1; + temp2 = _mm_aeskeygenassist_si128(temp1, 0x08); + temp1 = key_expansion_aes128_ni_assist(temp1, temp2); + RoundKey[4] = temp1; + temp2 = _mm_aeskeygenassist_si128(temp1, 0x10); + temp1 = key_expansion_aes128_ni_assist(temp1, temp2); + RoundKey[5] = temp1; + temp2 = _mm_aeskeygenassist_si128(temp1, 0x20); + temp1 = key_expansion_aes128_ni_assist(temp1, temp2); + RoundKey[6] = temp1; + temp2 = _mm_aeskeygenassist_si128(temp1, 0x40); + temp1 = key_expansion_aes128_ni_assist(temp1, temp2); + RoundKey[7] = temp1; + temp2 = _mm_aeskeygenassist_si128(temp1, 0x80); + temp1 = key_expansion_aes128_ni_assist(temp1, temp2); + RoundKey[8] = temp1; + temp2 = _mm_aeskeygenassist_si128(temp1, 0x1B); + temp1 = key_expansion_aes128_ni_assist(temp1, temp2); + RoundKey[9] = temp1; + temp2 = _mm_aeskeygenassist_si128(temp1, 0x36); + temp1 = key_expansion_aes128_ni_assist(temp1, temp2); + RoundKey[10] = temp1; + RoundKey[11] = _mm_aesimc_si128(RoundKey[9]); + RoundKey[12] = _mm_aesimc_si128(RoundKey[8]); + RoundKey[13] = _mm_aesimc_si128(RoundKey[7]); + RoundKey[14] = _mm_aesimc_si128(RoundKey[6]); + RoundKey[15] = _mm_aesimc_si128(RoundKey[5]); + RoundKey[16] = _mm_aesimc_si128(RoundKey[4]); + RoundKey[17] = _mm_aesimc_si128(RoundKey[3]); + RoundKey[18] = _mm_aesimc_si128(RoundKey[2]); + RoundKey[19] = _mm_aesimc_si128(RoundKey[1]); +} + +inline static void key_expansion_aes192_ni_assist(__m128i* temp1, __m128i* temp2, __m128i* temp3) { + __m128i temp4; + *temp2 = _mm_shuffle_epi32(*temp2, 0x55); + temp4 = _mm_slli_si128(*temp1, 0x04); + *temp1 = _mm_xor_si128(*temp1, temp4); + temp4 = _mm_slli_si128(temp4, 0x04); + *temp1 = _mm_xor_si128(*temp1, temp4); + temp4 = _mm_slli_si128(temp4, 0x04); + *temp1 = _mm_xor_si128(*temp1, temp4); + *temp1 = _mm_xor_si128(*temp1, *temp2); + *temp2 = _mm_shuffle_epi32(*temp1, 0xFF); + temp4 = _mm_slli_si128(*temp3, 0x04); + *temp3 = _mm_xor_si128(*temp3, temp4); + *temp3 = _mm_xor_si128(*temp3, *temp2); +} + +static void key_expansion_aes192_ni(__m128i* RoundKey, const uint8_t* Key) { + __m128i temp1, temp2, temp3; + temp1 = _mm_loadu_si128((__m128i*)&Key[0]); + temp3 = _mm_loadu_si128((__m128i*)&Key[16]); + RoundKey[0] = temp1; + RoundKey[1] = temp3; + temp2 = _mm_aeskeygenassist_si128(temp3, 0x01); + key_expansion_aes192_ni_assist(&temp1, &temp2, &temp3); + *(__m128d*)& RoundKey[1] = _mm_shuffle_pd(*(__m128d*)&RoundKey[1], *(__m128d*)&temp1, 0); + *(__m128d*)& RoundKey[2] = _mm_shuffle_pd(*(__m128d*)&temp1, *(__m128d*)&temp3, 1); + temp2 = _mm_aeskeygenassist_si128(temp3, 0x02); + key_expansion_aes192_ni_assist(&temp1, &temp2, &temp3); + RoundKey[3] = temp1; + RoundKey[4] = temp3; + temp2 = _mm_aeskeygenassist_si128(temp3, 0x04); + key_expansion_aes192_ni_assist(&temp1, &temp2, &temp3); + *(__m128d*)& RoundKey[4] = _mm_shuffle_pd(*(__m128d*)&RoundKey[4], *(__m128d*)&temp1, 0); + *(__m128d*)& RoundKey[5] = _mm_shuffle_pd(*(__m128d*)&temp1, *(__m128d*)&temp3, 1); + temp2 = _mm_aeskeygenassist_si128(temp3, 0x08); + key_expansion_aes192_ni_assist(&temp1, &temp2, &temp3); + RoundKey[6] = temp1; + RoundKey[7] = temp3; + temp2 = _mm_aeskeygenassist_si128(temp3, 0x10); + key_expansion_aes192_ni_assist(&temp1, &temp2, &temp3); + *(__m128d*)& RoundKey[7] = _mm_shuffle_pd(*(__m128d*)&RoundKey[7], *(__m128d*)&temp1, 0); + *(__m128d*)& RoundKey[8] = _mm_shuffle_pd(*(__m128d*)&temp1, *(__m128d*)&temp3, 1); + temp2 = _mm_aeskeygenassist_si128(temp3, 0x20); + key_expansion_aes192_ni_assist(&temp1, &temp2, &temp3); + RoundKey[9] = temp1; + RoundKey[10] = temp3; + temp2 = _mm_aeskeygenassist_si128(temp3, 0x40); + key_expansion_aes192_ni_assist(&temp1, &temp2, &temp3); + *(__m128d*)& RoundKey[10] = _mm_shuffle_pd(*(__m128d*)&RoundKey[10], *(__m128d*)&temp1, 0); + *(__m128d*)& RoundKey[11] = _mm_shuffle_pd(*(__m128d*)&temp1, *(__m128d*)&temp3, 1); + temp2 = _mm_aeskeygenassist_si128(temp3, 0x80); + key_expansion_aes192_ni_assist(&temp1, &temp2, &temp3); + RoundKey[12] = temp1; + RoundKey[13] = _mm_aesimc_si128(RoundKey[11]); + RoundKey[14] = _mm_aesimc_si128(RoundKey[10]); + RoundKey[15] = _mm_aesimc_si128(RoundKey[9]); + RoundKey[16] = _mm_aesimc_si128(RoundKey[8]); + RoundKey[17] = _mm_aesimc_si128(RoundKey[7]); + RoundKey[18] = _mm_aesimc_si128(RoundKey[6]); + RoundKey[19] = _mm_aesimc_si128(RoundKey[5]); + RoundKey[20] = _mm_aesimc_si128(RoundKey[4]); + RoundKey[21] = _mm_aesimc_si128(RoundKey[3]); + RoundKey[22] = _mm_aesimc_si128(RoundKey[2]); + RoundKey[23] = _mm_aesimc_si128(RoundKey[1]); +} + +inline static void key_expansion_aes256_ni_assist_1(__m128i* temp1, __m128i* temp2) { + __m128i temp4; + *temp2 = _mm_shuffle_epi32(*temp2, 0xFF); + temp4 = _mm_slli_si128(*temp1, 0x04); + *temp1 = _mm_xor_si128(*temp1, temp4); + temp4 = _mm_slli_si128(temp4, 0x04); + *temp1 = _mm_xor_si128(*temp1, temp4); + temp4 = _mm_slli_si128(temp4, 0x04); + *temp1 = _mm_xor_si128(*temp1, temp4); + *temp1 = _mm_xor_si128(*temp1, *temp2); +} + +inline static void key_expansion_aes256_ni_assist_2(__m128i* temp1, __m128i* temp3) { + __m128i temp2, temp4; + temp4 = _mm_aeskeygenassist_si128(*temp1, 0x00); + temp2 = _mm_shuffle_epi32(temp4, 0xAA); + temp4 = _mm_slli_si128(*temp3, 0x04); + *temp3 = _mm_xor_si128(*temp3, temp4); + temp4 = _mm_slli_si128(temp4, 0x04); + *temp3 = _mm_xor_si128(*temp3, temp4); + temp4 = _mm_slli_si128(temp4, 0x04); + *temp3 = _mm_xor_si128(*temp3, temp4); + *temp3 = _mm_xor_si128(*temp3, temp2); +} + +static void key_expansion_aes256_ni(__m128i* RoundKey, const uint8_t* Key) { + __m128i temp1, temp2, temp3; + temp1 = _mm_loadu_si128((__m128i*)&Key[0]); + temp3 = _mm_loadu_si128((__m128i*)&Key[16]); + RoundKey[0] = temp1; + RoundKey[1] = temp3; + temp2 = _mm_aeskeygenassist_si128(temp3, 0x01); + key_expansion_aes256_ni_assist_1(&temp1, &temp2); + RoundKey[2] = temp1; + key_expansion_aes256_ni_assist_2(&temp1, &temp3); + RoundKey[3] = temp3; + temp2 = _mm_aeskeygenassist_si128(temp3, 0x02); + key_expansion_aes256_ni_assist_1(&temp1, &temp2); + RoundKey[4] = temp1; + key_expansion_aes256_ni_assist_2(&temp1, &temp3); + RoundKey[5] = temp3; + temp2 = _mm_aeskeygenassist_si128(temp3, 0x04); + key_expansion_aes256_ni_assist_1(&temp1, &temp2); + RoundKey[6] = temp1; + key_expansion_aes256_ni_assist_2(&temp1, &temp3); + RoundKey[7] = temp3; + temp2 = _mm_aeskeygenassist_si128(temp3, 0x08); + key_expansion_aes256_ni_assist_1(&temp1, &temp2); + RoundKey[8] = temp1; + key_expansion_aes256_ni_assist_2(&temp1, &temp3); + RoundKey[9] = temp3; + temp2 = _mm_aeskeygenassist_si128(temp3, 0x10); + key_expansion_aes256_ni_assist_1(&temp1, &temp2); + RoundKey[10] = temp1; + key_expansion_aes256_ni_assist_2(&temp1, &temp3); + RoundKey[11] = temp3; + temp2 = _mm_aeskeygenassist_si128(temp3, 0x20); + key_expansion_aes256_ni_assist_1(&temp1, &temp2); + RoundKey[12] = temp1; + key_expansion_aes256_ni_assist_2(&temp1, &temp3); + RoundKey[13] = temp3; + temp2 = _mm_aeskeygenassist_si128(temp3, 0x40); + key_expansion_aes256_ni_assist_1(&temp1, &temp2); + RoundKey[14] = temp1; + RoundKey[15] = _mm_aesimc_si128(RoundKey[13]); + RoundKey[16] = _mm_aesimc_si128(RoundKey[12]); + RoundKey[17] = _mm_aesimc_si128(RoundKey[11]); + RoundKey[18] = _mm_aesimc_si128(RoundKey[10]); + RoundKey[19] = _mm_aesimc_si128(RoundKey[9]); + RoundKey[20] = _mm_aesimc_si128(RoundKey[8]); + RoundKey[21] = _mm_aesimc_si128(RoundKey[7]); + RoundKey[22] = _mm_aesimc_si128(RoundKey[6]); + RoundKey[23] = _mm_aesimc_si128(RoundKey[5]); + RoundKey[24] = _mm_aesimc_si128(RoundKey[4]); + RoundKey[25] = _mm_aesimc_si128(RoundKey[3]); + RoundKey[26] = _mm_aesimc_si128(RoundKey[2]); + RoundKey[27] = _mm_aesimc_si128(RoundKey[1]); +} + +void aes128_init_ctx(aes128_ctx* ctx, const uint8_t* key) { + if (aes_ni) + key_expansion_aes128_ni(ctx->RoundKeyNI, key); + else + key_expansion_aes128(ctx->RoundKey, key); +} + +void aes192_init_ctx(aes192_ctx* ctx, const uint8_t* key) { + if (aes_ni) + key_expansion_aes192_ni(ctx->RoundKeyNI, key); + else + key_expansion_aes192(ctx->RoundKey, key); +} + +void aes256_init_ctx(aes256_ctx* ctx, const uint8_t* key) { + if (aes_ni) + key_expansion_aes256_ni(ctx->RoundKeyNI, key); + else + key_expansion_aes256(ctx->RoundKey, key); +} + +void aes128_init_ctx_iv(aes128_ctx* ctx, const uint8_t* key, const uint8_t* iv) { + if (aes_ni) + key_expansion_aes128_ni(ctx->RoundKeyNI, key); + else + key_expansion_aes128(ctx->RoundKey, key); + memcpy(ctx->Iv, iv, AES_BLOCKLEN); +} + +void aes192_init_ctx_iv(aes192_ctx* ctx, const uint8_t* key, const uint8_t* iv) { + if (aes_ni) + key_expansion_aes192_ni(ctx->RoundKeyNI, key); + else + key_expansion_aes192(ctx->RoundKey, key); + memcpy(ctx->Iv, iv, AES_BLOCKLEN); +} + +void aes256_init_ctx_iv(aes256_ctx* ctx, const uint8_t* key, const uint8_t* iv) { + if (aes_ni) + key_expansion_aes256_ni(ctx->RoundKeyNI, key); + else + key_expansion_aes256(ctx->RoundKey, key); + memcpy(ctx->Iv, iv, AES_BLOCKLEN); +} + +void aes128_ctx_set_iv(aes128_ctx* ctx, const uint8_t* iv) { + memcpy(ctx->Iv, iv, AES_BLOCKLEN); +} + +void aes192_ctx_set_iv(aes192_ctx* ctx, const uint8_t* iv) { + memcpy(ctx->Iv, iv, AES_BLOCKLEN); +} + +void aes256_ctx_set_iv(aes256_ctx* ctx, const uint8_t* iv) { + memcpy(ctx->Iv, iv, AES_BLOCKLEN); +} + +// This function adds the round key to state. +// The round key is added to the state by an XOR function. +inline static void add_round_key(uint8_t round, state_t* state, const uint8_t* RoundKey) { + uint8_t i, j; + for (i = 0; i < 4; ++i) + for (j = 0; j < 4; ++j) + (*state)[i][j] ^= RoundKey[(round * Nb * 4) + (i * Nb) + j]; +} + +// The SubBytes Function Substitutes the values in the +// state matrix with values in an S-box. +inline static void sub_bytes(state_t* state) { + uint8_t i, j; + for (i = 0; i < 4; ++i) + for (j = 0; j < 4; ++j) + (*state)[j][i] = sbox[(*state)[j][i]]; +} + +// The ShiftRows() function shifts the rows in the state to the left. +// Each row is shifted with different offset. +// Offset = Row number. So the first row is not shifted. +inline static void shift_rows(state_t* state) { + uint8_t temp; + + // Rotate first row 1 columns to left + temp = (*state)[0][1]; + (*state)[0][1] = (*state)[1][1]; + (*state)[1][1] = (*state)[2][1]; + (*state)[2][1] = (*state)[3][1]; + (*state)[3][1] = temp; + + // Rotate second row 2 columns to left + temp = (*state)[0][2]; + (*state)[0][2] = (*state)[2][2]; + (*state)[2][2] = temp; + + temp = (*state)[1][2]; + (*state)[1][2] = (*state)[3][2]; + (*state)[3][2] = temp; + + // Rotate third row 3 columns to left + temp = (*state)[0][3]; + (*state)[0][3] = (*state)[3][3]; + (*state)[3][3] = (*state)[2][3]; + (*state)[2][3] = (*state)[1][3]; + (*state)[1][3] = temp; +} + +inline static uint8_t xtime(uint8_t x) { + return ((x<<1) ^ (((x>>7) & 1) * 0x1b)); +} + +// MixColumns function mixes the columns of the state matrix +inline static void mix_columns(state_t* state) { + uint8_t i; + uint8_t Tmp, Tm, t; + for (i = 0; i < 4; ++i) { + t = (*state)[i][0]; + Tmp = (*state)[i][0] ^ (*state)[i][1] ^ (*state)[i][2] ^ (*state)[i][3] ; + Tm = (*state)[i][0] ^ (*state)[i][1] ; Tm = xtime(Tm); (*state)[i][0] ^= Tm ^ Tmp ; + Tm = (*state)[i][1] ^ (*state)[i][2] ; Tm = xtime(Tm); (*state)[i][1] ^= Tm ^ Tmp ; + Tm = (*state)[i][2] ^ (*state)[i][3] ; Tm = xtime(Tm); (*state)[i][2] ^= Tm ^ Tmp ; + Tm = (*state)[i][3] ^ t ; Tm = xtime(Tm); (*state)[i][3] ^= Tm ^ Tmp ; + } +} + +inline static uint8_t Multiply(uint8_t x, uint8_t y) { + return ((y & 1) * x) ^ + (((y >> 1) & 0x01) * xtime(x)) ^ + (((y >> 2) & 0x01) * xtime(xtime(x))) ^ + (((y >> 3) & 0x01) * xtime(xtime(xtime(x)))); +} + +// MixColumns function mixes the columns of the state matrix. +// The method used to multiply may be difficult to understand for the inexperienced. +// Please use the references to gain more information. +inline static void inv_mix_columns(state_t* state) { + int32_t i; + uint8_t a, b, c, d; + for (i = 0; i < 4; ++i) { + a = (*state)[i][0]; + b = (*state)[i][1]; + c = (*state)[i][2]; + d = (*state)[i][3]; + + (*state)[i][0] = Multiply(a, 0x0E) ^ Multiply(b, 0x0B) ^ Multiply(c, 0x0D) ^ Multiply(d, 0x09); + (*state)[i][1] = Multiply(a, 0x09) ^ Multiply(b, 0x0E) ^ Multiply(c, 0x0B) ^ Multiply(d, 0x0D); + (*state)[i][2] = Multiply(a, 0x0D) ^ Multiply(b, 0x09) ^ Multiply(c, 0x0E) ^ Multiply(d, 0x0B); + (*state)[i][3] = Multiply(a, 0x0B) ^ Multiply(b, 0x0D) ^ Multiply(c, 0x09) ^ Multiply(d, 0x0E); + } +} + +// The SubBytes Function Substitutes the values in the +// state matrix with values in an S-box. +inline static void inv_sub_bytes(state_t* state) { + uint8_t i, j; + for (i = 0; i < 4; ++i) + for (j = 0; j < 4; ++j) + (*state)[j][i] = rsbox[(*state)[j][i]]; +} + +inline static void inv_shift_rows(state_t* state) { + uint8_t temp; + + // Rotate first row 1 columns to right + temp = (*state)[3][1]; + (*state)[3][1] = (*state)[2][1]; + (*state)[2][1] = (*state)[1][1]; + (*state)[1][1] = (*state)[0][1]; + (*state)[0][1] = temp; + + // Rotate second row 2 columns to right + temp = (*state)[0][2]; + (*state)[0][2] = (*state)[2][2]; + (*state)[2][2] = temp; + + temp = (*state)[1][2]; + (*state)[1][2] = (*state)[3][2]; + (*state)[3][2] = temp; + + // Rotate third row 3 columns to right + temp = (*state)[0][3]; + (*state)[0][3] = (*state)[1][3]; + (*state)[1][3] = (*state)[2][3]; + (*state)[2][3] = (*state)[3][3]; + (*state)[3][3] = temp; +} + +// Cipher is the main function that encrypts the PlainText. +inline static void cipher_aes128(state_t* state, const uint8_t* RoundKey) { + uint8_t round = 0; + + // Add the First round key to the state before starting the rounds. + add_round_key(0, state, RoundKey); + + // There will be Nr rounds. + // The first Nr-1 rounds are identical. + // These Nr rounds are executed in the loop below. + // Last one without MixColumns() + for (round = 1; ; ++round) { + sub_bytes(state); + shift_rows(state); + if (round == Nr128) + break; + + mix_columns(state); + add_round_key(round, state, RoundKey); + } + // Add round key to last round + add_round_key(Nr128, state, RoundKey); +} + +inline static void cipher_aes192(state_t* state, const uint8_t* RoundKey) { + uint8_t round = 0; + + // Add the First round key to the state before starting the rounds. + add_round_key(0, state, RoundKey); + + // There will be Nr rounds. + // The first Nr-1 rounds are identical. + // These Nr rounds are executed in the loop below. + // Last one without MixColumns() + for (round = 1; ; ++round) { + sub_bytes(state); + shift_rows(state); + if (round == Nr192) + break; + + mix_columns(state); + add_round_key(round, state, RoundKey); + } + // Add round key to last round + add_round_key(Nr192, state, RoundKey); +} + +inline static void cipher_aes256(state_t* state, const uint8_t* RoundKey) { + uint8_t round = 0; + + // Add the First round key to the state before starting the rounds. + add_round_key(0, state, RoundKey); + + // There will be Nr rounds. + // The first Nr-1 rounds are identical. + // These Nr rounds are executed in the loop below. + // Last one without MixColumns() + for (round = 1; ; ++round) { + sub_bytes(state); + shift_rows(state); + if (round == Nr256) + break; + + mix_columns(state); + add_round_key(round, state, RoundKey); + } + // Add round key to last round + add_round_key(Nr256, state, RoundKey); +} + +inline static void cipher_aes128_ni(void* state, __m128i* round_key) { + __m128i m = _mm_loadu_si128((__m128i*)state); + m = _mm_xor_si128(m, round_key[0]); + m = _mm_aesenc_si128(m, round_key[1]); + m = _mm_aesenc_si128(m, round_key[2]); + m = _mm_aesenc_si128(m, round_key[3]); + m = _mm_aesenc_si128(m, round_key[4]); + m = _mm_aesenc_si128(m, round_key[5]); + m = _mm_aesenc_si128(m, round_key[6]); + m = _mm_aesenc_si128(m, round_key[7]); + m = _mm_aesenc_si128(m, round_key[8]); + m = _mm_aesenc_si128(m, round_key[9]); + m = _mm_aesenclast_si128(m, round_key[10]); + _mm_storeu_si128((__m128i*)state, m); +} + +inline static void cipher_aes192_ni(void* state, __m128i* round_key) { + __m128i m = _mm_loadu_si128((__m128i*)state); + m = _mm_xor_si128(m, round_key[0]); + m = _mm_aesenc_si128(m, round_key[1]); + m = _mm_aesenc_si128(m, round_key[2]); + m = _mm_aesenc_si128(m, round_key[3]); + m = _mm_aesenc_si128(m, round_key[4]); + m = _mm_aesenc_si128(m, round_key[5]); + m = _mm_aesenc_si128(m, round_key[6]); + m = _mm_aesenc_si128(m, round_key[7]); + m = _mm_aesenc_si128(m, round_key[8]); + m = _mm_aesenc_si128(m, round_key[9]); + m = _mm_aesenc_si128(m, round_key[10]); + m = _mm_aesenc_si128(m, round_key[11]); + m = _mm_aesenclast_si128(m, round_key[12]); + _mm_storeu_si128((__m128i*)state, m); +} + +inline static void cipher_aes256_ni(void* state, __m128i* round_key) { + __m128i m = _mm_loadu_si128((__m128i*)state); + m = _mm_xor_si128(m, round_key[0]); + m = _mm_aesenc_si128(m, round_key[1]); + m = _mm_aesenc_si128(m, round_key[2]); + m = _mm_aesenc_si128(m, round_key[3]); + m = _mm_aesenc_si128(m, round_key[4]); + m = _mm_aesenc_si128(m, round_key[5]); + m = _mm_aesenc_si128(m, round_key[6]); + m = _mm_aesenc_si128(m, round_key[7]); + m = _mm_aesenc_si128(m, round_key[8]); + m = _mm_aesenc_si128(m, round_key[9]); + m = _mm_aesenc_si128(m, round_key[10]); + m = _mm_aesenc_si128(m, round_key[11]); + m = _mm_aesenc_si128(m, round_key[12]); + m = _mm_aesenc_si128(m, round_key[13]); + m = _mm_aesenclast_si128(m, round_key[14]); + _mm_storeu_si128((__m128i*)state, m); +} + +inline static void inv_cipher_aes128(state_t* state, const uint8_t* RoundKey) { + uint8_t round = 0; + + // Add the First round key to the state before starting the rounds. + add_round_key(Nr128, state, RoundKey); + + // There will be Nr rounds. + // The first Nr-1 rounds are identical. + // These Nr rounds are executed in the loop below. + // Last one without InvMixColumn() + for (round = (Nr128 - 1); ; --round) { + inv_shift_rows(state); + inv_sub_bytes(state); + add_round_key(round, state, RoundKey); + if (round == 0) + break; + + inv_mix_columns(state); + } + +} + +inline static void inv_cipher_aes192(state_t* state, const uint8_t* RoundKey) { + uint8_t round = 0; + + // Add the First round key to the state before starting the rounds. + add_round_key(Nr192, state, RoundKey); + + // There will be Nr rounds. + // The first Nr-1 rounds are identical. + // These Nr rounds are executed in the loop below. + // Last one without InvMixColumn() + for (round = (Nr192 - 1); ; --round) { + inv_shift_rows(state); + inv_sub_bytes(state); + add_round_key(round, state, RoundKey); + if (round == 0) + break; + + inv_mix_columns(state); + } + +} + +inline static void inv_cipher_aes256(state_t* state, const uint8_t* RoundKey) { + uint8_t round = 0; + + // Add the First round key to the state before starting the rounds. + add_round_key(Nr256, state, RoundKey); + + // There will be Nr rounds. + // The first Nr-1 rounds are identical. + // These Nr rounds are executed in the loop below. + // Last one without InvMixColumn() + for (round = (Nr256 - 1); ; --round) { + inv_shift_rows(state); + inv_sub_bytes(state); + add_round_key(round, state, RoundKey); + if (round == 0) + break; + + inv_mix_columns(state); + } + +} + +inline static void inv_cipher_aes128_ni(void* state, __m128i* round_key) { + __m128i m = _mm_loadu_si128((__m128i*)state); + m = _mm_xor_si128(m, round_key[10]); + m = _mm_aesdec_si128(m, round_key[11]); + m = _mm_aesdec_si128(m, round_key[12]); + m = _mm_aesdec_si128(m, round_key[13]); + m = _mm_aesdec_si128(m, round_key[14]); + m = _mm_aesdec_si128(m, round_key[15]); + m = _mm_aesdec_si128(m, round_key[16]); + m = _mm_aesdec_si128(m, round_key[17]); + m = _mm_aesdec_si128(m, round_key[18]); + m = _mm_aesdec_si128(m, round_key[19]); + m = _mm_aesdeclast_si128(m, round_key[0]); + _mm_storeu_si128((__m128i*)state, m); +} + +inline static void inv_cipher_aes192_ni(void* state, __m128i* round_key) { + __m128i m = _mm_loadu_si128((__m128i*)state); + m = _mm_xor_si128(m, round_key[12]); + m = _mm_aesdec_si128(m, round_key[13]); + m = _mm_aesdec_si128(m, round_key[14]); + m = _mm_aesdec_si128(m, round_key[15]); + m = _mm_aesdec_si128(m, round_key[16]); + m = _mm_aesdec_si128(m, round_key[17]); + m = _mm_aesdec_si128(m, round_key[18]); + m = _mm_aesdec_si128(m, round_key[19]); + m = _mm_aesdec_si128(m, round_key[20]); + m = _mm_aesdec_si128(m, round_key[21]); + m = _mm_aesdec_si128(m, round_key[22]); + m = _mm_aesdec_si128(m, round_key[23]); + m = _mm_aesdeclast_si128(m, round_key[0]); + _mm_storeu_si128((__m128i*)state, m); +} + +inline static void inv_cipher_aes256_ni(void* state, __m128i* round_key) { + __m128i m = _mm_loadu_si128((__m128i*)state); + m = _mm_xor_si128(m, round_key[14]); + m = _mm_aesdec_si128(m, round_key[15]); + m = _mm_aesdec_si128(m, round_key[16]); + m = _mm_aesdec_si128(m, round_key[17]); + m = _mm_aesdec_si128(m, round_key[18]); + m = _mm_aesdec_si128(m, round_key[19]); + m = _mm_aesdec_si128(m, round_key[20]); + m = _mm_aesdec_si128(m, round_key[21]); + m = _mm_aesdec_si128(m, round_key[22]); + m = _mm_aesdec_si128(m, round_key[23]); + m = _mm_aesdec_si128(m, round_key[24]); + m = _mm_aesdec_si128(m, round_key[25]); + m = _mm_aesdec_si128(m, round_key[26]); + m = _mm_aesdec_si128(m, round_key[27]); + m = _mm_aesdeclast_si128(m, round_key[0]); + _mm_storeu_si128((__m128i*)state, m); +} + +/*****************************************************************************/ +/* Public functions: */ +/*****************************************************************************/ +void aes128_ecb_encrypt(aes128_ctx* ctx, uint8_t* buf) { + // The next function call encrypts the PlainText with the Key using AES algorithm. + if (aes_ni) + cipher_aes128_ni(buf, ctx->RoundKeyNI); + else + cipher_aes128((state_t*)buf, ctx->RoundKey); +} + +void aes192_ecb_encrypt(aes192_ctx* ctx, uint8_t* buf) { + // The next function call encrypts the PlainText with the Key using AES algorithm. + if (aes_ni) + cipher_aes192_ni(buf, ctx->RoundKeyNI); + else + cipher_aes192((state_t*)buf, ctx->RoundKey); +} + +void aes256_ecb_encrypt(aes256_ctx* ctx, uint8_t* buf) { + // The next function call encrypts the PlainText with the Key using AES algorithm. + if (aes_ni) + cipher_aes256_ni(buf, ctx->RoundKeyNI); + else + cipher_aes256((state_t*)buf, ctx->RoundKey); +} + +void aes128_ecb_decrypt(aes128_ctx* ctx, uint8_t* buf) { + // The next function call decrypts the PlainText with the Key using AES algorithm. + if (aes_ni) + inv_cipher_aes128_ni(buf, ctx->RoundKeyNI); + else + inv_cipher_aes128((state_t*)buf, ctx->RoundKey); +} + +void aes192_ecb_decrypt(aes192_ctx* ctx, uint8_t* buf) { + // The next function call decrypts the PlainText with the Key using AES algorithm. + if (aes_ni) + inv_cipher_aes192_ni(buf, ctx->RoundKeyNI); + else + inv_cipher_aes192((state_t*)buf, ctx->RoundKey); +} + +void aes256_ecb_decrypt(aes256_ctx* ctx, uint8_t* buf) { + // The next function call decrypts the PlainText with the Key using AES algorithm. + if (aes_ni) + inv_cipher_aes256_ni(buf, ctx->RoundKeyNI); + else + inv_cipher_aes256((state_t*)buf, ctx->RoundKey); +} + +void aes128_ecb_encrypt_buffer(aes128_ctx* ctx, uint8_t* buf, size_t length) { + if (aes_ni) + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) + cipher_aes128_ni(buf, ctx->RoundKeyNI); + else + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) + cipher_aes128((state_t*)buf, ctx->RoundKey); +} + +void aes192_ecb_encrypt_buffer(aes192_ctx* ctx, uint8_t* buf, size_t length) { + if (aes_ni) + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) + cipher_aes192_ni(buf, ctx->RoundKeyNI); + else + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) + cipher_aes192((state_t*)buf, ctx->RoundKey); +} + +void aes256_ecb_encrypt_buffer(aes256_ctx* ctx, uint8_t* buf, size_t length) { + if (aes_ni) + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) + cipher_aes256_ni(buf, ctx->RoundKeyNI); + else + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) + cipher_aes256((state_t*)buf, ctx->RoundKey); +} + +void aes128_ecb_decrypt_buffer(aes128_ctx* ctx, uint8_t* buf, size_t length) { + if (aes_ni) + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) + inv_cipher_aes128_ni(buf, ctx->RoundKeyNI); + else + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) + inv_cipher_aes128((state_t*)buf, ctx->RoundKey); +} + +void aes192_ecb_decrypt_buffer(aes192_ctx* ctx, uint8_t* buf, size_t length) { + if (aes_ni) + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) + inv_cipher_aes192_ni(buf, ctx->RoundKeyNI); + else + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) + inv_cipher_aes192((state_t*)buf, ctx->RoundKey); +} + +void aes256_ecb_decrypt_buffer(aes256_ctx* ctx, uint8_t* buf, size_t length) { + if (aes_ni) + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) + inv_cipher_aes256_ni(buf, ctx->RoundKeyNI); + else + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) + inv_cipher_aes256((state_t*)buf, ctx->RoundKey); +} + +inline static void XorWithIv(uint8_t* buf, const uint8_t* Iv) { + _mm_storeu_si128((__m128i*)buf, + _mm_xor_si128( + _mm_loadu_si128((const __m128i*)buf), + _mm_loadu_si128((const __m128i*)Iv) + ) + ); +} + +void aes128_cbc_encrypt(aes128_ctx* ctx, uint8_t buf[AES_BLOCKLEN]) { + uint8_t* Iv = ctx->Iv; + if (aes_ni) { + XorWithIv(buf, Iv); + cipher_aes128_ni(buf, ctx->RoundKeyNI); + Iv = buf; + } + else { + XorWithIv(buf, Iv); + cipher_aes128((state_t*)buf, ctx->RoundKey); + Iv = buf; + } + /* store Iv in ctx for next call */ + memcpy(ctx->Iv, Iv, AES_BLOCKLEN); +} + +void aes192_cbc_encrypt(aes192_ctx* ctx, uint8_t buf[AES_BLOCKLEN]) { + uint8_t* Iv = ctx->Iv; + if (aes_ni) { + XorWithIv(buf, Iv); + cipher_aes192_ni(buf, ctx->RoundKeyNI); + Iv = buf; + } + else { + XorWithIv(buf, Iv); + cipher_aes192((state_t*)buf, ctx->RoundKey); + Iv = buf; + } + /* store Iv in ctx for next call */ + memcpy(ctx->Iv, Iv, AES_BLOCKLEN); +} + +void aes256_cbc_encrypt(aes256_ctx* ctx, uint8_t buf[AES_BLOCKLEN]) { + uint8_t* Iv = ctx->Iv; + if (aes_ni) { + XorWithIv(buf, Iv); + cipher_aes256_ni(buf, ctx->RoundKeyNI); + Iv = buf; + } + else { + XorWithIv(buf, Iv); + cipher_aes256((state_t*)buf, ctx->RoundKey); + Iv = buf; + } + /* store Iv in ctx for next call */ + memcpy(ctx->Iv, Iv, AES_BLOCKLEN); +} + +void aes128_cbc_decrypt(aes128_ctx* ctx, uint8_t buf[AES_BLOCKLEN]) { + uint8_t storeNextIv[AES_BLOCKLEN]; + if (aes_ni) { + memcpy(storeNextIv, buf, AES_BLOCKLEN); + inv_cipher_aes128_ni(buf, ctx->RoundKeyNI); + XorWithIv(buf, ctx->Iv); + memcpy(ctx->Iv, storeNextIv, AES_BLOCKLEN); + } + else { + memcpy(storeNextIv, buf, AES_BLOCKLEN); + inv_cipher_aes128((state_t*)buf, ctx->RoundKey); + XorWithIv(buf, ctx->Iv); + memcpy(ctx->Iv, storeNextIv, AES_BLOCKLEN); + } +} + +void aes192_cbc_decrypt(aes192_ctx* ctx, uint8_t buf[AES_BLOCKLEN]) { + uint8_t storeNextIv[AES_BLOCKLEN]; + if (aes_ni) { + memcpy(storeNextIv, buf, AES_BLOCKLEN); + inv_cipher_aes192_ni(buf, ctx->RoundKeyNI); + XorWithIv(buf, ctx->Iv); + memcpy(ctx->Iv, storeNextIv, AES_BLOCKLEN); + } + else { + memcpy(storeNextIv, buf, AES_BLOCKLEN); + inv_cipher_aes192((state_t*)buf, ctx->RoundKey); + XorWithIv(buf, ctx->Iv); + memcpy(ctx->Iv, storeNextIv, AES_BLOCKLEN); + } +} + +void aes256_cbc_decrypt(aes256_ctx* ctx, uint8_t buf[AES_BLOCKLEN]) { + uint8_t storeNextIv[AES_BLOCKLEN]; + if (aes_ni) { + memcpy(storeNextIv, buf, AES_BLOCKLEN); + inv_cipher_aes256_ni(buf, ctx->RoundKeyNI); + XorWithIv(buf, ctx->Iv); + memcpy(ctx->Iv, storeNextIv, AES_BLOCKLEN); + } + else { + memcpy(storeNextIv, buf, AES_BLOCKLEN); + inv_cipher_aes256((state_t*)buf, ctx->RoundKey); + XorWithIv(buf, ctx->Iv); + memcpy(ctx->Iv, storeNextIv, AES_BLOCKLEN); + } +} + +void aes128_cbc_encrypt_buffer(aes128_ctx *ctx, uint8_t* buf, size_t length) { + uint8_t* Iv = ctx->Iv; + if (aes_ni) + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) { + XorWithIv(buf, Iv); + cipher_aes128_ni(buf, ctx->RoundKeyNI); + Iv = buf; + } + else + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) { + XorWithIv(buf, Iv); + cipher_aes128((state_t*)buf, ctx->RoundKey); + Iv = buf; + } + /* store Iv in ctx for next call */ + memcpy(ctx->Iv, Iv, AES_BLOCKLEN); +} + +void aes192_cbc_encrypt_buffer(aes192_ctx *ctx, uint8_t* buf, size_t length) { + uint8_t* Iv = ctx->Iv; + if (aes_ni) + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) { + XorWithIv(buf, Iv); + cipher_aes192_ni(buf, ctx->RoundKeyNI); + Iv = buf; + } + else + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) { + XorWithIv(buf, Iv); + cipher_aes192((state_t*)buf, ctx->RoundKey); + Iv = buf; + } + /* store Iv in ctx for next call */ + memcpy(ctx->Iv, Iv, AES_BLOCKLEN); +} + +void aes256_cbc_encrypt_buffer(aes256_ctx *ctx, uint8_t* buf, size_t length) { + uint8_t* Iv = ctx->Iv; + if (aes_ni) + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) { + XorWithIv(buf, Iv); + cipher_aes256_ni(buf, ctx->RoundKeyNI); + Iv = buf; + } + else + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) { + XorWithIv(buf, Iv); + cipher_aes256((state_t*)buf, ctx->RoundKey); + Iv = buf; + } + /* store Iv in ctx for next call */ + memcpy(ctx->Iv, Iv, AES_BLOCKLEN); +} + +void aes128_cbc_decrypt_buffer(aes128_ctx* ctx, uint8_t* buf, size_t length) { + uint8_t storeNextIv[AES_BLOCKLEN]; + if (aes_ni) + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) { + memcpy(storeNextIv, buf, AES_BLOCKLEN); + inv_cipher_aes128_ni(buf, ctx->RoundKeyNI); + XorWithIv(buf, ctx->Iv); + memcpy(ctx->Iv, storeNextIv, AES_BLOCKLEN); + } + else + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) { + memcpy(storeNextIv, buf, AES_BLOCKLEN); + inv_cipher_aes128((state_t*)buf, ctx->RoundKey); + XorWithIv(buf, ctx->Iv); + memcpy(ctx->Iv, storeNextIv, AES_BLOCKLEN); + } +} + +void aes192_cbc_decrypt_buffer(aes192_ctx* ctx, uint8_t* buf, size_t length) { + uint8_t storeNextIv[AES_BLOCKLEN]; + if (aes_ni) + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) { + memcpy(storeNextIv, buf, AES_BLOCKLEN); + inv_cipher_aes192_ni(buf, ctx->RoundKeyNI); + XorWithIv(buf, ctx->Iv); + memcpy(ctx->Iv, storeNextIv, AES_BLOCKLEN); + } + else + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) { + memcpy(storeNextIv, buf, AES_BLOCKLEN); + inv_cipher_aes192((state_t*)buf, ctx->RoundKey); + XorWithIv(buf, ctx->Iv); + memcpy(ctx->Iv, storeNextIv, AES_BLOCKLEN); + } +} + +void aes256_cbc_decrypt_buffer(aes256_ctx* ctx, uint8_t* buf, size_t length) { + uint8_t storeNextIv[AES_BLOCKLEN]; + if (aes_ni) + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) { + memcpy(storeNextIv, buf, AES_BLOCKLEN); + inv_cipher_aes256_ni(buf, ctx->RoundKeyNI); + XorWithIv(buf, ctx->Iv); + memcpy(ctx->Iv, storeNextIv, AES_BLOCKLEN); + } + else + for (size_t i = 0; i < length; i += AES_BLOCKLEN, buf += AES_BLOCKLEN) { + memcpy(storeNextIv, buf, AES_BLOCKLEN); + inv_cipher_aes256((state_t*)buf, ctx->RoundKey); + XorWithIv(buf, ctx->Iv); + memcpy(ctx->Iv, storeNextIv, AES_BLOCKLEN); + } +} + +/* Symmetrical operation: same function for encrypting as for decrypting. Note any IV/nonce should never be reused with the same key */ +void aes128_ctr_xcrypt_buffer(aes128_ctx* ctx, uint8_t* buf, uint32_t length) { + uint8_t buffer[AES_BLOCKLEN]; + + int32_t bi; + size_t i; + for (i = 0, bi = AES_BLOCKLEN; i < length; i++, bi++) { + if (bi == AES_BLOCKLEN) { /* we need to regen xor compliment in buffer */ + memcpy(buffer, ctx->Iv, AES_BLOCKLEN); + if (aes_ni) + cipher_aes128_ni(buffer, ctx->RoundKeyNI); + else + cipher_aes128((state_t*)buffer, ctx->RoundKey); + + /* Increment Iv and handle overflow */ + for (bi = (AES_BLOCKLEN - 1); bi >= 0; bi--) { + /* inc will overflow */ + if (ctx->Iv[bi] == 255) { + ctx->Iv[bi] = 0; + continue; + } + ctx->Iv[bi]++; + break; + } + bi = 0; + } + + buf[i] ^= buffer[bi]; + } +} + +void aes192_ctr_xcrypt_buffer(aes192_ctx* ctx, uint8_t* buf, uint32_t length) { + uint8_t buffer[AES_BLOCKLEN]; + + int32_t bi; + size_t i; + for (i = 0, bi = AES_BLOCKLEN; i < length; i++, bi++) { + if (bi == AES_BLOCKLEN) { /* we need to regen xor compliment in buffer */ + memcpy(buffer, ctx->Iv, AES_BLOCKLEN); + if (aes_ni) + cipher_aes192_ni(buffer, ctx->RoundKeyNI); + else + cipher_aes192((state_t*)buffer, ctx->RoundKey); + + /* Increment Iv and handle overflow */ + for (bi = (AES_BLOCKLEN - 1); bi >= 0; bi--) { + /* inc will overflow */ + if (ctx->Iv[bi] == 255) { + ctx->Iv[bi] = 0; + continue; + } + ctx->Iv[bi]++; + break; + } + bi = 0; + } + + buf[i] ^= buffer[bi]; + } +} + +void aes256_ctr_xcrypt_buffer(aes256_ctx* ctx, uint8_t* buf, uint32_t length) { + uint8_t buffer[AES_BLOCKLEN]; + + int32_t bi; + size_t i; + for (i = 0, bi = AES_BLOCKLEN; i < length; i++, bi++) { + if (bi == AES_BLOCKLEN) { /* we need to regen xor compliment in buffer */ + memcpy(buffer, ctx->Iv, AES_BLOCKLEN); + if (aes_ni) + cipher_aes256_ni(buffer, ctx->RoundKeyNI); + else + cipher_aes256((state_t*)buffer, ctx->RoundKey); + + /* Increment Iv and handle overflow */ + for (bi = (AES_BLOCKLEN - 1); bi >= 0; bi--) { + /* inc will overflow */ + if (ctx->Iv[bi] == 255) { + ctx->Iv[bi] = 0; + continue; + } + ctx->Iv[bi]++; + break; + } + bi = 0; + } + + buf[i] ^= buffer[bi]; + } +} diff --git a/src/KKdLib/aes.hpp b/src/KKdLib/aes.hpp new file mode 100644 index 0000000..73fbefc --- /dev/null +++ b/src/KKdLib/aes.hpp @@ -0,0 +1,90 @@ +/* + Original: https://github.com/kokke/tiny-AES-c +*/ + +#ifndef _AES_H_ +#define _AES_H_ + +#include "default.hpp" +#include + +//#define AES128 1 +//#define AES192 1 +//#define AES256 1 + +#define AES_BLOCKLEN 16 // Block length in bytes - AES is 128b block only + +#define AES128_KEYLEN 16 // Key length in bytes +#define AES128_keyExpSize 176 +#define AES192_KEYLEN 24 +#define AES192_keyExpSize 208 +#define AES256_KEYLEN 32 +#define AES256_keyExpSize 240 + +struct aes128_ctx { + union { + uint8_t RoundKey[AES128_keyExpSize]; + __m128i RoundKeyNI[AES128_keyExpSize / sizeof(__m128i) + 9]; + }; + uint8_t Iv[AES_BLOCKLEN]; +}; + +struct aes192_ctx { + union { + uint8_t RoundKey[AES192_keyExpSize]; + __m128i RoundKeyNI[AES192_keyExpSize / sizeof(__m128i) + 11]; + }; + uint8_t Iv[AES_BLOCKLEN]; +}; + +struct aes256_ctx { + union { + uint8_t RoundKey[AES256_keyExpSize]; + __m128i RoundKeyNI[AES256_keyExpSize / sizeof(__m128i) + 13]; + }; + uint8_t Iv[AES_BLOCKLEN]; +}; + +void aes128_init_ctx(aes128_ctx* ctx, const uint8_t* key); +void aes192_init_ctx(aes192_ctx* ctx, const uint8_t* key); +void aes256_init_ctx(aes256_ctx* ctx, const uint8_t* key); +void aes128_init_ctx_iv(aes128_ctx* ctx, const uint8_t* key, const uint8_t* iv); +void aes192_init_ctx_iv(aes192_ctx* ctx, const uint8_t* key, const uint8_t* iv); +void aes256_init_ctx_iv(aes256_ctx* ctx, const uint8_t* key, const uint8_t* iv); +void aes128_ctx_set_iv(aes128_ctx* ctx, const uint8_t* iv); +void aes192_ctx_set_iv(aes192_ctx* ctx, const uint8_t* iv); +void aes256_ctx_set_iv(aes256_ctx* ctx, const uint8_t* iv); + +// buffer size is exactly AES_BLOCKLEN bytes; +// you need only AES_init_ctx as IV is not used in ECB +void aes128_ecb_encrypt(aes128_ctx* ctx, uint8_t* buf); +void aes192_ecb_encrypt(aes192_ctx* ctx, uint8_t* buf); +void aes256_ecb_encrypt(aes256_ctx* ctx, uint8_t* buf); +void aes128_ecb_decrypt(aes128_ctx* ctx, uint8_t* buf); +void aes192_ecb_decrypt(aes192_ctx* ctx, uint8_t* buf); +void aes256_ecb_decrypt(aes256_ctx* ctx, uint8_t* buf); +void aes128_ecb_encrypt_buffer(aes128_ctx* ctx, uint8_t* buf, size_t length); +void aes192_ecb_encrypt_buffer(aes192_ctx* ctx, uint8_t* buf, size_t length); +void aes256_ecb_encrypt_buffer(aes256_ctx* ctx, uint8_t* buf, size_t length); +void aes128_ecb_decrypt_buffer(aes128_ctx* ctx, uint8_t* buf, size_t length); +void aes192_ecb_decrypt_buffer(aes192_ctx* ctx, uint8_t* buf, size_t length); +void aes256_ecb_decrypt_buffer(aes256_ctx* ctx, uint8_t* buf, size_t length); + +void aes128_cbc_encrypt(aes128_ctx* ctx, uint8_t buf[AES_BLOCKLEN]); +void aes192_cbc_encrypt(aes192_ctx* ctx, uint8_t buf[AES_BLOCKLEN]); +void aes256_cbc_encrypt(aes256_ctx* ctx, uint8_t buf[AES_BLOCKLEN]); +void aes128_cbc_decrypt(aes128_ctx* ctx, uint8_t buf[AES_BLOCKLEN]); +void aes192_cbc_decrypt(aes192_ctx* ctx, uint8_t buf[AES_BLOCKLEN]); +void aes256_cbc_decrypt(aes256_ctx* ctx, uint8_t buf[AES_BLOCKLEN]); +void aes128_cbc_encrypt_buffer(aes128_ctx* ctx, uint8_t* buf, size_t length); +void aes192_cbc_encrypt_buffer(aes192_ctx* ctx, uint8_t* buf, size_t length); +void aes256_cbc_encrypt_buffer(aes256_ctx* ctx, uint8_t* buf, size_t length); +void aes128_cbc_decrypt_buffer(aes128_ctx* ctx, uint8_t* buf, size_t length); +void aes192_cbc_decrypt_buffer(aes192_ctx* ctx, uint8_t* buf, size_t length); +void aes256_cbc_decrypt_buffer(aes256_ctx* ctx, uint8_t* buf, size_t length); + +void aes128_ctr_xcrypt_buffer(aes128_ctx* ctx, uint8_t* buf, uint32_t length); +void aes192_ctr_xcrypt_buffer(aes192_ctx* ctx, uint8_t* buf, uint32_t length); +void aes256_ctr_xcrypt_buffer(aes256_ctx* ctx, uint8_t* buf, uint32_t length); + +#endif // _AES_H_ diff --git a/src/KKdLib/default.cpp b/src/KKdLib/default.cpp new file mode 100644 index 0000000..526ec0b --- /dev/null +++ b/src/KKdLib/default.cpp @@ -0,0 +1,345 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "default.hpp" + +void* force_malloc(size_t size) { + if (!size) + return 0; + + void* buf = 0; + while (!buf) + buf = malloc(size); + memset(buf, 0, size); + return buf; +} + +wchar_t* utf8_to_utf16(const char* s) { + if (!s) + return 0; + + uint32_t c = 0; + size_t length = utf8_to_utf16_length(s); + + wchar_t* str = force_malloc(length + 1); + if (!str) + return 0; + + size_t j = 0; + size_t l = 0; + while (*s) { + char t = *s++; + if (!(t & 0x80)) { + l = 0; + str[j++] = t; + c = 0; + continue; + } + else if ((t & 0xFC) == 0xF8) + continue; + else if ((t & 0xF8) == 0xF0) { + c = t & 0x07; + l = 3; + } + else if ((t & 0xF0) == 0xE0) { + c = t & 0x0F; + l = 2; + } + else if ((t & 0xE0) == 0xC0) { + c = t & 0x1F; + l = 1; + } + else if ((t & 0xC0) == 0x80) { + c = (c << 6) | (t & 0x3F); + l--; + } + + if (!l) { + if (c <= 0xD7FF || (c >= 0xE000 && c <= 0xFFFF)) + str[j++] = c; + else if (c >= 0x10000 && c <= 0x10FFFF) { + c -= 0x10000; + str[j++] = 0xD800 | ((c >> 10) & 0x3FF); + str[j++] = 0xDC00 | (c & 0x3FF); + } + c = 0; + } + } + str[length] = 0; + return str; +} + +wchar_t* utf8_to_utf16(const char* s, size_t length) { + if (!s || !length) + return 0; + + uint32_t c = 0; + size_t _length = utf8_to_utf16_length(s); + + wchar_t* str = force_malloc(_length + 1); + if (!str) + return 0; + + size_t j = 0; + size_t l = 0; + while (*s && length) { + char t = *s++; + length--; + if (!(t & 0x80)) { + l = 0; + str[j++] = t; + c = 0; + continue; + } + else if ((t & 0xFC) == 0xF8) + continue; + else if ((t & 0xF8) == 0xF0) { + c = t & 0x07; + l = 3; + } + else if ((t & 0xF0) == 0xE0) { + c = t & 0x0F; + l = 2; + } + else if ((t & 0xE0) == 0xC0) { + c = t & 0x1F; + l = 1; + } + else if ((t & 0xC0) == 0x80) { + c = (c << 6) | (t & 0x3F); + l--; + } + + if (!l) { + if (c <= 0xD7FF || (c >= 0xE000 && c <= 0xFFFF)) + str[j++] = c; + else if (c >= 0x10000 && c <= 0x10FFFF) { + c -= 0x10000; + str[j++] = 0xD800 | ((c >> 10) & 0x3FF); + str[j++] = 0xDC00 | (c & 0x3FF); + } + c = 0; + } + } + str[_length] = 0; + return str; +} + +char* utf16_to_utf8(const wchar_t* s) { + if (!s) + return 0; + + uint32_t c = 0; + size_t length = utf16_to_utf8_length(s); + + char* str = force_malloc(length + 1); + if (!str) + return 0; + + size_t j = 0; + while (*s) { + c = *s++; + if ((c & 0xFC00) == 0xD800) { + if (!*s) + break; + + wchar_t _c = *s++; + if ((_c & 0xFC00) != 0xDC00) + continue; + + c &= 0x3FF; + c <<= 10; + c |= _c & 0x3FF; + c += 0x10000; + } + else if ((c & 0xFC00) == 0xDC00) + continue; + + if (c <= 0x7F) + str[j++] = (uint8_t)c; + else if (c <= 0x7FF) { + str[j++] = (uint8_t)(0xC0 | ((c >> 6) & 0x1F)); + str[j++] = (uint8_t)(0x80 | (c & 0x3F)); + } + else if ((c >= 0x800 && c <= 0xD7FF) || (c >= 0xE000 && c <= 0xFFFF)) { + str[j++] = (uint8_t)(0xE0 | ((c >> 12) & 0xF)); + str[j++] = (uint8_t)(0x80 | ((c >> 6) & 0x3F)); + str[j++] = (uint8_t)(0x80 | (c & 0x3F)); + } + else if (c >= 0x10000 && c <= 0x10FFFF) { + str[j++] = (uint8_t)(0xF0 | ((c >> 18) & 0x7)); + str[j++] = (uint8_t)(0x80 | ((c >> 12) & 0x3F)); + str[j++] = (uint8_t)(0x80 | ((c >> 6) & 0x3F)); + str[j++] = (uint8_t)(0x80 | (c & 0x3F)); + } + } + str[length] = 0; + return str; +} + +char* utf16_to_utf8(const wchar_t* s, size_t length) { + if (!s || !length) + return 0; + + uint32_t c = 0; + size_t _length = utf16_to_utf8_length(s); + + char* str = force_malloc(_length + 1); + if (!str) + return 0; + + size_t j = 0; + while (*s && length) { + c = *s++; + length--; + if ((c & 0xFC00) == 0xD800) { + if (!*s) + break; + + wchar_t _c = *s++; + if ((_c & 0xFC00) != 0xDC00) + continue; + + c &= 0x3FF; + c <<= 10; + c |= _c & 0x3FF; + c += 0x10000; + } + else if ((c & 0xFC00) == 0xDC00) + continue; + + if (c <= 0x7F) + str[j++] = (uint8_t)c; + else if (c <= 0x7FF) { + str[j++] = (uint8_t)(0xC0 | ((c >> 6) & 0x1F)); + str[j++] = (uint8_t)(0x80 | (c & 0x3F)); + } + else if ((c >= 0x800 && c <= 0xD7FF) || (c >= 0xE000 && c <= 0xFFFF)) { + str[j++] = (uint8_t)(0xE0 | ((c >> 12) & 0xF)); + str[j++] = (uint8_t)(0x80 | ((c >> 6) & 0x3F)); + str[j++] = (uint8_t)(0x80 | (c & 0x3F)); + } + else if (c >= 0x10000 && c <= 0x10FFFF) { + str[j++] = (uint8_t)(0xF0 | ((c >> 18) & 0x7)); + str[j++] = (uint8_t)(0x80 | ((c >> 12) & 0x3F)); + str[j++] = (uint8_t)(0x80 | ((c >> 6) & 0x3F)); + str[j++] = (uint8_t)(0x80 | (c & 0x3F)); + } + } + str[_length] = 0; + return str; +} + +std::wstring utf8_to_utf16(const std::string& s) { + size_t length = s.size(); + if (!length) + return {}; + + uint32_t c = 0; + size_t _length = utf8_to_utf16_length(s.c_str()); + + std::wstring str; + str.resize(_length); + + char* _s = (char*)s.data(); + wchar_t* _str = (wchar_t*)str.data(); + + size_t l = 0; + while (*_s && length) { + char t = *_s++; + length--; + if (!(t & 0x80)) { + l = 0; + *_str++ = t; + c = 0; + continue; + } + else if ((t & 0xFC) == 0xF8) + continue; + else if ((t & 0xF8) == 0xF0) { + c = t & 0x07; + l = 3; + } + else if ((t & 0xF0) == 0xE0) { + c = t & 0x0F; + l = 2; + } + else if ((t & 0xE0) == 0xC0) { + c = t & 0x1F; + l = 1; + } + else if ((t & 0xC0) == 0x80) { + c = (c << 6) | (t & 0x3F); + l--; + } + + if (!l) { + if (c <= 0xD7FF || (c >= 0xE000 && c <= 0xFFFF)) + *_str++ = c; + else if (c >= 0x10000 && c <= 0x10FFFF) { + c -= 0x10000; + *_str++ = 0xD800 | ((c >> 10) & 0x3FF); + } + c = 0; + } + } + *_str++ = 0; + return str; +} + +std::string utf16_to_utf8(const std::wstring& s) { + size_t length = s.size(); + if (!length) + return {}; + + uint32_t c = 0; + size_t _length = utf16_to_utf8_length(s.c_str()); + + std::string str; + str.resize(_length); + + wchar_t* _s = (wchar_t*)s.data(); + char* _str = (char*)str.data(); + + while (*_s && length) { + c = *_s++; + length--; + if ((c & 0xFC00) == 0xD800) { + if (!*_s) + break; + + wchar_t _c = *_s++; + if ((_c & 0xFC00) != 0xDC00) + continue; + + c &= 0x3FF; + c <<= 10; + c |= _c & 0x3FF; + c += 0x10000; + } + else if ((c & 0xFC00) == 0xDC00) + continue; + + if (c <= 0x7F) + *_str++ = (uint8_t)c; + else if (c <= 0x7FF) { + *_str++ = (uint8_t)(0xC0 | ((c >> 6) & 0x1F)); + *_str++ = (uint8_t)(0x80 | (c & 0x3F)); + } + else if ((c >= 0x800 && c <= 0xD7FF) || (c >= 0xE000 && c <= 0xFFFF)) { + *_str++ = (uint8_t)(0xE0 | ((c >> 12) & 0xF)); + *_str++ = (uint8_t)(0x80 | ((c >> 6) & 0x3F)); + *_str++ = (uint8_t)(0x80 | (c & 0x3F)); + } + else if (c >= 0x10000 && c <= 0x10FFFF) { + *_str++ = (uint8_t)(0xF0 | ((c >> 18) & 0x7)); + *_str++ = (uint8_t)(0x80 | ((c >> 12) & 0x3F)); + *_str++ = (uint8_t)(0x80 | ((c >> 6) & 0x3F)); + *_str++ = (uint8_t)(0x80 | (c & 0x3F)); + } + } + *_str++ = 0; + return str; +} diff --git a/src/KKdLib/default.hpp b/src/KKdLib/default.hpp new file mode 100644 index 0000000..c23e28f --- /dev/null +++ b/src/KKdLib/default.hpp @@ -0,0 +1,543 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#define NOMINMAX +#define _USE_MATH_DEFINES +#include "types.hpp" +#include "prj/math.hpp" +#include +#include +#include +#include +#define WIN32_LEAN_AND_MEAN +#include +#include +#include + +#pragma warning( push ) +#pragma warning( disable: 26812 ) + +template +inline void free_def(T& ptr) { + if (ptr) + free(ptr); + ptr = 0; +} + +template +struct null_def { + bool has_value; + T value; +}; + +template +inline T max_def(const T left, const U right) { + return left > right ? left : right; +} + +template +inline T min_def(const T left, const U right) { + return left < right ? left : right; +} + +template +inline T mult_min_max_def(const T value, const U pos_scale, const U neg_scale) { + return (value >= (T)0) ? (value * pos_scale) : (value * neg_scale); +} + +template +inline T div_min_max_def(const T value, const U pos_scale, const U neg_scale) { + return (value >= (T)0) ? (value / pos_scale) : (value / neg_scale); +} + +template +inline T clamp_def(const T value, const U min, const V max) { + return min_def(max_def(value, min), max); +} + +template +inline T align_val(const T value, const U align) { + return (value + align - (T)1) / align * align; +} + +template +inline T align_val_divide(const T value, const U align, const V div) { + return (value + align - (T)1) / div; +} + +#define BUF_SIZE 4096 + +#define RAD_TO_DEG ((double_t)(180.0 / M_PI)) + +#define DEG_TO_RAD ((double_t)(M_PI / 180.0)) + +#define RAD_TO_DEG_FLOAT ((float_t)(180.0 / M_PI)) + +#define DEG_TO_RAD_FLOAT ((float_t)(M_PI / 180.0)) + +inline float __CRTDECL ctgf(float _X) { + return 1.0f / tanf(_X); +} + +inline float __CRTDECL ctghf(float _X) { + return 1.0f / tanhf(_X); +} + +inline float __CRTDECL actgf(float _X) { + return 1.0f / atanf(_X); +} + +inline float __CRTDECL actghf(float _X) { + return 1.0f / atanhf(_X); +} + +inline double __CRTDECL ctg(double _X) { + return 1.0 / tan(_X); +} + +inline double __CRTDECL ctgh(double _X) { + return 1.0 / tanh(_X); +} + +inline double __CRTDECL actg(double _X) { + return 1.0 / atan(_X); +} + +inline double __CRTDECL actgh(double _X) { + return 1.0 / atanh(_X); +} + +template +inline T lerp_def(T x, T y, U blend) { + return ((U)1 - blend) * x + blend * y; +} + +extern void* force_malloc(size_t size); + +template +inline T* force_malloc() { + return (T*)force_malloc(sizeof(T)); +} + +template +inline T* force_malloc(size_t size) { + return (T*)force_malloc(sizeof(T) * (size)); +} + +#define enum_or(s0, s1) \ +(s0) = (decltype(s0))((int32_t)(s0) | (s1)) + +#define enum_xor(s0, s1) \ +(s0) = (decltype(s0))((int32_t)(s0) ^ (s1)) + +#define enum_and(s0, s1) \ +(s0) = (decltype(s0))((int32_t)(s0) & (s1)) + +#define enum_not(s0, s1) \ +(s0) = (decltype(s0))(~(int32_t)(s0)) + +inline int16_t load_reverse_endianness_int16_t(const void* ptr) { + return (int16_t)_byteswap_ushort(*(uint16_t*)ptr); +} + +inline uint16_t load_reverse_endianness_uint16_t(const void* ptr) { + return (uint16_t)_byteswap_ushort(*(uint16_t*)ptr); +} + +inline int32_t load_reverse_endianness_int32_t(const void* ptr) { + return (int32_t)_byteswap_ulong(*(uint32_t*)ptr); +} + +inline uint32_t load_reverse_endianness_uint32_t(const void* ptr) { + return (uint32_t)_byteswap_ulong(*(uint32_t*)ptr); +} + +inline int64_t load_reverse_endianness_int64_t(const void* ptr) { + return (int64_t)_byteswap_uint64(*(uint64_t*)ptr); +} + +inline uint64_t load_reverse_endianness_uint64_t(const void* ptr) { + return (uint64_t)_byteswap_uint64(*(uint64_t*)ptr); +} + +inline ssize_t load_reverse_endianness_ssize_t(const void* ptr) { + return (ssize_t)_byteswap_uint64(*(uint64_t*)ptr); +} + +inline size_t load_reverse_endianness_size_t(const void* ptr) { + return (size_t)_byteswap_uint64(*(uint64_t*)ptr); +} + +inline float_t load_reverse_endianness_float_t(const void* ptr) { + uint32_t v = (uint32_t)_byteswap_ulong(*(uint32_t*)ptr); + return *(float_t*)&v; +} + +inline double_t load_reverse_endianness_double_t(const void* ptr) { + uint64_t v = (uint64_t)_byteswap_uint64(*(uint64_t*)ptr); + return *(double_t*)&v; +} + +inline void store_reverse_endianness_int16_t(void* ptr, int16_t value) { + *(int16_t*)ptr = (int16_t)_byteswap_ushort((uint16_t)value); +} + +inline void store_reverse_endianness_int16_t(void* ptr, uint16_t value) { + *(int16_t*)ptr = (int16_t)_byteswap_ushort(value); +} + +inline void store_reverse_endianness_uint16_t(void* ptr, int16_t value) { + *(uint16_t*)ptr = (uint16_t)_byteswap_ushort((uint16_t)value); +} + +inline void store_reverse_endianness_uint16_t(void* ptr, uint16_t value) { + *(uint16_t*)ptr = (uint16_t)_byteswap_ushort(value); +} + +inline void store_reverse_endianness_int32_t(void* ptr, int32_t value) { + *(int32_t*)ptr = (int32_t)_byteswap_ulong((uint32_t)value); +} + +inline void store_reverse_endianness_int32_t(void* ptr, uint32_t value) { + *(int32_t*)ptr = (int32_t)_byteswap_ulong(value); +} + +inline void store_reverse_endianness_uint32_t(void* ptr, int32_t value) { + *(uint32_t*)ptr = (uint32_t)_byteswap_ulong((uint32_t)value); +} + +inline void store_reverse_endianness_uint32_t(void* ptr, uint32_t value) { + *(uint32_t*)ptr = (uint32_t)_byteswap_ulong(value); +} + +inline void store_reverse_endianness_int64_t(void* ptr, int64_t value) { + *(int64_t*)ptr = (int64_t)_byteswap_uint64((uint64_t)value); +} + +inline void store_reverse_endianness_int64_t(void* ptr, uint64_t value) { + *(int64_t*)ptr = (int64_t)_byteswap_uint64(value); +} + +inline void store_reverse_endianness_uint64_t(void* ptr, int64_t value) { + *(uint64_t*)ptr = (uint64_t)_byteswap_uint64((uint64_t)value); +} + +inline void store_reverse_endianness_uint64_t(void* ptr, uint64_t value) { + *(uint64_t*)ptr = (uint64_t)_byteswap_uint64(value); +} + +inline void store_reverse_endianness_ssize_t(void* ptr, ssize_t value) { + *(ssize_t*)ptr = (ssize_t)_byteswap_uint64((uint64_t)value); +} + +inline void store_reverse_endianness_ssize_t(void* ptr, size_t value) { + *(ssize_t*)ptr = (ssize_t)_byteswap_uint64((uint64_t)value); +} + +inline void store_reverse_endianness_size_t(void* ptr, ssize_t value) { + *(size_t*)ptr = (size_t)_byteswap_uint64((uint64_t)value); +} + +inline void store_reverse_endianness_size_t(void* ptr, size_t value) { + *(size_t*)ptr = (size_t)_byteswap_uint64((uint64_t)value); +} + +inline void store_reverse_endianness_float_t(void* ptr, float_t value) { + *(uint32_t*)ptr = (uint32_t)_byteswap_ulong(*(uint32_t*)&value); +} + +inline void store_reverse_endianness_double_t(void* ptr, double_t value) { + *(uint64_t*)ptr = (uint64_t)_byteswap_uint64(*(uint64_t*)&value); +} + +inline int16_t reverse_endianness_int16_t(int16_t value) { + return (int16_t)_byteswap_ushort((uint16_t)value); +} + +inline int16_t reverse_endianness_int16_t(uint16_t value) { + return (int16_t)_byteswap_ushort(value); +} + +inline uint16_t reverse_endianness_uint16_t(int16_t value) { + return (uint16_t)_byteswap_ushort((uint16_t)value); +} + +inline uint16_t reverse_endianness_uint16_t(uint16_t value) { + return (uint16_t)_byteswap_ushort(value); +} + +inline int32_t reverse_endianness_int32_t(int32_t value) { + return (int32_t)_byteswap_ulong((uint32_t)value); +} + +inline int32_t reverse_endianness_int32_t(uint32_t value) { + return (int32_t)_byteswap_ulong(value); +} + +inline uint32_t reverse_endianness_uint32_t(int32_t value) { + return (uint32_t)_byteswap_ulong((uint32_t)value); +} + +inline uint32_t reverse_endianness_uint32_t(uint32_t value) { + return (uint32_t)_byteswap_ulong(value); +} + +inline int64_t reverse_endianness_int64_t(int64_t value) { + return (int64_t)_byteswap_uint64((uint64_t)value); +} + +inline int64_t reverse_endianness_int64_t(uint64_t value) { + return (int64_t)_byteswap_uint64(value); +} + +inline uint64_t reverse_endianness_uint64_t(int64_t value) { + return (uint64_t)_byteswap_uint64((uint64_t)value); +} + +inline uint64_t reverse_endianness_uint64_t(uint64_t value) { + return (uint64_t)_byteswap_uint64(value); +} + +inline ssize_t reverse_endianness_ssize_t(ssize_t value) { + return (ssize_t)_byteswap_uint64((uint64_t)value); +} + +inline ssize_t reverse_endianness_ssize_t(size_t value) { + return (ssize_t)_byteswap_uint64((uint64_t)value); +} + +inline size_t reverse_endianness_size_t(ssize_t value) { + return (size_t)_byteswap_uint64((uint64_t)value); +} + +inline size_t reverse_endianness_size_t(size_t value) { + return (size_t)_byteswap_uint64((uint64_t)value); +} + +inline float_t reverse_endianness_float_t(float_t value) { + uint32_t v = (uint32_t)_byteswap_ulong(*(uint32_t*)&value); + return *(float_t*)&v; +} + +inline double_t reverse_endianness_double_t(double_t value) { + uint64_t v = (uint64_t)_byteswap_uint64(*(uint64_t*)&value); + return *(double_t*)&v; +} + +inline void printf_debug(const char* fmt, ...) { +#ifdef DEBUG + va_list args; + va_start(args, fmt); + vprintf(fmt, args); + va_end(args); +#endif +} + +inline constexpr size_t utf8_length(const char* s) { + if (!s) + return 0; + + size_t len = 0; + while (*s++) + len++; + return len; +} + +inline constexpr size_t utf16_length(const wchar_t* s) { + if (!s) + return 0; + + size_t len = 0; + while (*s++) + len++; + return len; +} + +inline constexpr bool utf8_check_for_ascii_only(const char* s) { + char c = 0; + while (c = *s++) + if (c & 0x80) + return false; + return true; +} + +inline constexpr size_t utf8_to_utf16_length(const char* s) { + if (!s) + return 0; + + uint32_t c = 0; + size_t length = 0; + size_t l = 0; + + while (*s) { + char t = *s++; + if (!(t & 0x80)) { + l = 0; + length++; + c = 0; + continue; + } + else if ((t & 0xFC) == 0xF8) + continue; + else if ((t & 0xF8) == 0xF0) { + c = t & 0x07; + l = 3; + } + else if ((t & 0xF0) == 0xE0) { + c = t & 0x0F; + l = 2; + } + else if ((t & 0xE0) == 0xC0) { + c = t & 0x1F; + l = 1; + } + else if ((t & 0xC0) == 0x80) { + c = (c << 6) | (t & 0x3F); + l--; + } + + if (!l) { + if (c <= 0xD7FF || (c >= 0xE000 && c <= 0xFFFF)) + length++; + else if (c >= 0x10000 && c <= 0x10FFFF) { + length++; + length++; + } + c = 0; + } + } + return length; +} + +inline constexpr size_t utf8_to_utf16_length(const char* s, size_t length) { + if (!s || !length) + return 0; + + uint32_t c = 0; + size_t _length = 0; + size_t l = 0; + + while (*s && length) { + char t = *s++; + length--; + if (!(t & 0x80)) { + l = 0; + _length++; + c = 0; + continue; + } + else if ((t & 0xFC) == 0xF8) + continue; + else if ((t & 0xF8) == 0xF0) { + c = t & 0x07; + l = 3; + } + else if ((t & 0xF0) == 0xE0) { + c = t & 0x0F; + l = 2; + } + else if ((t & 0xE0) == 0xC0) { + c = t & 0x1F; + l = 1; + } + else if ((t & 0xC0) == 0x80) { + c = (c << 6) | (t & 0x3F); + l--; + } + + if (!l) { + if (c <= 0xD7FF || (c >= 0xE000 && c <= 0xFFFF)) + _length++; + else if (c >= 0x10000 && c <= 0x10FFFF) { + _length++; + _length++; + } + c = 0; + } + } + return _length; +} + +inline constexpr size_t utf16_to_utf8_length(const wchar_t* s) { + if (!s) + return 0; + + uint32_t c = 0; + size_t length = 0; + while (*s) { + c = *s++; + if ((c & 0xFC00) == 0xD800) { + if (!*s) + break; + + wchar_t _c = *s++; + if ((_c & 0xFC00) != 0xDC00) + continue; + + c &= 0x3FF; + c <<= 10; + c |= _c & 0x3FF; + c += 0x10000; + } + else if ((c & 0xFC00) == 0xDC00) + continue; + + if (c <= 0x7F) + length++; + else if (c <= 0x7FF) + length += 2; + else if ((c >= 0x800 && c <= 0xD7FF) || (c >= 0xE000 && c <= 0xFFFF)) + length += 3; + else if (c >= 0x10000 && c <= 0x10FFFF) + length += 4; + } + return length; +} + +inline constexpr size_t utf16_to_utf8_length(const wchar_t* s, size_t length) { + if (!s || !length) + return 0; + + uint32_t c = 0; + size_t _length = 0; + while (*s && length) { + c = *s++; + length--; + if ((c & 0xFC00) == 0xD800) { + if (!*s) + break; + + wchar_t _c = *s++; + if ((_c & 0xFC00) != 0xDC00) + continue; + + c &= 0x3FF; + c <<= 10; + c |= _c & 0x3FF; + c += 0x10000; + } + else if ((c & 0xFC00) == 0xDC00) + continue; + + if (c <= 0x7F) + _length++; + else if (c <= 0x7FF) + _length += 2; + else if ((c >= 0x800 && c <= 0xD7FF) || (c >= 0xE000 && c <= 0xFFFF)) + _length += 3; + else if (c >= 0x10000 && c <= 0x10FFFF) + _length += 4; + } + return _length; +} + +extern wchar_t* utf8_to_utf16(const char* s); +extern wchar_t* utf8_to_utf16(const char* s, size_t length); +extern char* utf16_to_utf8(const wchar_t* s); +extern char* utf16_to_utf8(const wchar_t* s, size_t length); +extern std::wstring utf8_to_utf16(const std::string& s); +extern std::string utf16_to_utf8(const std::wstring& s); diff --git a/src/KKdLib/deflate.cpp b/src/KKdLib/deflate.cpp new file mode 100644 index 0000000..24c9507 --- /dev/null +++ b/src/KKdLib/deflate.cpp @@ -0,0 +1,187 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "deflate.hpp" +#include + +namespace deflate { + static int32_t compress_static(struct libdeflate_compressor* c, const void* src, size_t src_length, + void** dst, size_t* dst_length, int32_t compression_level, mode mode); + static int32_t decompress_static(struct libdeflate_decompressor* d, const void* src, size_t src_length, + void** dst, size_t* dst_length, mode mode); + + int32_t compress(const void* src, size_t src_length, void** dst, + size_t* dst_length, int32_t compression_level, mode mode) { + if (!src_length) + return -1; + else if (!src) + return -2; + else if (!dst) + return -3; + else if (!dst_length) + return -4; + else if (mode < MODE_DEFLATE || mode > MODE_ZLIB) + return -5; + + struct libdeflate_compressor* c = libdeflate_alloc_compressor(compression_level); + if (!c) + return -5; + + int32_t result = compress_static(c, src, src_length, dst, dst_length, compression_level, mode); + libdeflate_free_compressor(c); + return result >= 0 ? result : result - 0x10; + } + + int32_t compress_gzip(const void* src, size_t src_length, void** dst, + size_t* dst_length, int32_t compression_level, const char* file_name) { + if (!src_length) + return -1; + else if (!src) + return -2; + else if (!dst) + return -3; + else if (!dst_length) + return -4; + + struct libdeflate_compressor* c = libdeflate_alloc_compressor(compression_level); + if (!c) + return -5; + + int32_t result = compress_static(c, src, src_length, dst, dst_length, compression_level, MODE_GZIP); + libdeflate_free_compressor(c); + if (result < 0) + return result - 0x10; + + size_t file_name_length = utf8_length(file_name); + void* temp = force_malloc(*dst_length + file_name_length + 1); + size_t t = (size_t)temp; + size_t d = (size_t)*dst; + memcpy((void*)t, (void*)d, 0x0A); + memcpy((void*)(t + 0x0A), file_name, file_name_length + 1); + memcpy((void*)(t + 0x0A + file_name_length + 1), (void*)(d + 0x0A), *dst_length - 0x0A); + ((uint8_t*)t)[0x03] |= 0x08; + return result; + } + + int32_t decompress(const void* src, size_t src_length, void** dst, + size_t* dst_length, mode mode) { + if (!src_length) + return -1; + else if (!src) + return -2; + else if (!dst) + return -3; + else if (!dst_length) + return -4; + else if (mode < MODE_DEFLATE || mode > MODE_ZLIB) + return -5; + + if (!*dst_length) + *dst_length = 1; + struct libdeflate_decompressor* d = libdeflate_alloc_decompressor(); + int32_t result = decompress_static(d, src, src_length, dst, dst_length, mode); + libdeflate_free_decompressor(d); + return result >= 0 ? result : result - 0x10; + } + + static int32_t compress_static(struct libdeflate_compressor* c, const void* src, size_t src_length, + void** dst, size_t* dst_length, int32_t compression_level, mode mode) { + size_t dst_max_length; + switch (mode) { + case MODE_GZIP: + dst_max_length = libdeflate_gzip_compress_bound(c, src_length); + break; + case MODE_ZLIB: + dst_max_length = libdeflate_zlib_compress_bound(c, src_length); + break; + default: + dst_max_length = libdeflate_deflate_compress_bound(c, src_length); + break; + } + + *dst = force_malloc(dst_max_length); + if (!*dst) + return -1; + + size_t dst_act_length; + switch (mode) { + case MODE_GZIP: + dst_act_length = libdeflate_gzip_compress(c, src, src_length, *dst, dst_max_length); + break; + case MODE_ZLIB: + dst_act_length = libdeflate_zlib_compress(c, src, src_length, *dst, dst_max_length); + break; + default: + dst_act_length = libdeflate_deflate_compress(c, src, src_length, *dst, dst_max_length); + break; + } + + if (dst_act_length == dst_max_length) { + *dst_length = dst_act_length; + return 0; + } + + void* temp = force_malloc(dst_act_length); + if (!temp) { + free_def(*dst); + return -2; + } + + memcpy(temp, *dst, dst_act_length); + free_def(*dst); + *dst = temp; + *dst_length = dst_act_length; + return 0; + } + + static int32_t decompress_static(struct libdeflate_decompressor* d, const void* src, size_t src_length, + void** dst, size_t* dst_length, mode mode) { + *dst = force_malloc(*dst_length); + if (!*dst) + return -1; + + size_t dst_act_length = 0; + enum libdeflate_result result; + switch (mode) { + case MODE_GZIP: + result = libdeflate_gzip_decompress(d, src, src_length, *dst, *dst_length, &dst_act_length); + break; + case MODE_ZLIB: + result = libdeflate_zlib_decompress(d, src, src_length, *dst, *dst_length, &dst_act_length); + break; + default: + result = libdeflate_deflate_decompress(d, src, src_length, *dst, *dst_length, &dst_act_length); + break; + } + + switch (result) { + case LIBDEFLATE_BAD_DATA: + free_def(*dst); + return -2; + break; + case LIBDEFLATE_INSUFFICIENT_SPACE: + free_def(*dst); + *dst_length *= 2; + decompress_static(d, src, src_length, dst, dst_length, mode); + break; + default: + if (dst_act_length >= *dst_length) + break; + + void* temp = force_malloc(dst_act_length); + if (!temp) { + free_def(*dst); + return -3; + } + + memcpy(temp, *dst, dst_act_length); + free_def(*dst); + *dst = temp; + *dst_length = dst_act_length; + break; + } + return 0; + } +} diff --git a/src/KKdLib/deflate.hpp b/src/KKdLib/deflate.hpp new file mode 100644 index 0000000..e62904b --- /dev/null +++ b/src/KKdLib/deflate.hpp @@ -0,0 +1,23 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "default.hpp" + +namespace deflate { + enum mode { + MODE_DEFLATE = 0, + MODE_GZIP = 1, + MODE_ZLIB = 2, + }; + + extern int32_t compress(const void* src, size_t src_length, void** dst, + size_t* dst_length, int32_t compression_level, mode mode); + extern int32_t compress_gzip(const void* src, size_t src_length, void** dst, + size_t* dst_length, int32_t compression_level, const char* file_name = 0); + extern int32_t decompress(const void* src, size_t src_length, void** dst, + size_t* dst_length, mode mode); +} diff --git a/src/KKdLib/divafile.cpp b/src/KKdLib/divafile.cpp new file mode 100644 index 0000000..b0217a0 --- /dev/null +++ b/src/KKdLib/divafile.cpp @@ -0,0 +1,155 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "divafile.hpp" +#include "io/file_stream.hpp" +#include "aes.hpp" +#include "str_utils.hpp" + +namespace divafile { + static const uint8_t key[] = { + 0x66, 0x69, 0x6C, 0x65, 0x20, 0x61, 0x63, 0x63, + 0x65, 0x73, 0x73, 0x20, 0x64, 0x65, 0x6E, 0x79 + }; + + void decrypt(const char* path) { + wchar_t* file_buf = utf8_to_utf16(path); + decrypt(file_buf); + free_def(file_buf); + } + + void decrypt(const wchar_t* path) { + wchar_t* file_temp = str_utils_add(path, L"_dec"); + file_stream s_enc; + s_enc.open(path, L"rb"); + if (s_enc.check_not_null()) { + uint64_t signature = s_enc.read_uint64_t(); + if (signature == 0x454C494641564944) { + uint32_t stream_length = s_enc.read_uint32_t(); + uint32_t file_length = s_enc.read_uint32_t(); + void* data = force_malloc(stream_length); + s_enc.read(data, stream_length); + + aes128_ctx ctx; + aes128_init_ctx(&ctx, key); + aes128_ecb_decrypt_buffer(&ctx, (uint8_t*)data, stream_length); + + file_stream s_dec; + s_dec.open(file_temp, L"wb"); + s_dec.write(data, min_def(file_length, stream_length)); + free_def(data); + } + } + free_def(file_temp); + } + + void decrypt(void* enc_data, void** dec_data, size_t* dec_size) { + if (!enc_data || !dec_data || !dec_size) + return; + + *dec_data = 0; + *dec_size = 0; + + size_t d = (size_t)enc_data; + + uint64_t signature = *(uint64_t*)d; + if (signature != 0x454C494641564944) + return; + + uint32_t stream_length = *(uint32_t*)(d + 8); + uint32_t file_length = *(uint32_t*)(d + 12); + void* data = force_malloc(stream_length); + memcpy(data, (void*)(d + 16), stream_length); + + aes128_ctx ctx; + aes128_init_ctx(&ctx, key); + aes128_ecb_decrypt_buffer(&ctx, (uint8_t*)data, stream_length); + + *dec_data = data; + *dec_size = file_length; + } + + void decrypt(stream& enc, memory_stream& dec) { + size_t pos = enc.get_position(); + uint64_t signature = enc.read_uint64_t(); + if (signature != 0x454C494641564944) { + std::vector data; + enc.set_position(0, SEEK_SET); + int64_t length = enc.get_length(); + data.resize(length); + enc.read(data.data(), length); + enc.set_position(pos, SEEK_SET); + dec.open(data); + return; + } + + uint32_t stream_length = enc.read_uint32_t(); + uint32_t file_length = enc.read_uint32_t(); + void* data = force_malloc(stream_length); + enc.read(data, stream_length); + + aes128_ctx ctx; + aes128_init_ctx(&ctx, key); + aes128_ecb_decrypt_buffer(&ctx, (uint8_t*)data, stream_length); + + dec.open(data, file_length); + free_def(data); + } + + void encrypt(const char* path) { + wchar_t* file_buf = utf8_to_utf16(path); + encrypt(file_buf); + free_def(file_buf); + } + + void encrypt(const wchar_t* path) { + wchar_t* file_temp = str_utils_add(path, L"_enc"); + file_stream s_dec; + s_dec.open(path, L"rb"); + if (s_dec.check_not_null()) { + size_t len = s_dec.length; + size_t len_align = align_val(len, 0x10); + + void* data = force_malloc(len_align); + s_dec.read(data, len); + + aes128_ctx ctx; + aes128_init_ctx(&ctx, key); + aes128_ecb_encrypt_buffer(&ctx, (uint8_t*)data, len_align); + + file_stream s_enc; + s_enc.open(file_temp, L"wb"); + s_enc.write_uint64_t(0x454C494641564944); + s_enc.write_uint32_t((uint32_t)len_align); + s_enc.write_uint32_t((uint32_t)len); + s_enc.write(data, len_align); + free_def(data); + } + free_def(file_temp); + } + + void encrypt(void* dec_data, size_t dec_size, void** enc_data, size_t* enc_size) { + if (!dec_data || !dec_size || !enc_data || !enc_size) + return; + + size_t len = dec_size; + size_t len_align = align_val(len, 0x10); + + void* data = force_malloc(len_align + 0x10); + size_t d = (size_t)data; + memcpy((void*)(d + 16), dec_data, len); + + aes128_ctx ctx; + aes128_init_ctx(&ctx, key); + aes128_ecb_encrypt_buffer(&ctx, (uint8_t*)(d + 16), len_align); + + *(uint64_t*)d = 0x454C494641564944; + *(uint32_t*)(d + 8) = (uint32_t)len_align; + *(uint32_t*)(d + 12) = (uint32_t)len; + + *enc_data = data; + *enc_size = len_align + 0x10; + } +} diff --git a/src/KKdLib/divafile.hpp b/src/KKdLib/divafile.hpp new file mode 100644 index 0000000..005df9b --- /dev/null +++ b/src/KKdLib/divafile.hpp @@ -0,0 +1,19 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "default.hpp" +#include "io/memory_stream.hpp" + +namespace divafile { + extern void decrypt(const char* path); + extern void decrypt(const wchar_t* path); + extern void decrypt(void* enc_data, void** dec_data, size_t* dec_size); + extern void decrypt(stream& enc, memory_stream& dec); + extern void encrypt(const char* path); + extern void encrypt(const wchar_t* path); + extern void encrypt(void* dec_data, size_t dec_size, void** enc_data, size_t* enc_size); +} diff --git a/src/KKdLib/f2/enrs.cpp b/src/KKdLib/f2/enrs.cpp new file mode 100644 index 0000000..31dacc5 --- /dev/null +++ b/src/KKdLib/f2/enrs.cpp @@ -0,0 +1,286 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "enrs.hpp" + +enum enrs_value_type { + ENRS_VALUE_INT8 = 0x0, + ENRS_VALUE_INT16 = 0x1, + ENRS_VALUE_INT32 = 0x2, + ENRS_VALUE_INVALID = 0x3, +}; + +inline static bool enrs_length_get_size_type(uint32_t* length, size_t val); +inline static bool enrs_length_get_size(uint32_t* length, size_t val); +static bool enrs_read_packed_value(stream& s, uint32_t* val); +static bool enrs_write_packed_value(stream& s, uint32_t val); +static bool enrs_read_packed_value_type(stream& s, uint32_t* val, enrs_type* type); +static bool enrs_write_packed_value_type(stream& s, uint32_t val, enrs_type type); + +enrs_entry::enrs_entry(): offset(), count(), size(), repeat_count() { + +} + +enrs_entry::enrs_entry(uint32_t offset, uint32_t count, uint32_t size, uint32_t repeat_count) { + this->offset = offset; + this->count = count; + this->size = size; + this->repeat_count = repeat_count; +} + +enrs_entry::~enrs_entry() { + +} + +void enrs_entry::append(uint32_t skip_bytes, uint32_t repeat_count, enrs_type type) { + sub.push_back({ skip_bytes, repeat_count, type }); +} + +void enrs_entry::append(enrs_sub_entry&& data) { + sub.push_back(data); +} + +enrs::enrs() { + +} + +enrs::~enrs() { + +} + +void enrs::apply(void* data) { + if (!data) + return; + + uint8_t* d = (uint8_t*)data; + uint8_t* temp; + for (enrs_entry& i : vec) { + d += i.offset; + for (size_t j = 0; j < i.repeat_count; j++) { + temp = d + i.size * j; + for (enrs_sub_entry& k : i.sub) { + temp += k.skip_bytes; + switch (k.type) { + case ENRS_WORD: + for (size_t l = k.repeat_count; l; l--) { + *(uint16_t*)temp = reverse_endianness_uint16_t(*(uint16_t*)temp); + temp += 2; + } + break; + case ENRS_DWORD: + for (size_t l = k.repeat_count; l; l--) { + *(uint32_t*)temp = reverse_endianness_uint32_t(*(uint32_t*)temp); + temp += 4; + } + break; + case ENRS_QWORD: + for (size_t l = k.repeat_count; l; l--) { + *(uint64_t*)temp = reverse_endianness_uint64_t(*(uint64_t*)temp); + temp += 8; + } + break; + } + } + } + } +} + +uint32_t enrs::length() { + uint32_t l = 0x10; + uint32_t o = 0; + for (enrs_entry& i : vec) { + uint32_t offset = i.offset; + i.count = (uint32_t)i.sub.size(); + if (&i != vec.data() && (&i)[-1].count < 1) { + o += (uint32_t)((size_t)(&i)[-1].size * (&i)[-1].repeat_count); + if (i.count > 0) { + offset += o; + o = 0; + } + } + + if (i.count < 1) + continue; + + if (enrs_length_get_size(&l, offset) + || enrs_length_get_size(&l, i.count) + || enrs_length_get_size(&l, i.size) + || enrs_length_get_size(&l, i.repeat_count)) + goto End; + + if (i.repeat_count < 1 || i.count > 0x40000000) + continue; + + for (enrs_sub_entry& j : i.sub) { + if (enrs_length_get_size_type(&l, j.skip_bytes) + || enrs_length_get_size(&l, j.repeat_count)) + goto End; + } + } +End: + l = align_val(l, 0x10); + return l; +} + +void enrs::read(stream& s) { + vec.clear(); + + s.read_uint32_t(); + size_t l = s.read_uint32_t(); + s.read_uint32_t(); + s.read_uint32_t(); + + vec.reserve(l); + for (size_t i = 0; i < l; i++) { + enrs_entry entry; + if (enrs_read_packed_value(s, &entry.offset) + || enrs_read_packed_value(s, &entry.count) + || enrs_read_packed_value(s, &entry.size) + || enrs_read_packed_value(s, &entry.repeat_count)) + return; + + if (!entry.count || !entry.repeat_count) { + vec.push_back(entry); + continue; + } + + entry.sub.reserve(entry.count); + for (size_t j = 0; j < entry.count; j++) { + enrs_sub_entry sub_entry = {}; + if (enrs_read_packed_value_type(s, &sub_entry.skip_bytes, &sub_entry.type) + || enrs_read_packed_value(s, &sub_entry.repeat_count)) + return; + entry.sub.push_back(sub_entry); + } + vec.push_back(entry); + } +} + +void enrs::write(stream& s) { + uint32_t o = 0; + size_t length = enrs::length(); + s.write_uint32_t(0); + s.write_uint32_t((uint32_t)vec.size()); + s.write_uint32_t(0); + s.write_uint32_t(0); + for (enrs_entry& i : vec) { + uint32_t offset = i.offset; + i.count = (uint32_t)i.sub.size(); + if (&i != vec.data() && (&i)[-1].count < 1) { + o += (uint32_t)((size_t)(&i)[-1].size * (&i)[-1].repeat_count); + if ((&i)->count > 0) { + offset += o; + o = 0; + } + } + + if (i.count < 1) + continue; + + if (enrs_write_packed_value(s, offset) + || enrs_write_packed_value(s, i.count) + || enrs_write_packed_value(s, i.size) + || enrs_write_packed_value(s, i.repeat_count)) + goto End; + + if (i.repeat_count < 1) + continue; + + for (enrs_sub_entry& j : i.sub) + if (enrs_write_packed_value_type(s, j.skip_bytes, j.type) + || enrs_write_packed_value(s, j.repeat_count)) + goto End; + } + +End: + s.align_write(0x10); +} + +inline static bool enrs_length_get_size_type(uint32_t* length, size_t val) { + *length += val < 0x10 ? 1 : val < 0x1000 ? 2 : val < 0x10000000 ? 4 : 1; + return val >= 0x10000000; +} + +inline static bool enrs_length_get_size(uint32_t* length, size_t val) { + if (!length) + return true; + + *length += val < 0x40 ? 1 : val < 0x4000 ? 2 : val < 0x40000000 ? 4 : 1; + return val >= 0x40000000; +} + +static bool enrs_read_packed_value(stream& s, uint32_t* val) { + *val = s.read_uint8_t(); + enrs_value_type value = (enrs_value_type)((*val >> 6) & 0x3); + *val &= 0x3F; + + if (value == ENRS_VALUE_INT32) + *val = (((((*val << 8) | s.read_uint8_t()) << 8) | s.read_uint8_t()) << 8) | s.read_uint8_t(); + else if (value == ENRS_VALUE_INT16) + *val = (*val << 8) | s.read_uint8_t(); + else if (value == ENRS_VALUE_INVALID) { + *val = 0; + return true; + } + return false; +} + +static bool enrs_write_packed_value(stream& s, uint32_t val) { + if (val < 0x40) + s.write_uint8_t((uint8_t)((ENRS_VALUE_INT8 << 6) | (val & 0x3F))); + else if (val < 0x4000) { + s.write_uint8_t((uint8_t)((ENRS_VALUE_INT16 << 6) | ((val >> 8) & 0x3F))); + s.write_uint8_t((uint8_t)val); + } + else if (val < 0x40000000) { + s.write_uint8_t((uint8_t)((ENRS_VALUE_INT32 << 6) | ((val >> 24) & 0x3F))); + s.write_uint8_t((uint8_t)(val >> 16)); + s.write_uint8_t((uint8_t)(val >> 8)); + s.write_uint8_t((uint8_t)val); + } + else { + s.write_uint8_t(ENRS_VALUE_INVALID << 6); + return true; + } + return false; +} + +static bool enrs_read_packed_value_type(stream& s, uint32_t* val, enrs_type* type) { + *val = s.read_uint8_t(); + enrs_value_type value = (enrs_value_type)((*val >> 6) & 0x3); + *type = (enrs_type)((*val >> 4) & 0x3); + *val &= 0xF; + + if (value == ENRS_VALUE_INT32) + *val = (((((*val << 8) | s.read_uint8_t()) << 8) | s.read_uint8_t()) << 8) | s.read_uint8_t(); + else if (value == ENRS_VALUE_INT16) + *val = (*val << 8) | s.read_uint8_t(); + else if (value == ENRS_VALUE_INVALID) { + *val = 0; + return true; + } + return false; +} + +static bool enrs_write_packed_value_type(stream& s, uint32_t val, enrs_type type) { + uint8_t t = ((uint8_t)type & 0x3) << 4; + if (val < 0x10) + s.write_uint8_t((uint8_t)((ENRS_VALUE_INT8 << 6) | t | (val & 0xF))); + else if (val < 0x1000) { + s.write_uint8_t((uint8_t)((ENRS_VALUE_INT16 << 6) | t | ((val >> 8) & 0xF))); + s.write_uint8_t((uint8_t)val); + } + else if (val < 0x10000000) { + s.write_uint8_t((uint8_t)((ENRS_VALUE_INT32 << 6) | t | ((val >> 24) & 0xF))); + s.write_uint8_t((uint8_t)(val >> 16)); + s.write_uint8_t((uint8_t)(val >> 8)); + s.write_uint8_t((uint8_t)val); + } + else { + s.write_uint8_t(ENRS_VALUE_INVALID << 6); + return true; + } + return false; +} diff --git a/src/KKdLib/f2/enrs.hpp b/src/KKdLib/f2/enrs.hpp new file mode 100644 index 0000000..aed551c --- /dev/null +++ b/src/KKdLib/f2/enrs.hpp @@ -0,0 +1,50 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include +#include "../default.hpp" +#include "../io/stream.hpp" + +enum enrs_type { + ENRS_WORD = 0x0, + ENRS_DWORD = 0x1, + ENRS_QWORD = 0x2, + ENRS_INVALID = 0x3, +}; + +struct enrs_sub_entry { + uint32_t skip_bytes; + uint32_t repeat_count; + enrs_type type; +}; + +struct enrs_entry { + uint32_t offset; + uint32_t count; + uint32_t size; + uint32_t repeat_count; + std::vector sub; + + enrs_entry(); + enrs_entry(uint32_t offset, uint32_t count, uint32_t size, uint32_t repeat_count); + ~enrs_entry(); + + void append(uint32_t skip_bytes, uint32_t repeat_count, enrs_type type); + void append(enrs_sub_entry&& data); +}; + +struct enrs { + std::vector vec; + + enrs(); + ~enrs(); + + void apply(void* data); + uint32_t length(); + void read(stream& s); + void write(stream& s); +}; diff --git a/src/KKdLib/f2/header.cpp b/src/KKdLib/f2/header.cpp new file mode 100644 index 0000000..36462c5 --- /dev/null +++ b/src/KKdLib/f2/header.cpp @@ -0,0 +1,37 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "header.hpp" + +void f2_header_read(stream& s, f2_header* h) { + memset(h, 0, sizeof(f2_header)); + + if (s.check_null()) + return; + + s.read(h, 0x20); + if (h->length == 0x40) + s.read((uint8_t*)h + 0x20, 0x20); +} + +void f2_header_write(stream& s, f2_header* h, bool extended) { + if (s.check_null()) + return; + + h->length = extended ? 0x40 : 0x20; + s.write(h, h->length); +} + +void f2_header_write_end_of_container(stream& s, uint32_t depth) { + if (s.check_null()) + return; + + f2_header h = {}; + h.signature = reverse_endianness_uint32_t('EOFC'); + h.length = 0x20; + h.depth = depth; + h.use_section_size = true; + f2_header_write(s, &h, false); +} diff --git a/src/KKdLib/f2/header.hpp b/src/KKdLib/f2/header.hpp new file mode 100644 index 0000000..a955252 --- /dev/null +++ b/src/KKdLib/f2/header.hpp @@ -0,0 +1,39 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "../default.hpp" +#include "../io/stream.hpp" + +struct f2_header { + union { + char signature_char[0x04]; // 0x00 + uint32_t signature; // 0x00 + }; + uint32_t data_size; // 0x04 + uint32_t length; // 0x08 + union { // 0x0C + struct { + uint32_t flags0 : 27; + uint32_t use_big_endian : 1; + uint32_t use_section_size : 1; + uint32_t flags1 : 3; + }; + uint32_t flags; + }; + uint32_t depth; // 0x10 + uint32_t section_size; // 0x14 + uint32_t version; // 0x18 + uint32_t unknown0; // 0x1C + uint32_t murmurhash; // 0x20 + uint32_t unknown1[3]; // 0x24 + uint32_t inner_signature; // 0x30 + uint32_t unknown2[3]; // 0x34 +}; + +extern void f2_header_read(stream& s, f2_header* h); +extern void f2_header_write(stream& s, f2_header* h, bool extended); +extern void f2_header_write_end_of_container(stream& s, uint32_t depth); \ No newline at end of file diff --git a/src/KKdLib/f2/pof.cpp b/src/KKdLib/f2/pof.cpp new file mode 100644 index 0000000..700e11c --- /dev/null +++ b/src/KKdLib/f2/pof.cpp @@ -0,0 +1,206 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "pof.hpp" + +enum pof_value_type { + POF_VALUE_INVALID = 0x0, + POF_VALUE_INT8 = 0x1, + POF_VALUE_INT16 = 0x2, + POF_VALUE_INT32 = 0x3, +}; + +inline static bool pof_length_get_size(uint32_t* length, size_t val); +static size_t pof_read_offsets_count(stream& s); +inline static bool pof_write_packed_value(stream& s, size_t val); + +pof::pof() : shift_x() { + +} + +pof::~pof() { + +} + +void pof::add(stream& s, int64_t offset) { + vec.push_back(s.get_position() + offset); +} + +void pof::read(stream& s) { + vec.clear(); + + size_t length = pof_read_offsets_count(s); + + uint8_t bit_shift = (uint8_t)(shift_x ? 3 : 2); + size_t l = s.read_uint32_t() - 4ULL; + + vec.reserve(length); + + size_t i = 0; + size_t j = 0; + size_t offset = 0; + while (i < l) { + size_t v = s.read_uint8_t(); + pof_value_type value = (pof_value_type)((v >> 6) & 0x03); + v &= 0x3F; + + if (value == POF_VALUE_INT32) { + v = (((((v << 8) | s.read_uint8_t()) << 8) | s.read_uint8_t()) << 8) | s.read_uint8_t(); + i += 3; + } + else if (value == POF_VALUE_INT16) { + v = (v << 8) | s.read_uint8_t(); + i++; + } + else if (value == POF_VALUE_INVALID) + break; + + offset += v; + vec.push_back(offset << bit_shift); + i++; + j++; + } +} + +void pof::write(stream& s) { + size_t j = 0; + size_t o = 0; + uint8_t bit_shift = (uint8_t)(shift_x ? 3 : 2); + size_t v = ((size_t)1 << bit_shift) - 1; + size_t l = length(); + if (shift_x) + s.write_uint32_t((uint32_t)l); + else + s.write_uint32_t((uint32_t)align_val(l, 4)); + + for (int64_t& i : vec) { + o = i; + if (o & v) { + pof_write_packed_value(s, 0x7FFFFFFF); + break; + } + + size_t k = o - j; + if (&i != vec.data() && !k) + continue; + j = o; + o = k; + + pof_write_packed_value(s, o >> bit_shift); + } + + size_t pos = s.get_position(); + for (size_t c = align_val(pos, 0x10) - pos; c > 0; c--) + s.write_uint8_t(0); +} + +uint32_t pof::length() { + uint32_t l = 4; + size_t j = 0; + uint8_t bit_shift = (uint8_t)(shift_x ? 3 : 2); + size_t v = ((size_t)1 << bit_shift) - 1; + + for (int64_t& i : vec) { + size_t o = i; + if (o & v) + break; + else if (&i != vec.data()) { + size_t k = o - j; + if (!k) + continue; + j = o; + o = k; + } + else + j = o; + + if (pof_length_get_size(&l, o >> bit_shift)) + break; + } + return l; +} + +inline void io_write_offset_pof_add(stream& s, int64_t val, + int32_t offset, bool is_x, pof* pof) { + if (!is_x) { + if (val) + val += offset; + pof->add(s, offset); + s.write_int32_t_reverse_endianness((int32_t)val); + } + else { + s.align_write(0x08); + pof->add(s, 0); + s.write_int64_t_reverse_endianness(val); + } +} + +inline void io_write_offset_f2_pof_add(stream& s, int64_t val, + int32_t offset, pof* pof) { + if (val) + val += offset; + pof->add(s, offset); + s.write_int32_t_reverse_endianness((int32_t)val); +} + +inline void io_write_offset_x_pof_add(stream& s, int64_t val, pof* pof) { + s.align_write(0x08); + pof->add(s, 0); + s.write_int64_t_reverse_endianness(val); +} + +inline static bool pof_length_get_size(uint32_t* length, size_t val) { + *length += val < 0x40 ? 1 : val < 0x4000 ? 2 : val < 0x40000000 ? 4 : 1; + return val >= 0x40000000; +} + +static size_t pof_read_offsets_count(stream& s) { + size_t pos = s.get_position(); + size_t i, j, l; + pof_value_type val; + + l = s.read_uint32_t() - 4ULL; + i = 0; + j = 0; + while (i < l) { + val = (pof_value_type)((s.read_uint8_t() >> 6) & 0x03); + if (val == POF_VALUE_INT32) { + s.read_uint8_t(); + s.read_uint8_t(); + s.read_uint8_t(); + i += 3; + } + else if (val == POF_VALUE_INT16) { + s.read_uint8_t(); + i++; + } + else if (val != POF_VALUE_INT8) + break; + i++; + j++; + } + s.set_position( pos, SEEK_SET); + return j; +} + +static bool pof_write_packed_value(stream& s, size_t val) { + if (val < 0x40) + s.write_uint8_t((uint8_t)((POF_VALUE_INT8 << 6) | (val & 0x3F))); + else if (val < 0x4000) { + s.write_uint8_t((uint8_t)((POF_VALUE_INT16 << 6) | ((val >> 8) & 0x3F))); + s.write_uint8_t((uint8_t)val); + } + else if (val < 0x40000000) { + s.write_uint8_t((uint8_t)((POF_VALUE_INT32 << 6) | ((val >> 24) & 0x3F))); + s.write_uint8_t((uint8_t)(val >> 16)); + s.write_uint8_t((uint8_t)(val >> 8)); + s.write_uint8_t((uint8_t)val); + } + else { + s.write_uint8_t(POF_VALUE_INVALID << 6); + return true; + } + return false; +} diff --git a/src/KKdLib/f2/pof.hpp b/src/KKdLib/f2/pof.hpp new file mode 100644 index 0000000..2985ce5 --- /dev/null +++ b/src/KKdLib/f2/pof.hpp @@ -0,0 +1,29 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include +#include "../default.hpp" +#include "../io/stream.hpp" + +struct pof { + std::vector vec; + bool shift_x; + + pof(); + ~pof(); + + void add(stream& s, int64_t offset); + void read(stream& s); + void write(stream& s); + uint32_t length(); +}; + +extern void io_write_offset_pof_add(stream& s, int64_t val, + int32_t offset, bool is_x, pof* pof); +extern void io_write_offset_f2_pof_add(stream& s, int64_t val, + int32_t offset, pof* pof); +extern void io_write_offset_x_pof_add(stream& s, int64_t val, pof* pof); diff --git a/src/KKdLib/f2/struct.cpp b/src/KKdLib/f2/struct.cpp new file mode 100644 index 0000000..4727597 --- /dev/null +++ b/src/KKdLib/f2/struct.cpp @@ -0,0 +1,245 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "struct.hpp" +#include "../io/file_stream.hpp" +#include "../io/memory_stream.hpp" +#include "../io/path.hpp" +#include "../divafile.hpp" + +static void f2_struct_read_data(stream& s, f2_struct* st, f2_header* h); +static void f2_struct_write_inner(stream& s, f2_struct* st, uint32_t depth, bool use_depth, bool shift_x); +static void f2_struct_get_length(f2_struct* s, bool shift_x); +static void f2_struct_write_pof(stream& s, pof* pof, uint32_t depth, bool shift_x); +static void f2_struct_write_enrs(stream& s, enrs* enrs, uint32_t depth); + +f2_struct::f2_struct() : header() { + +} + +f2_struct::~f2_struct() { + +} + +void f2_struct::read(const char* path) { + if (!path) + return; + + file_stream s; + s.open(path, "rb"); + if (s.check_not_null()) { + memory_stream ms; + divafile::decrypt(s, ms); + + f2_header h; + f2_header_read(ms, &h); + f2_struct_read_data(ms, this, &h); + } +} + +void f2_struct::read(const wchar_t* path) { + if (!path) + return; + + file_stream s; + s.open(path, L"rb"); + if (s.check_not_null()) { + memory_stream ms; + divafile::decrypt(s, ms); + + f2_header h; + f2_header_read(ms, &h); + f2_struct_read_data(ms, this, &h); + } +} + +void f2_struct::read(const void* data, size_t size) { + if (!data || !size) + return; + + memory_stream s; + s.open(data, size); + if (s.check_not_null()) { + memory_stream ms; + divafile::decrypt(s, ms); + + f2_header h; + f2_header_read(ms, &h); + f2_struct_read_data(ms, this, &h); + } +} + +void f2_struct::read(stream& s) { + if (s.check_null()) + return; + + memory_stream ms; + divafile::decrypt(s, ms); + + f2_header h; + f2_header_read(ms, &h); + f2_struct_read_data(ms, this, &h); +} + +void f2_struct::write(const char* path, bool use_depth, bool shift_x) { + if (!path) + return; + + file_stream s; + s.open(path, "wb"); + if (s.check_not_null()) { + f2_struct_get_length(this, shift_x); + f2_struct_write_inner(s, this, 0, use_depth, shift_x); + } +} + +void f2_struct::write(const wchar_t* path, bool use_depth, bool shift_x) { + if (!path) + return; + + file_stream s; + s.open(path, L"wb"); + if (s.check_not_null()) { + f2_struct_get_length(this, shift_x); + f2_struct_write_inner(s, this, 0, use_depth, shift_x); + } +} + +void f2_struct::write(void** data, size_t* size, bool use_depth, bool shift_x) { + if (!data || !size) + return; + + f2_struct_get_length(this, shift_x); + memory_stream s; + s.open(0, header.data_size + 0x40ULL); + f2_struct_write_inner(s, this, 0, use_depth, shift_x); + + s.copy(data, size); +} + +void f2_struct::write(stream& s, bool use_depth, bool shift_x) { + if (s.check_null()) + return; + + f2_struct_get_length(this, shift_x); + f2_struct_write_inner(s, this, 0, use_depth, shift_x); + +} + +static void f2_struct_get_length(f2_struct* s, bool shift_x) { + bool has_pof = s->pof.vec.size() > 0 ? true : false; + bool has_enrs = s->enrs.vec.size() > 0 ? true : false; + bool has_sub_structs = s->sub_structs.size() > 0 ? true : false; + + s->header.section_size = (uint32_t)s->data.size(); + + uint32_t l = s->header.section_size; + if (has_enrs) { + uint32_t len = s->enrs.length(); + l += 0x20 + align_val(len, 0x10); + } + + if (has_pof) { + s->pof.shift_x = shift_x; + uint32_t len = s->pof.length(); + l += 0x20 + align_val(len, 0x10); + } + + if (has_sub_structs) + for (f2_struct& i : s->sub_structs) { + f2_struct_get_length(&i, shift_x); + l += i.header.data_size; + l += i.header.length; + } + + if (has_enrs || has_pof || has_sub_structs) + l += 0x20; + + s->header.data_size = l; +} + +static void f2_struct_read_data(stream& s, f2_struct* st, f2_header* h) { + uint32_t l = h->use_section_size ? h->section_size : h->data_size; + uint32_t depth = h->depth; + st->header = *h; + if (l) { + st->data.resize(l); + s.read(st->data.data(), l); + } + + uint32_t sig; + size_t length = (size_t)h->data_size - l; + size_t position = 0; + while (length > position) { + f2_header_read(s, h); + sig = reverse_endianness_uint32_t(h->signature); + l = h->use_section_size ? h->section_size : h->data_size; + position += (size_t)h->length + l; + if (sig == 'EOFC') + break; + else if (sig == 'ENRS') { + size_t pos = s.get_position(); + st->enrs.read(s); + s.set_position(pos + l, SEEK_SET); + } + else if ((sig & 0xFFFFFFF0) == 'POF0') { + size_t pos = s.get_position(); + st->pof.shift_x = sig == 'POF1'; + st->pof.read(s); + s.set_position(pos + l, SEEK_SET); + } + else { + st->sub_structs.push_back({}); + f2_struct_read_data(s, &st->sub_structs.back(), h); + } + } +} + +static void f2_struct_write_inner(stream& s, f2_struct* st, uint32_t depth, bool use_depth, bool shift_x) { + bool has_pof = st->pof.vec.size() > 0 ? true : false; + bool has_enrs = st->enrs.vec.size() > 0 ? true : false; + bool has_sub_structs = st->sub_structs.size() > 0 ? true : false; + + st->header.depth = use_depth ? depth : 0; + f2_header_write(s, &st->header, st->header.length == 0x40); + if (st->data.size()) + s.write(st->data.data(), st->data.size()); + if (has_enrs) + f2_struct_write_enrs(s, &st->enrs, use_depth ? depth + 1 : 0); + if (has_pof) + f2_struct_write_pof(s, &st->pof, use_depth ? depth + 1 : 0, shift_x); + if (has_sub_structs) + for (f2_struct& i : st->sub_structs) + f2_struct_write_inner(s, &i, depth + 1, use_depth, shift_x); + if (has_enrs || has_pof || has_sub_structs) + f2_header_write_end_of_container(s, use_depth ? depth + 1 : 0); + if (!depth) + f2_header_write_end_of_container(s, 0); +} + +static void f2_struct_write_pof(stream& s, pof* pof, uint32_t depth, bool shift_x) { + pof->shift_x = shift_x; + size_t len = pof->length(); + f2_header h = {}; + h.signature = shift_x ? reverse_endianness_uint32_t('POF1') : reverse_endianness_uint32_t('POF0'); + h.length = 0x20; + h.depth = depth; + h.use_section_size = true; + h.data_size = h.section_size = (uint32_t)align_val(len, 0x10); + f2_header_write(s, &h, false); + pof->write(s); +} + +static void f2_struct_write_enrs(stream& s, enrs* enrs, uint32_t depth) { + size_t len = enrs->length(); + f2_header h = {}; + h.signature = reverse_endianness_uint32_t('ENRS'); + h.length = 0x20; + h.depth = depth; + h.use_section_size = true; + h.data_size = h.section_size = (uint32_t)align_val(len, 0x10); + f2_header_write(s, &h, false); + enrs->write(s); +} diff --git a/src/KKdLib/f2/struct.hpp b/src/KKdLib/f2/struct.hpp new file mode 100644 index 0000000..d60f406 --- /dev/null +++ b/src/KKdLib/f2/struct.hpp @@ -0,0 +1,34 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "../default.hpp" +#include "enrs.hpp" +#include "header.hpp" +#include "pof.hpp" + +struct f2_struct; + +struct f2_struct { + f2_header header; + std::vector data; + std::vector sub_structs; + + enrs enrs; + pof pof; + + f2_struct(); + ~f2_struct(); + + void read(const char* path); + void read(const wchar_t* path); + void read(const void* data, size_t size); + void read(stream& s); + void write(const char* path, bool use_depth = true, bool shift_x = false); + void write(const wchar_t* path, bool use_depth = true, bool shift_x = false); + void write(void** data, size_t* size, bool use_depth = true, bool shift_x = false); + void write(stream& s, bool use_depth = true, bool shift_x = false); +}; diff --git a/src/KKdLib/farc.cpp b/src/KKdLib/farc.cpp new file mode 100644 index 0000000..690b3e8 --- /dev/null +++ b/src/KKdLib/farc.cpp @@ -0,0 +1,950 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "farc.hpp" +#include "io/path.hpp" +#include "io/file_stream.hpp" +#include "io/memory_stream.hpp" +#include "aes.hpp" +#include "deflate.hpp" +#include "hash.hpp" +#include "str_utils.hpp" + +static const uint8_t key[] = { + 0x70, 0x72, 0x6F, 0x6A, 0x65, 0x63, 0x74, 0x5F, + 0x64, 0x69, 0x76, 0x61, 0x2E, 0x62, 0x69, 0x6E, +}; + +static const uint8_t key_ft[] = { + 0x13, 0x72, 0xD5, 0x7B, 0x6E, 0x9E, 0x31, 0xEB, + 0xA2, 0x39, 0xB8, 0x3C, 0x15, 0x57, 0xC6, 0xBB, +}; + +static errno_t farc_get_files(farc* f); +static void farc_pack_files(farc* f, stream& s, farc_signature signature, farc_flags flags, bool get_files = false); +static errno_t farc_read_header(farc* f, stream& s); +static void farc_unpack_files(farc* f, stream& s, bool save); +static void farc_unpack_file(farc* f, farc_file* ff); +static void farc_unpack_file(farc* f, stream& s, farc_file* ff, + bool save = false, char* temp_path = 0, size_t dir_len = 0); +static void farc_write_padding(farc* f, stream& s, size_t size, bool x = false); + +farc::farc() : flags(), ft() { + signature = FARC_FArC; + compression_level = 12; + alignment = 0x10; +} + +farc::~farc() { + +} + +farc_file* farc::add_file(const char* name) { + if (!name) + return 0; + + uint64_t name_hash = hash_utf8_fnv1a64m(name, true); + for (farc_file& i : files) + if (hash_string_fnv1a64m(i.name, true) == name_hash) + return &i; + + size_t files_count = files.size(); + void** data_temp = new void* [files_count]; + void** data_comp_temp = new void* [files_count]; + if (!data_temp || !data_comp_temp) { + if (data_temp) + delete[] data_temp; + if (data_comp_temp) + delete[] data_comp_temp; + return 0; + } + + for (size_t i = 0; i < files_count; i++) { + data_temp[i] = files[i].data; + data_comp_temp[i] = files[i].data_compressed; + files[i].data = 0; + files[i].data_compressed = 0; + } + + files.push_back({}); + files.back().name.assign(name); + + for (size_t i = 0; i < files_count; i++) { + files[i].data = data_temp[i]; + files[i].data_compressed = data_comp_temp[i]; + data_temp[i] = 0; + data_comp_temp[i] = 0; + } + + delete[] data_temp; + delete[] data_comp_temp; + + return &files.back(); +} + +farc_file* farc::add_file(const wchar_t* name) { + if (!name) + return 0; + + uint64_t name_hash = hash_utf16_fnv1a64m(name, true); + for (farc_file& i : files) + if (hash_string_fnv1a64m(i.name, true) == name_hash) + return &i; + + size_t files_count = files.size(); + void** data_temp = new void* [files_count]; + void** data_comp_temp = new void* [files_count]; + if (!data_temp || !data_comp_temp) { + if (data_temp) + delete[] data_temp; + if (data_comp_temp) + delete[] data_comp_temp; + } + + for (size_t i = 0; i < files_count; i++) { + data_temp[i] = files[i].data; + data_comp_temp[i] = files[i].data_compressed; + files[i].data = 0; + files[i].data_compressed = 0; + } + + files.push_back({}); + if (name) { + char* name_temp = utf16_to_utf8(name); + files.back().name.assign(name_temp); + free_def(name_temp); + } + + for (size_t i = 0; i < files_count; i++) { + files[i].data = data_temp[i]; + files[i].data_compressed = data_comp_temp[i]; + data_temp[i] = 0; + data_comp_temp[i] = 0; + } + + delete[] data_temp; + delete[] data_comp_temp; + + return &files.back(); +} + +const char* farc::get_file_name(uint32_t hash) { + if (!hash || hash == hash_murmurhash_empty) + return 0; + + for (farc_file& i : files) { + const char* l_str = i.name.c_str(); + const char* t = strrchr(l_str, '.'); + size_t l_len = i.name.size(); + if (t) + l_len = t - l_str; + + if (hash_murmurhash(l_str, l_len) == hash) + return l_str; + } + return 0; +} + +size_t farc::get_file_size(const char* name) { + if (!name) + return 0; + + uint64_t name_hash = hash_utf8_fnv1a64m(name, true); + for (farc_file& i : files) + if (hash_string_fnv1a64m(i.name, true) == name_hash) + return i.size; + return 0; +} + +size_t farc::get_file_size(const wchar_t* name) { + if (!name) + return 0; + + uint64_t name_hash = hash_utf16_fnv1a64m(name, true); + for (farc_file& i : files) + if (hash_string_fnv1a64m(i.name, true) == name_hash) + return i.size; + return 0; +} + +size_t farc::get_file_size(uint32_t hash) { + if (!hash || hash == hash_murmurhash_empty) + return 0; + + for (farc_file& i : files) { + const char* l_str = i.name.c_str(); + const char* t = strrchr(l_str, '.'); + size_t l_len = i.name.size(); + if (t) + l_len = t - l_str; + + if (hash_murmurhash(l_str, l_len) == hash) + return i.size; + } + return 0; +} + +bool farc::has_file(const char* name) { + if (!name) + return false; + + uint64_t name_hash = hash_utf8_fnv1a64m(name, true); + for (farc_file& i : files) + if (hash_string_fnv1a64m(i.name, true) == name_hash) + return true; + return false; +} + +bool farc::has_file(const wchar_t* name) { + if (!name) + return false; + + uint64_t name_hash = hash_utf16_fnv1a64m(name, true); + for (farc_file& i : files) + if (hash_string_fnv1a64m(i.name, true) == name_hash) + return true; + return false; +} + +bool farc::has_file(uint32_t hash) { + if (!hash || hash == hash_murmurhash_empty) + return false; + + for (farc_file& i : files) { + const char* l_str = i.name.c_str(); + const char* t = strrchr(l_str, '.'); + size_t l_len = i.name.size(); + if (t) + l_len = t - l_str; + + if (hash_murmurhash(l_str, l_len) == hash) + return true; + } + return false; +} + +void farc::read(const char* path, bool unpack, bool save) { + if (!path) + return; + + wchar_t* path_buf = utf8_to_utf16(path); + read(path_buf, unpack, save); + free_def(path_buf); +} + +void farc::read(const wchar_t* path, bool unpack, bool save) { + if (!path) + return; + + files.clear(); + + wchar_t full_path_buf[MAX_PATH]; + wchar_t* full_path = _wfullpath(full_path_buf, path, MAX_PATH); + + if (!full_path) + return; + else if (!path_check_file_exists(full_path_buf)) + return; + + char* dir_temp = utf16_to_utf8(full_path_buf); + size_t dir_temp_len = utf8_length(dir_temp); + file_path.assign(dir_temp, dir_temp_len); + directory_path.assign(dir_temp, dir_temp_len); + free_def(dir_temp); + + const char* dot = strrchr(directory_path.c_str(), '.'); + if (dot) + directory_path = directory_path.substr(0, dot - directory_path.c_str()); + + file_stream s; + s.open(file_path.c_str(), "rb"); + if (s.check_not_null() && !farc_read_header(this, s) && unpack) + farc_unpack_files(this, s, save); +} + +void farc::read(const void* data, size_t size, bool unpack) { + if (!data || !size) + return; + + files.clear(); + file_path.clear(); + + directory_path.clear(); + + memory_stream s; + s.open(data, size); + if (!farc_read_header(this, s) && unpack) + farc_unpack_files(this, s, false); +} + +farc_file* farc::read_file(const char* name) { + if (!name) + return 0; + + uint64_t name_hash = hash_utf8_fnv1a64m(name, true); + for (farc_file& i : files) + if (hash_string_fnv1a64m(i.name, true) == name_hash) { + farc_unpack_file(this, &i); + return &i; + } + return 0; +} + +farc_file* farc::read_file(const wchar_t* name) { + if (!name) + return 0; + + uint64_t name_hash = hash_utf16_fnv1a64m(name, true); + for (farc_file& i : files) + if (hash_string_fnv1a64m(i.name, true) == name_hash) { + farc_unpack_file(this, &i); + return &i; + } + return 0; +} + +farc_file* farc::read_file(uint32_t hash) { + if (!hash || hash == hash_murmurhash_empty) + return 0; + + for (farc_file& i : files) { + const char* l_str = i.name.c_str(); + const char* t = strrchr(l_str, '.'); + size_t l_len = i.name.size(); + if (t) + l_len = t - l_str; + + if (hash_murmurhash(l_str, l_len) == hash) { + farc_unpack_file(this, &i); + return &i; + } + } + return 0; +} + +void farc::write(const char* path, farc_signature signature, farc_flags flags, bool get_files) { + if (!path) + return; + + wchar_t* path_buf = utf8_to_utf16(path); + write(path_buf, signature, flags, get_files); + free_def(path_buf); +} + +void farc::write(const wchar_t* path, farc_signature signature, farc_flags flags, bool get_files) { + if (!path) + return; + + if (get_files) + files.clear(); + + wchar_t full_path_buf[MAX_PATH]; + wchar_t* full_path = _wfullpath(full_path_buf, path, MAX_PATH); + + if (!full_path) + return; + else if (get_files && !path_check_directory_exists(full_path_buf)) + return; + + char* dir_temp = utf16_to_utf8(full_path_buf); + size_t dir_temp_len = utf8_length(dir_temp); + directory_path.assign(dir_temp, dir_temp_len); + file_path.assign(dir_temp, dir_temp_len); + file_path.append(".farc", 5); + free_def(dir_temp); + + if (!get_files || (get_files && !farc_get_files(this))) { + file_stream s; + s.open(file_path.c_str(), "wb"); + if (s.check_not_null()) + farc_pack_files(this, s, signature, flags, get_files); + } +} + +void farc::write(void** data, size_t* size, farc_signature signature, farc_flags flags) { + if (!data || !size) + return; + + directory_path.clear(); + file_path.clear(); + + memory_stream s; + s.open(); + farc_pack_files(this, s, signature, flags); +} + +bool farc::load_file(void* data, const char* path, const char* file, uint32_t hash) { + size_t file_len = utf8_length(file); + if (file_len < 5 || memcmp(&file[file_len - 5], ".farc", 6)) + return false; + + size_t path_len = utf8_length(path); + if (path_len + file_len + 2 > 0x1000) + return false; + + char buf[0x1000]; + memcpy(buf, path, path_len); + memcpy(buf + path_len, file, file_len + 1); + if (!path_check_file_exists(buf)) + return false; + + farc* f = (farc*)data; + f->read(buf, true, false); + return !!f->files.size(); +} + +static errno_t farc_get_files(farc* f) { + f->files.clear(); + f->files.shrink_to_fit(); + + std::vector files = path_get_files(f->directory_path.c_str()); + if (files.size() < 1) + return -1; + + f->files = std::vector(files.size()); + for (farc_file& i : f->files) + i.name = files[&i - f->files.data()]; + return 0; +} + +static void farc_pack_files(farc* f, stream& s, farc_signature signature, farc_flags flags, bool get_files) { + bool plain = false; + for (farc_file& i : f->files) { + bool is_a3da = i.name.find(".a3da") == i.name.size() - 5; + bool is_diva = i.name.find(".diva") == i.name.size() - 5; + bool is_vag = i.name.find(".vag" ) == i.name.size() - 4; + + if (is_a3da || is_diva || is_vag) { + plain = true; + break; + } + } + + if (plain) + signature = FARC_FArc; + + bool compressed = false; + bool encrypted = false; + switch (signature) { + case FARC_FARC: + f->flags = flags; + compressed = !!(flags & FARC_GZIP); + encrypted = !!(flags & FARC_AES); + break; + case FARC_FArC: + f->flags = (farc_flags)0; + compressed = true; + break; + default: + f->flags = (farc_flags)0; + break; + } + + size_t header_length = 0; + switch (signature) { + case FARC_FARC: + header_length += sizeof(int32_t) * 5; + break; + default: + header_length += sizeof(int32_t); + break; + } + + if (signature == FARC_FArc) + for (farc_file& i : f->files) { + header_length += i.name.size() + 1; + header_length += sizeof(int32_t) * 2; + } + else + for (farc_file& i : f->files) { + header_length += i.name.size() + 1; + header_length += sizeof(int32_t) * 3; + } + + size_t align = header_length + 8; + s.set_position(align_val(align, f->alignment), SEEK_SET); + size_t dir_len = f->directory_path.size(); + + aes128_ctx ctx; + if (signature == FARC_FARC) + aes128_init_ctx(&ctx, key); + + f->compression_level = clamp_def(f->compression_level, 0, 12); + + if (get_files) { + char* temp = force_malloc(dir_len + 2 + MAX_PATH); + memcpy(temp, f->directory_path.c_str(), sizeof(char) * dir_len); + temp[dir_len] = '\\'; + for (farc_file& i : f->files) { + if (i.name.size()) { + memcpy(temp + dir_len + 1, i.name.c_str(), sizeof(char) * i.name.size()); + temp[dir_len + 1 + i.name.size()] = '\0'; + } + + i.offset = s.get_position(); + i.size = 0; + i.size_compressed = 0; + free_def(i.data); + free_def(i.data_compressed); + i.compressed = false; + i.encrypted = false; + + file_stream s_t; + s_t.open(temp, "rb"); + if (s_t.check_null()) + continue; + + size_t file_len = s_t.get_length(); + + i.size = file_len; + i.data = force_malloc(file_len); + s_t.read(i.data, file_len); + i.compressed = compressed; + i.encrypted = encrypted; + i.data_changed = false; + } + free_def(temp); + } + + for (farc_file& i : f->files) { + i.offset = s.get_position(); + size_t file_len = i.size; + + if (i.encrypted && encrypted) { + void* t1; + size_t t1_len; + if (i.compressed && compressed) { + if (!i.data_compressed || i.data_changed) { + free_def(i.data_compressed); + deflate::compress_gzip(i.data, file_len, &i.data_compressed, + &i.size_compressed, f->compression_level, i.name.c_str()); + } + t1 = i.data_compressed; + t1_len = i.size_compressed; + } + else { + i.compressed = false; + free_def(i.data_compressed); + t1 = i.data; + t1_len = i.size; + } + + size_t t2_len = align_val(t1_len, f->alignment); + uint8_t* t2 = (uint8_t*)force_malloc(t2_len); + memcpy(t2, t1, t1_len); + memset(t2 + t1_len, 0x78, t2_len - t1_len); + + aes128_ecb_encrypt_buffer(&ctx, t2, t2_len); + + s.write(t2, t2_len); + free_def(t2); + } + else if (i.compressed && compressed) { + i.encrypted = false; + if (!i.data_compressed || i.data_changed) { + free_def(i.data_compressed); + deflate::compress_gzip(i.data, file_len, &i.data_compressed, + &i.size_compressed, f->compression_level, i.name.c_str()); + } + s.write(i.data_compressed, i.size_compressed); + farc_write_padding(f, s, i.size_compressed, signature != FARC_FArc); + } + else { + i.compressed = false; + i.encrypted = false; + free_def(i.data_compressed); + s.write(i.data, file_len); + farc_write_padding(f, s, i.size); + } + i.data_changed = false; + } + + s.set_position(0, SEEK_SET); + switch (signature) { + case FARC_FArc: + default: + s.write_uint32_t_reverse_endianness(FARC_FArc, true); + s.write_uint32_t_reverse_endianness((int32_t)header_length, true); + s.write_uint32_t_reverse_endianness(f->alignment, true); + break; + case FARC_FArC: + s.write_uint32_t_reverse_endianness(FARC_FArC, true); + s.write_uint32_t_reverse_endianness((int32_t)header_length, true); + s.write_uint32_t_reverse_endianness(f->alignment, true); + break; + case FARC_FARC: + s.write_uint32_t_reverse_endianness(FARC_FARC, true); + s.write_uint32_t_reverse_endianness((int32_t)header_length, true); + s.write_uint32_t_reverse_endianness(f->flags, true); + s.write_uint32_t_reverse_endianness(0x00, true); + s.write_uint32_t_reverse_endianness(f->alignment, true); + s.write_uint32_t_reverse_endianness(0x00, true); + s.write_uint32_t_reverse_endianness(0x00, true); + break; + } + + switch (signature) { + case FARC_FArc: + for (farc_file& i : f->files) { + s.write_string_null_terminated(i.name); + s.write_int32_t_reverse_endianness((int32_t)i.offset, true); + s.write_int32_t_reverse_endianness((int32_t)i.size, true); + } + break; + case FARC_FArC: + for (farc_file& i : f->files) { + s.write_string_null_terminated(i.name); + s.write_int32_t_reverse_endianness((int32_t)i.offset, true); + if (i.compressed) { + s.write_int32_t_reverse_endianness((int32_t)i.size_compressed, true); + s.write_int32_t_reverse_endianness((int32_t)i.size, true); + } + else { + s.write_int32_t_reverse_endianness((int32_t)i.size, true); + s.write_int32_t_reverse_endianness(0x00, true); + } + } + break; + case FARC_FARC: + for (farc_file& i : f->files) { + s.write_string_null_terminated(i.name); + s.write_int32_t_reverse_endianness((int32_t)i.offset, true); + if (i.compressed) { + s.write_int32_t_reverse_endianness((int32_t)i.size_compressed, true); + s.write_int32_t_reverse_endianness((int32_t)i.size, true); + } + else { + s.write_int32_t_reverse_endianness((int32_t)i.size, true); + s.write_int32_t_reverse_endianness(0x00, true); + } + s.write_int32_t_reverse_endianness( + (int32_t)((i.compressed ? FARC_GZIP : 0x00) | (i.encrypted ? FARC_AES : 0x00)), true); + } + break; + } + + farc_write_padding(f, s, header_length + 0x08, signature != FARC_FArc); +} + +static errno_t farc_read_header(farc* f, stream& s) { + if (!f || s.check_null()) + return -1; + + s.set_position(0, SEEK_SET); + f->signature = (farc_signature)s.read_uint32_t_reverse_endianness(true); + switch (f->signature) { + case FARC_FArc: + case FARC_FArC: + case FARC_FARC: + break; + default: + return -2; + } + + f->ft = false; + + uint32_t header_length = s.read_uint32_t_reverse_endianness(true); + if (f->signature == FARC_FARC) { + f->flags = (farc_flags)s.read_uint32_t_reverse_endianness(true); + s.read_uint32_t(); + uint32_t alignment = s.read_uint32_t_reverse_endianness(true); + bool modern = !!s.read_uint32_t_reverse_endianness(true); + + // If alignment is way too big, then it might be part of IV + f->ft = (f->flags & FARC_AES) && modern && alignment > 0x1000; + s.set_position(0x10, SEEK_SET); + header_length -= 0x08; + } + + f->files.clear(); + + uint8_t* d_t; + uint8_t* dt; + int32_t length = 0; + + dt = d_t = force_malloc(header_length); + s.read(d_t, header_length); + if (f->ft) { + header_length -= 0x10; + + aes128_ctx ctx; + aes128_init_ctx_iv(&ctx, key_ft, dt); + dt += 0x10; + aes128_cbc_decrypt_buffer(&ctx, dt, header_length); + + header_length -= ((uint8_t*)dt)[header_length - 1]; // PKCS7 Padding + } + + if (f->ft) { + f->alignment = load_reverse_endianness_uint32_t((void*)dt); + uint32_t files_count = load_reverse_endianness_uint32_t((void*)(dt + 8)); + uint32_t entry_size = load_reverse_endianness_uint32_t((void*)(dt + 12)); + dt += sizeof(uint32_t) * 4; + + f->files.clear(); + f->files.reserve(files_count); + while ((dt - d_t < header_length + 0x08LL) && files_count) { + size_t length = 0; + while (dt[length]) + length++; + + farc_file ff; + ff.name.assign((const char*)dt, length); + dt += length + 1; + ff.offset = (size_t)load_reverse_endianness_uint32_t((void*)dt); + ff.size_compressed = (size_t)load_reverse_endianness_uint32_t((void*)(dt + 4)); + ff.size = (size_t)load_reverse_endianness_uint32_t((void*)(dt + 8)); + farc_flags flags = (farc_flags)load_reverse_endianness_uint32_t((void*)(dt + 12)); + + if (ff.size) + ff.compressed = true; + else { + ff.size = ff.size_compressed; + ff.size_compressed = 0; + ff.compressed = false; + } + + ff.encrypted = ((f->flags | flags) & FARC_AES) != 0; + f->files.push_back(ff); + dt += entry_size; + files_count--; + } + + free_def(d_t); + return 0; + } + + size_t entry_size; + if (f->signature == FARC_FARC) { + f->alignment = load_reverse_endianness_uint32_t((void*)dt); + uint32_t entry_offset = load_reverse_endianness_uint32_t((void*)(dt + 4)); + uint32_t header_offset = load_reverse_endianness_uint32_t((void*)(dt + 8)); + dt += sizeof(uint32_t) * 3; + + entry_size = sizeof(uint32_t) * 3 + entry_offset; + } + else if (f->signature != FARC_FArc) { + f->alignment = load_reverse_endianness_uint32_t((void*)dt); + dt += sizeof(uint32_t); + + entry_size = sizeof(uint32_t) * 3; + } + else { + f->alignment = load_reverse_endianness_uint32_t((void*)dt); + dt += sizeof(uint32_t); + + entry_size = sizeof(uint32_t) * 2; + } + + size_t count = 0; + uint8_t* position = dt; + + while (dt - d_t < header_length) { + while (*dt++); + dt += entry_size; + count++; + } + dt = position; + + bool encrypted = !!(f->flags & FARC_AES); + + f->files.clear(); + f->files.resize(count); + if (f->signature != FARC_FArc) + for (farc_file& i : f->files) { + size_t length = 0; + while (dt[length]) + length++; + i.name.assign((const char*)dt, length); + dt += length + 1; + i.offset = (size_t)load_reverse_endianness_uint32_t((void*)dt); + i.size_compressed = (size_t)load_reverse_endianness_uint32_t((void*)(dt + 4)); + i.size = (size_t)load_reverse_endianness_uint32_t((void*)(dt + 8)); + + if (i.size) + i.compressed = true; + else { + i.size = i.size_compressed; + i.size_compressed = 0; + i.compressed = false; + } + + i.encrypted = encrypted; + dt += entry_size; + } + else + for (farc_file& i : f->files) { + size_t length = 0; + while (dt[length]) + length++; + i.name.assign((const char*)dt, length); + dt += length + 1; + i.offset = (size_t)load_reverse_endianness_uint32_t((void*)dt); + i.size = (size_t)load_reverse_endianness_uint32_t((void*)(dt + 4)); + + i.size_compressed = 0; + i.compressed = false; + i.encrypted = false; + dt += sizeof(int32_t) * 2; + } + + free_def(d_t); + return 0; +} + +static void farc_unpack_files(farc* f, stream& s, bool save) { + if (!f || s.check_null() || !f->files.size()) + return; + + size_t max_path_len = 0; + size_t dir_len = f->directory_path.size(); + for (farc_file& i : f->files) { + size_t path_len = dir_len + 1 + utf8_to_utf16_length(i.name.c_str()); + if (max_path_len < path_len) + max_path_len = path_len; + } + + if (save) { + wchar_t* dir_temp = utf8_to_utf16(f->directory_path.c_str()); + CreateDirectoryW(dir_temp, 0); + free_def(dir_temp); + } + + char* temp_path = force_malloc(max_path_len + 1); + memcpy(temp_path, f->directory_path.c_str(), sizeof(char) * dir_len); + temp_path[dir_len] = '\\'; + + for (farc_file& i : f->files) + farc_unpack_file(f, s, &i, save, temp_path, dir_len); + + free_def(temp_path); +} + +static void farc_unpack_file(farc* f, farc_file* ff) { + if (ff->data) + return; + else if (ff->data_compressed) { + deflate::decompress(ff->data_compressed, ff->size_compressed, + &ff->data, &ff->size, deflate::MODE_GZIP); + return; + } + + file_stream s; + s.open(f->file_path.c_str(), "rb"); + if (s.check_not_null()) + farc_unpack_file(f, s, ff); +} + +static void farc_unpack_file(farc* f, stream& s, farc_file* ff, bool save, char* temp_path, size_t dir_len) { + if (!f || s.check_null()) + return; + + if (ff->data) + free_def(ff->data); + + s.set_position(ff->offset, SEEK_SET); + + if (f->signature == FARC_FArc) { + ff->data_compressed = 0; + ff->data = force_malloc(ff->size); + s.read(ff->data, ff->size); + } + else if (f->signature == FARC_FArC) { + if (ff->compressed) { + ff->data_compressed = force_malloc(ff->size_compressed); + s.read(ff->data_compressed, ff->size_compressed); + deflate::decompress(ff->data_compressed, ff->size_compressed, + &ff->data, &ff->size, deflate::MODE_GZIP); + } + else { + ff->data = force_malloc(ff->size); + s.read(ff->data, ff->size); + } + } + else if (ff->compressed || ff->encrypted) { + size_t temp_s = ff->compressed ? ff->size_compressed : ff->size; + temp_s = ff->encrypted ? align_val(temp_s, f->alignment) : temp_s; + void* temp = force_malloc(temp_s); + s.read(temp, temp_s); + + size_t t = (size_t)temp; + if (ff->encrypted) + if (f->ft) { + temp_s -= 0x10; + t += 0x10; + + aes128_ctx ctx; + aes128_init_ctx_iv(&ctx, key_ft, (uint8_t*)temp); + aes128_cbc_decrypt_buffer(&ctx, (uint8_t*)t, temp_s); + + ff->size_compressed = temp_s - ((uint8_t*)t)[temp_s - 1]; // PKCS7 Padding + } + else { + aes128_ctx ctx; + aes128_init_ctx(&ctx, key); + aes128_ecb_decrypt_buffer(&ctx, (uint8_t*)t, temp_s); + } + + if (ff->compressed) { + ff->data_compressed = force_malloc(ff->size_compressed); + memcpy(ff->data_compressed, (void*)t, ff->size_compressed); + deflate::decompress(ff->data_compressed, ff->size_compressed, + &ff->data, &ff->size, deflate::MODE_GZIP); + } + else { + ff->data_compressed = 0; + ff->data = force_malloc(ff->size); + memcpy(ff->data, (void*)t, ff->size); + } + free_def(temp); + } + else { + ff->data_compressed = 0; + ff->data = force_malloc(ff->size); + s.read(ff->data, ff->size); + } + ff->data_changed = false; + + if (!save) + return; + + if (ff->data) { + if (ff->name.size()) { + memcpy(temp_path + dir_len + 1, ff->name.c_str(), ff->name.size()); + temp_path[dir_len + 1 + ff->name.size()] = '\0'; + + file_stream temp_s; + temp_s.open(temp_path, "wb"); + if (temp_s.check_not_null()) + temp_s.write(ff->data, ff->size); + } + + free(ff->data); + ff->data = 0; + } + + if (ff->data_compressed) { + free(ff->data_compressed); + ff->data_compressed = 0; + } +} + +static void farc_write_padding(farc* f, stream& s, size_t size, bool x) { + size_t align = align_val(size, f->alignment) - size; + if (!x) { + uint8_t padding[] = { + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + }; + s.write(padding, align); + } + else { + uint8_t padding[] = { + 0x78, 0x78, 0x78, 0x78, 0x78, 0x78, 0x78, 0x78, + 0x78, 0x78, 0x78, 0x78, 0x78, 0x78, 0x78, 0x78, + }; + s.write(padding, align); + } +} diff --git a/src/KKdLib/farc.hpp b/src/KKdLib/farc.hpp new file mode 100644 index 0000000..14c124d --- /dev/null +++ b/src/KKdLib/farc.hpp @@ -0,0 +1,84 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include +#include +#include "default.hpp" + +enum farc_signature { + FARC_FArc = 'FArc', + FARC_FArC = 'FArC', + FARC_FARC = 'FARC', +}; + +enum farc_flags { + FARC_NONE = 0x00, + FARC_GZIP = 0x02, + FARC_AES = 0x04, +}; + +struct farc_file { + std::string name; + size_t offset; + size_t size; + size_t size_compressed; + void* data; + void* data_compressed; + bool compressed; + bool encrypted; + bool data_changed; + + inline farc_file() : offset(), size(), size_compressed(), data(), + data_compressed(), compressed(), encrypted(), data_changed() { + + } + + inline ~farc_file() { + if (data) + free(data); + if (data_compressed) + free(data_compressed); + } +}; + +struct farc { + std::string file_path; + std::string directory_path; + std::vector files; + farc_signature signature; + farc_flags flags; + int32_t compression_level; + uint32_t alignment; + bool ft; + + farc(); + ~farc(); + + farc_file* add_file(const char* name); + farc_file* add_file(const wchar_t* name); + const char* get_file_name(uint32_t hash); + size_t get_file_size(const char* name); + size_t get_file_size(const wchar_t* name); + size_t get_file_size(uint32_t hash); + bool has_file(const char* name); + bool has_file(const wchar_t* name); + bool has_file(uint32_t hash); + void read(const char* path, bool unpack = true, bool save = false); + void read(const wchar_t* path, bool unpack = true, bool save = false); + void read(const void* data, size_t size, bool unpack = true); + farc_file* read_file(const char* name); + farc_file* read_file(const wchar_t* name); + farc_file* read_file(uint32_t hash); + void write(const char* path, farc_signature signature = FARC_FArC, + farc_flags flags = FARC_NONE, bool get_files = true); + void write(const wchar_t* path, farc_signature signature = FARC_FArC, + farc_flags flags = FARC_NONE, bool get_files = true); + void write(void** data, size_t* size, farc_signature signature = FARC_FArC, + farc_flags flags = FARC_NONE); + + static bool load_file(void* data, const char* path, const char* file, uint32_t hash); +}; diff --git a/src/KKdLib/half_t.cpp b/src/KKdLib/half_t.cpp new file mode 100644 index 0000000..a0714b6 --- /dev/null +++ b/src/KKdLib/half_t.cpp @@ -0,0 +1,82 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "half_t.hpp" + +float_t half_to_float_convert(half_t h) { + int32_t si32; + uint16_t sign = (h >> 15) & 0x001; + uint16_t exponent = (h >> 10) & 0x01F; + uint16_t mantissa = h & 0x3FF; + si32 = sign ? (int32_t)0x80000000 : 0x00000000; + + if (exponent == 0x1F) + si32 |= 0x7F800000; + else if (exponent != 0x00) + si32 |= ((int32_t)exponent - 0x0F + 0x7F) << 23; + si32 |= (int32_t)mantissa << 13; + return *(float_t*)&si32; +} + +half_t float_to_half_convert(float_t val) { + int32_t si32 = *(int32_t*)&val; + if (si32 == 0x00000000) + return FLOAT16_POSITIVE_ZERO; + else if ((uint64_t)si32 == 0x80000000) + return FLOAT16_NEGATIVE_ZERO; + + int16_t sign = (int16_t)((si32 >> 31) & 0x001); + int16_t exponent = (int16_t)((si32 >> 23) & 0x0FF); + int16_t mantissa = (int16_t)((si32 >> 13) & 0x3FF); + + if (exponent == 0xFF) + exponent = 0x1F; + else if (exponent != 0x00) { + exponent -= 0x7F - 0x0F; + if (exponent < 0x00) + exponent = mantissa = 0; + else if (exponent >= 0x1F) + exponent = 0x1F; + } + return (half_t)((sign << 15) | (exponent << 10) | mantissa); +} + +double_t half_to_double_convert(half_t h) { + int64_t si64; + uint16_t sign = (h >> 15) & 0x001; + uint16_t exponent = (h >> 10) & 0x01F; + uint16_t mantissa = h & 0x3FF; + si64 = sign ? (int64_t)0x8000000000000000 : 0x0000000000000000; + + if (exponent == 0x1F) + si64 |= 0x7FF0000000000000; + else if (exponent != 0x00) + si64 |= ((int64_t)exponent - 0x0F + 0x3FF) << 52; + si64 |= (int64_t)mantissa << 42; + return *(double_t*)&si64; +} + +half_t double_to_half_convert(double_t val) { + int64_t si64 = *(int64_t*)&val; + if (si64 == 0x0000000000000000) + return FLOAT16_POSITIVE_ZERO; + else if ((uint64_t)si64 == 0x8000000000000000) + return FLOAT16_NEGATIVE_ZERO; + + int16_t sign = (int16_t)((si64 >> 63) & 0x001); + int16_t exponent = (int16_t)((si64 >> 52) & 0x7FF); + int16_t mantissa = (int16_t)((si64 >> 42) & 0x3FF); + + if (exponent == 0x7FF) + exponent = 0x1F; + else if (exponent != 0x00) { + exponent -= 0x3FF - 0x0F; + if (exponent < 0x00) + exponent = mantissa = 0; + else if (exponent >= 0x1F) + exponent = 0x1F; + } + return (half_t)((sign << 15) | (exponent << 10) | mantissa); +} diff --git a/src/KKdLib/half_t.hpp b/src/KKdLib/half_t.hpp new file mode 100644 index 0000000..40b7ffd --- /dev/null +++ b/src/KKdLib/half_t.hpp @@ -0,0 +1,67 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "default.hpp" +#include + +#define FLOAT16_NAN ((half_t)0x7FFF) +#define FLOAT16_POSITIVE_NAN ((half_t)0x7FFF) +#define FLOAT16_NEGATIVE_NAN ((half_t)0xFFFF) +#define FLOAT16_POSITIVE_ZERO ((half_t)0x0000) +#define FLOAT16_NEGATIVE_ZERO ((half_t)0x8000) +#define FLOAT16_POSITIVE_INF ((half_t)0x7C00) +#define FLOAT16_NEGATIVE_INF ((half_t)0xFC00) + +#define HALF_MAX 65504 +#define HALF_MIN 0.00006103515625 + +typedef unsigned short half_t; + +inline half_t load_reverse_endianness_half_t(void* ptr) { + return (half_t)_byteswap_ushort(*(uint16_t*)ptr); +} + +inline void store_reverse_endianness_half_t(void* ptr, half_t value) { + *(half_t*)ptr = (half_t)_byteswap_ushort((uint16_t)value); +} + +inline half_t reverse_endianness_half_t(half_t value) { + return (half_t)_byteswap_ushort((uint16_t)value); +} + +extern float_t half_to_float_convert(half_t h); +extern half_t float_to_half_convert(float_t val); +extern double_t half_to_double_convert(half_t h); +extern half_t double_to_half_convert(double_t val); + +inline float_t half_to_float(half_t h) { + extern bool f16c; + if (f16c) + return _mm_cvtss_f32(_mm_cvtph_ps(_mm_cvtsi32_si128((uint16_t)h))); + return half_to_float_convert(h); +} + +inline half_t float_to_half(float_t val) { + extern bool f16c; + if (f16c) + return (half_t)_mm_cvtsi128_si32(_mm_cvtps_ph(_mm_set_ss(val), _MM_FROUND_CUR_DIRECTION)); + return float_to_half_convert(val); +} + +inline double_t half_to_double(half_t h) { + extern bool f16c; + if (f16c) + return _mm_cvtss_f32(_mm_cvtph_ps(_mm_cvtsi32_si128((uint16_t)h))); + return half_to_double_convert(h); +} + +inline half_t double_to_half(double_t val) { + extern bool f16c; + if (f16c) + return (half_t)_mm_cvtsi128_si32(_mm_cvtps_ph(_mm_cvtpd_ps(_mm_set_sd(val)), _MM_FROUND_CUR_DIRECTION)); + return double_to_half_convert(val); +} diff --git a/src/KKdLib/hash.cpp b/src/KKdLib/hash.cpp new file mode 100644 index 0000000..a2683e6 --- /dev/null +++ b/src/KKdLib/hash.cpp @@ -0,0 +1,225 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "hash.hpp" + +static const uint16_t hash_crc16_ccitt_table[] = { + 0x0000, 0x1021, 0x2042, 0x3063, 0x4084, 0x50A5, 0x60C6, 0x70E7, + 0x8108, 0x9129, 0xA14A, 0xB16B, 0xC18C, 0xD1AD, 0xE1CE, 0xF1EF, + 0x1231, 0x0210, 0x3273, 0x2252, 0x52B5, 0x4294, 0x72F7, 0x62D6, + 0x9339, 0x8318, 0xB37B, 0xA35A, 0xD3BD, 0xC39C, 0xF3FF, 0xE3DE, + 0x2462, 0x3443, 0x0420, 0x1401, 0x64E6, 0x74C7, 0x44A4, 0x5485, + 0xA56A, 0xB54B, 0x8528, 0x9509, 0xE5EE, 0xF5CF, 0xC5AC, 0xD58D, + 0x3653, 0x2672, 0x1611, 0x0630, 0x76D7, 0x66F6, 0x5695, 0x46B4, + 0xB75B, 0xA77A, 0x9719, 0x8738, 0xF7DF, 0xE7FE, 0xD79D, 0xC7BC, + 0x48C4, 0x58E5, 0x6886, 0x78A7, 0x0840, 0x1861, 0x2802, 0x3823, + 0xC9CC, 0xD9ED, 0xE98E, 0xF9AF, 0x8948, 0x9969, 0xA90A, 0xB92B, + 0x5AF5, 0x4AD4, 0x7AB7, 0x6A96, 0x1A71, 0x0A50, 0x3A33, 0x2A12, + 0xDBFD, 0xCBDC, 0xFBBF, 0xEB9E, 0x9B79, 0x8B58, 0xBB3B, 0xAB1A, + 0x6CA6, 0x7C87, 0x4CE4, 0x5CC5, 0x2C22, 0x3C03, 0x0C60, 0x1C41, + 0xEDAE, 0xFD8F, 0xCDEC, 0xDDCD, 0xAD2A, 0xBD0B, 0x8D68, 0x9D49, + 0x7E97, 0x6EB6, 0x5ED5, 0x4EF4, 0x3E13, 0x2E32, 0x1E51, 0x0E70, + 0xFF9F, 0xEFBE, 0xDFDD, 0xCFFC, 0xBF1B, 0xAF3A, 0x9F59, 0x8F78, + 0x9188, 0x81A9, 0xB1CA, 0xA1EB, 0xD10C, 0xC12D, 0xF14E, 0xE16F, + 0x1080, 0x00A1, 0x30C2, 0x20E3, 0x5004, 0x4025, 0x7046, 0x6067, + 0x83B9, 0x9398, 0xA3FB, 0xB3DA, 0xC33D, 0xD31C, 0xE37F, 0xF35E, + 0x02B1, 0x1290, 0x22F3, 0x32D2, 0x4235, 0x5214, 0x6277, 0x7256, + 0xB5EA, 0xA5CB, 0x95A8, 0x8589, 0xF56E, 0xE54F, 0xD52C, 0xC50D, + 0x34E2, 0x24C3, 0x14A0, 0x0481, 0x7466, 0x6447, 0x5424, 0x4405, + 0xA7DB, 0xB7FA, 0x8799, 0x97B8, 0xE75F, 0xF77E, 0xC71D, 0xD73C, + 0x26D3, 0x36F2, 0x0691, 0x16B0, 0x6657, 0x7676, 0x4615, 0x5634, + 0xD94C, 0xC96D, 0xF90E, 0xE92F, 0x99C8, 0x89E9, 0xB98A, 0xA9AB, + 0x5844, 0x4865, 0x7806, 0x6827, 0x18C0, 0x08E1, 0x3882, 0x28A3, + 0xCB7D, 0xDB5C, 0xEB3F, 0xFB1E, 0x8BF9, 0x9BD8, 0xABBB, 0xBB9A, + 0x4A75, 0x5A54, 0x6A37, 0x7A16, 0x0AF1, 0x1AD0, 0x2AB3, 0x3A92, + 0xFD2E, 0xED0F, 0xDD6C, 0xCD4D, 0xBDAA, 0xAD8B, 0x9DE8, 0x8DC9, + 0x7C26, 0x6C07, 0x5C64, 0x4C45, 0x3CA2, 0x2C83, 0x1CE0, 0x0CC1, + 0xEF1F, 0xFF3E, 0xCF5D, 0xDF7C, 0xAF9B, 0xBFBA, 0x8FD9, 0x9FF8, + 0x6E17, 0x7E36, 0x4E55, 0x5E74, 0x2E93, 0x3EB2, 0x0ED1, 0x1EF0, +}; + +// Empty string +const uint64_t hash_fnv1a64m_empty = 0xCBF29CE44FD0BFC1; +// Empty string +const uint32_t hash_murmurhash_empty = 0x0CAD3078; +// "NULL" string +const uint32_t hash_murmurhash_null = 0x5A009B23; +// Empty string +const uint32_t hash_crc16_ccitt_empty = 0xFFFF; + +// FNV 1a 64-bit Modified +// 0x1403B04D0 in SBZV_7.10 +uint64_t hash_fnv1a64m(const void* data, size_t size, bool make_upper) { + const uint8_t* d = (const uint8_t*)data; + + uint64_t hash = 0xCBF29CE484222325; + if (data) + if (make_upper) // Hash text UPPERCASE + for (size_t i = size; i; i--) { + uint8_t c = *d++; + if (c > 0x60 && c < 0x7B) + c -= 0x20; + hash ^= c; + hash *= 0x100000001B3; + } + else + for (size_t i = size; i; i--) { + hash ^= *d++; + hash *= 0x100000001B3; + } + return (hash >> 32) ^ hash; // Actual Modification +} + +// MurmurHash +// 0x814D7A9C in PCSB00554 +// 0x8134C304 in PCSB01007 +// 0x0069CEA4 in NPEB02013 +uint32_t hash_murmurhash(const void* data, size_t size, + uint32_t seed, bool upper, bool big_endian) { + const uint8_t* d = (const uint8_t*)data; + + uint32_t a = 0; + uint32_t b = 0; + uint32_t hash = 0; + size_t i = 0; + + const uint32_t m = 0x7FD652AD; + const int32_t r = 16; + + hash = seed + 0xDEADBEEF; + if (d) + if (upper) { + if (big_endian) + for (i = 0; size > 3; size -= 4, i += 4, d += 4) { + b = load_reverse_endianness_uint32_t(d); + hash += b; + hash *= m; + hash ^= hash >> r; + } + else + for (i = 0; size > 3; size -= 4, i += 4, d += 4) { + b = *(uint32_t*)d; + hash += b; + hash *= m; + hash ^= hash >> r; + } + + if (size > 0) { + if (size > 1) { + if (size > 2) + hash += (uint32_t)d[2] << 16; + hash += (uint32_t)d[1] << 8; + } + hash += d[0]; + hash *= m; + hash ^= hash >> r; + } + } + else { + if (big_endian) + for (i = 0; size > 3; size -= 4, i += 4, d += 4) { + b = load_reverse_endianness_uint32_t(d); + + a = b & 0xFF; + if (a > 0x60 && a < 0x7B) + a -= 0x20; + b = (b & 0xFFFFFF00) | a; + + a = (b >> 8) & 0xFF; + if (a > 0x60 && a < 0x7B) + a -= 0x20; + b = (b & 0xFFFF00FF) | (a << 8); + + a = (b >> 16) & 0xFF; + if (a > 0x60 && a < 0x7B) + a -= 0x20; + b = (b & 0xFF00FFFF) | (a << 16); + + a = (b >> 24) & 0xFF; + if (a > 0x60 && a < 0x7B) + a -= 0x20; + b = (b & 0x00FFFFFF) | (a << 24); + + hash += b; + hash *= m; + hash ^= hash >> r; + } + else + for (i = 0; size > 3; size -= 4, i += 4, d += 4) { + b = *(uint32_t*)d; + + a = b & 0xFF; + if (a > 0x60 && a < 0x7B) + a -= 0x20; + b = (b & 0xFFFFFF00) | a; + + a = (b >> 8) & 0xFF; + if (a > 0x60 && a < 0x7B) + a -= 0x20; + b = (b & 0xFFFF00FF) | (a << 8); + + a = (b >> 16) & 0xFF; + if (a > 0x60 && a < 0x7B) + a -= 0x20; + b = (b & 0xFF00FFFF) | (a << 16); + + a = (b >> 24) & 0xFF; + if (a > 0x60 && a < 0x7B) + a -= 0x20; + b = (b & 0x00FFFFFF) | (a << 24); + + hash += b; + hash *= m; + hash ^= hash >> r; + } + + switch (size) { + case 3: + b = d[2]; + if (b > 0x60 && b < 0x7B) + b -= 0x20; + hash += b << 16; + case 2: + b = d[1]; + if (b > 0x60 && b < 0x7B) + b -= 0x20; + hash += b << 8; + case 1: + b = d[0]; + if (b > 0x60 && b < 0x7B) + b -= 0x20; + hash += b; + + hash *= m; + hash ^= hash >> r; + break; + } + } + + hash *= m; + hash ^= hash >> 10; + hash *= m; + hash ^= hash >> 17; + return hash; +} + +// CRC16-CCITT +// 0x140011A90 in SBZV_7.10 +uint16_t hash_crc16_ccitt(const void* data, size_t size, bool make_upper) { + const uint8_t* d = (const uint8_t*)data; + + uint16_t hash = 0xFFFF; + if (make_upper) // Modification for only uppercase latin text + for (size_t i = size; i; i--) { + uint8_t a = *d++; + if (a > 0x60 && a < 0x7B) + a -= 0x20; + hash = (uint16_t)(hash_crc16_ccitt_table[(hash >> 8) ^ a] ^ (hash << 8)); + } + else + for (size_t i = size; i; i--) + hash = (uint16_t)(hash_crc16_ccitt_table[(hash >> 8) ^ *d++] ^ (hash << 8)); + return hash; +} diff --git a/src/KKdLib/hash.hpp b/src/KKdLib/hash.hpp new file mode 100644 index 0000000..4e3071f --- /dev/null +++ b/src/KKdLib/hash.hpp @@ -0,0 +1,178 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include +#include "default.hpp" + +// Empty string +extern const uint64_t hash_fnv1a64m_empty; +// Empty string +extern const uint32_t hash_murmurhash_empty; +// "NULL" string +extern const uint32_t hash_murmurhash_null; +// Empty string +extern const uint32_t hash_crc16_ccitt_empty; + +extern uint64_t hash_fnv1a64m(const void* data, size_t size, bool make_upper = false); +extern uint32_t hash_murmurhash(const void* data, size_t size, + uint32_t seed = 0, bool upper = false, bool big_endian = false); +extern uint16_t hash_crc16_ccitt(const void* data, size_t size, bool make_upper = false); + +inline uint64_t hash_utf8_fnv1a64m(const char* data, bool make_upper = false) { + return hash_fnv1a64m(data, utf8_length(data), make_upper); +} + +inline uint64_t hash_utf16_fnv1a64m(const wchar_t* data, bool make_upper = false) { + char* temp = utf16_to_utf8(data); + uint64_t hash = hash_fnv1a64m(temp, utf8_length(temp), make_upper); + free_def(temp); + return hash; +} + +inline uint64_t hash_string_fnv1a64m(const std::string& data, bool make_upper = false) { + return hash_fnv1a64m(data.c_str(), data.size(), make_upper); +} + +inline uint32_t hash_utf8_murmurhash(const char* data, uint32_t seed = 0, bool upper = false) { + return hash_murmurhash(data, utf8_length(data), seed, upper); +} + +inline uint32_t hash_utf16_murmurhash(const wchar_t* data, uint32_t seed = 0, bool upper = false) { + char* temp = utf16_to_utf8(data); + uint32_t hash = hash_murmurhash(temp, utf8_length(temp), seed, upper); + free_def(temp); + return hash; +} + +inline uint32_t hash_string_murmurhash(const std::string& data, uint32_t seed = 0, bool upper = false) { + return hash_murmurhash(data.c_str(), data.size(), seed, upper); +} + +inline uint16_t hash_utf8_crc16_ccitt(const char* data, bool make_upper = false) { + return hash_crc16_ccitt(data, utf8_length(data), make_upper); +} + +inline uint16_t hash_utf16_crc16_ccitt(const wchar_t* data, bool make_upper = false) { + char* temp = utf16_to_utf8(data); + uint32_t hash = hash_crc16_ccitt(temp, utf8_length(temp), make_upper); + free_def(temp); + return hash; +} + +inline uint16_t hash_string_crc16_ccitt(const std::string& data, bool make_upper = false) { + return hash_crc16_ccitt(data.c_str(), data.size(), make_upper); +} + +struct string_hash { + std::string str; + uint64_t hash_fnv1a64m; + uint32_t hash_murmurhash; + + inline string_hash() { + hash_fnv1a64m = hash_fnv1a64m_empty; + hash_murmurhash = hash_murmurhash_empty; + } + + inline string_hash(const char* str) { + this->str.assign(str); + this->hash_fnv1a64m = hash_string_fnv1a64m(this->str); + this->hash_murmurhash = hash_string_murmurhash(this->str); + } + + inline string_hash(const char* str, size_t length) { + this->str.assign(str, length); + this->hash_fnv1a64m = hash_string_fnv1a64m(this->str); + this->hash_murmurhash = hash_string_murmurhash(this->str); + } + + inline string_hash(std::string& str) { + this->str.assign(str); + this->hash_fnv1a64m = hash_string_fnv1a64m(this->str); + this->hash_murmurhash = hash_string_murmurhash(this->str); + } + + inline string_hash(std::string&& str) { + this->str.assign(str); + this->hash_fnv1a64m = hash_string_fnv1a64m(this->str); + this->hash_murmurhash = hash_string_murmurhash(this->str); + } + + inline ~string_hash() { + + } + + inline const char* c_str() { + return str.c_str(); + } + + inline void append(const char* str) { + this->str.append(str); + this->hash_fnv1a64m = hash_string_fnv1a64m(this->str); + this->hash_murmurhash = hash_string_murmurhash(this->str); + } + + inline void append(const char* str, size_t length) { + this->str.append(str, length); + this->hash_fnv1a64m = hash_string_fnv1a64m(this->str); + this->hash_murmurhash = hash_string_murmurhash(this->str); + } + + inline void append(std::string& str) { + this->str.append(str); + this->hash_fnv1a64m = hash_string_fnv1a64m(this->str); + this->hash_murmurhash = hash_string_murmurhash(this->str); + } + + inline void append(std::string&& str) { + this->str.append(str); + this->hash_fnv1a64m = hash_string_fnv1a64m(this->str); + this->hash_murmurhash = hash_string_murmurhash(this->str); + } + + inline void append(string_hash& str) { + this->str.append(str.str); + this->hash_fnv1a64m = hash_string_fnv1a64m(this->str); + this->hash_murmurhash = hash_string_murmurhash(this->str); + } + + inline void assign(const char* str) { + this->str.assign(str); + this->hash_fnv1a64m = hash_string_fnv1a64m(this->str); + this->hash_murmurhash = hash_string_murmurhash(this->str); + } + + inline void assign(const char* str, size_t length) { + this->str.assign(str, length); + this->hash_fnv1a64m = hash_string_fnv1a64m(this->str); + this->hash_murmurhash = hash_string_murmurhash(this->str); + } + + inline void assign(std::string& str) { + this->str.assign(str); + this->hash_fnv1a64m = hash_string_fnv1a64m(this->str); + this->hash_murmurhash = hash_string_murmurhash(this->str); + } + + inline void assign(std::string&& str) { + this->str.assign(str); + this->hash_fnv1a64m = hash_string_fnv1a64m(this->str); + this->hash_murmurhash = hash_string_murmurhash(this->str); + } + + inline void assign(string_hash& str) { + this->str.assign(str.str); + this->hash_fnv1a64m = hash_string_fnv1a64m(this->str); + this->hash_murmurhash = hash_string_murmurhash(this->str); + } + + inline void clear() { + str.clear(); + str.shrink_to_fit(); + hash_fnv1a64m = hash_fnv1a64m_empty; + hash_murmurhash = hash_murmurhash_empty; + } +}; diff --git a/src/KKdLib/interpolation.cpp b/src/KKdLib/interpolation.cpp new file mode 100644 index 0000000..ae4276a --- /dev/null +++ b/src/KKdLib/interpolation.cpp @@ -0,0 +1,248 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "interpolation.hpp" + +void interpolate_chs_reverse_value(float_t* arr, size_t length, + float_t& t1, float_t& t2, size_t f1, size_t f2, size_t f) { + vec2 t = vec2( + (float_t)(int64_t)(f - f1 + 0), + (float_t)(int64_t)(f - f1 + 1) + ) / (float_t)(int64_t)(f2 - f1); + vec2 t_2 = t * t; + vec2 t_3 = t_2 * t; + vec2 t_23 = 3.0f * t_2; + vec2 t_32 = 2.0f * t_3; + + vec2 h00 = t_32 - t_23 + 1.0f; + vec2 h01 = t_23 - t_32; + vec2 h10 = t_3 - 2.0f * t_2 + t; + vec2 h11 = t_3 - t_2; + + vec2 t1_t2 = *(vec2*)&arr[f] - h00 * arr[f1] - h01 * arr[f2]; + t1_t2 /= (t_2.x - t.x) * (t_2.y - t.y); + + t1 = -h11.y * t1_t2.x + h11.x * t1_t2.y; + t2 = h10.y * t1_t2.x - h10.x * t1_t2.y; +} + +void interpolate_chs_reverse(float_t* arr, size_t length, + float_t& t1, float_t& t2, size_t f1, size_t f2) { + t1 = 0.0f; + t2 = 0.0f; + + if (f2 - f1 - 2 < 1) + return; + + float_t _t1 = 0.0f; + float_t _t2 = 0.0f; + double_t tt1 = 0.0; + double_t tt2 = 0.0; + for (size_t i = f1 + 1; i < f2 - 1; i++) { + interpolate_chs_reverse_value(arr, length, _t1, _t2, f1, f2, i); + tt1 += _t1; + tt2 += _t2; + } + t1 = (float_t)(tt1 / (double_t)(f2 - f1 - 2)); + t2 = (float_t)(tt2 / (double_t)(f2 - f1 - 2)); +} + +int32_t interpolate_chs_reverse_sequence( + std::vector& values_src, std::vector& values, bool fast) { + size_t count = values_src.size(); + if (!count) + return 0; + else if (count == 1) { + if (values_src[0] != 0.0f) { + values.push_back({ 0, values_src[0] }); + return 1; + } + else + return 0; + } + else { + float_t val = values_src.data()[0]; + float_t* arr = &values_src.data()[1]; + for (size_t i = count - 1; i; i--) + if (val != *arr++) + break; + + if (arr == values_src.data() + count) + if (values_src[0] != 0.0f) { + values.push_back({ 0, values_src[0] }); + return 1; + } + else + return 0; + } + + float_t* arr = values_src.data(); + + const float_t reverse_bias = 0.0001f; + const int32_t reverse_min_count = 4; + + float_t* a = arr; + size_t left_count = count; + int32_t frame = 0; + int32_t prev_frame = 0; + float_t t2_old = 0.0f; + while (left_count > 0) { + if (left_count < reverse_min_count) { + if (left_count > 1) { + values.push_back({ (float_t)frame, a[0], t2_old, 0.0f }); + for (size_t j = 1; j < left_count - 1; j++) + values.push_back({ (float_t)(int64_t)(frame + j), a[j] }); + t2_old = 0.0f; + } + break; + } + + size_t i = 0; + size_t i_prev = 0; + float_t t1 = 0.0f; + float_t t2 = 0.0f; + float_t t1_prev = 0.0f; + float_t t2_prev = 0.0f; + bool has_prev_succeded = false; + bool has_error = false; + bool has_prev_error = false; + bool constant_prev = false; + + int32_t c = 0; + for (i = reverse_min_count - 1, i_prev = i; i < left_count; i++) { + bool constant = true; + for (size_t j = 1; j <= i; j++) + if (memcmp(&a[0], &a[j], sizeof(float_t))) { + constant = false; + break; + } + + if (!fast) { + double_t t1_accum = 0.0; + double_t t2_accum = 0.0; + for (size_t j = 1; j < i - 1; j++) { + float_t t1 = 0.0f; + float_t t2 = 0.0f; + interpolate_chs_reverse_value(a, left_count, t1, t2, 0, i, j); + t1_accum += t1; + t2_accum += t2; + } + t1 = (float_t)(t1_accum / (double_t)(i - 2)); + t2 = (float_t)(t2_accum / (double_t)(i - 2)); + } + else + interpolate_chs_reverse_value(a, left_count, t1, t2, 0, i, 1); + + has_error = false; + for (size_t j = 1; j < i; j++) { + float_t val = interpolate_chs_value(a[0], a[i], t1, t2, 0.0f, (float_t)i, (float_t)j); + if (fabsf(val - a[j]) > reverse_bias) { + has_error = true; + break; + } + } + + if (fabsf(t1) > 0.5f || fabsf(t2) > 0.5f) + has_error = true; + + if (!has_error) { + i_prev = i; + t1_prev = t1; + t2_prev = t2; + constant_prev = constant; + has_prev_error = false; + has_prev_succeded = true; + if (i < left_count) + continue; + } + + if (has_prev_succeded) { + i = i_prev; + t1 = t1_prev; + t2 = t2_prev; + constant = constant_prev; + has_error = false; + has_prev_succeded = false; + } + + if (!has_error) { + if (constant) { + t1 = 0.0f; + t2 = 0.0f; + } + + c = (int32_t)i; + values.push_back({ (float_t)frame, a[0], t2_old, t1 }); + t2_old = t2; + has_prev_error = false; + break; + } + + has_prev_error = true; + } + + if (has_prev_succeded) { + if (has_error) { + values.push_back({ (float_t)frame, a[0], t2_old, 0.0f }); + for (size_t j = 1; j < c; j++) + values.push_back({ (float_t)(int64_t)(frame + j), a[j] }); + t2_old = 0.0f; + } + else { + values.push_back({ (float_t)frame, a[0], t2_old, t1_prev }); + t2_old = t2_prev; + } + c = (int32_t)i; + } + else if (has_prev_error) { + values.push_back({ (float_t)frame, a[0], t2_old, 0.0f }); + t2_old = 0.0f; + c = 1; + } + + prev_frame = frame; + frame += c; + a += c; + left_count -= c; + } + + values.push_back({ (float_t)(int64_t)(count - 1), arr[count - 1], t2_old, 0.0f }); + + kft3* keys = values.data(); + size_t length = values.size(); + for (size_t i = 0; i < count; i++) { + float_t frame = (float_t)(int64_t)i; + + kft3* first_key = keys; + kft3* key = keys; + size_t _length = length; + size_t temp; + while (_length > 0) + if (frame < key[temp = _length / 2].frame) + _length = temp; + else { + key += temp + 1; + _length -= temp + 1; + } + + float_t val; + if (key == first_key) + val = first_key->value; + else if (key == &first_key[length]) + val = key[-1].value; + else + val = interpolate_linear_value(key[-1].value, key[0].value, + key[-1].frame, key[0].frame, frame); + + if (fabsf(val - arr[i]) > reverse_bias) + return 3; + } + + for (kft3& i : values) { + i.tangent1 = 0.0f; + i.tangent2 = 0.0f; + } + return 2; +} diff --git a/src/KKdLib/interpolation.hpp b/src/KKdLib/interpolation.hpp new file mode 100644 index 0000000..58c5352 --- /dev/null +++ b/src/KKdLib/interpolation.hpp @@ -0,0 +1,233 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "default.hpp" +#include "kf.hpp" +#include "vec.hpp" +#include + +inline float_t interpolate_linear_value(const float_t p1, const float_t p2, + const float_t f1, const float_t f2, const float_t f) { + if (p1 == p2) + return p1; + + float_t t = (f - f1) / (f2 - f1); + return (1.0f - t) * p1 + t * p2; +} + +inline vec2 interpolate_linear_value(const vec2 p1, const vec2 p2, + const vec2 f1, const vec2 f2, const vec2 f) { + if (p1 == p2) + return p1; + + __m128 _p1 = vec2::load_xmm(p1); + __m128 _p2 = vec2::load_xmm(p2); + __m128 _f1 = vec2::load_xmm(f1); + __m128 _f2 = vec2::load_xmm(f2); + __m128 _f = vec2::load_xmm(f); + + const __m128 _1 = vec4::load_xmm(1.0f); + + __m128 t = _mm_div_ps(_mm_sub_ps(_f, _f1), _mm_sub_ps(_f2, _f1)); + return vec2::store_xmm(_mm_add_ps(_mm_mul_ps(_p1, _mm_sub_ps(_1, t)), _mm_mul_ps(_p2, t))); +} + +inline vec3 interpolate_linear_value(const vec3 p1, const vec3 p2, + const vec3 f1, const vec3 f2, const vec3 f) { + if (p1 == p2) + return p1; + + __m128 _p1 = vec3::load_xmm(p1); + __m128 _p2 = vec3::load_xmm(p2); + __m128 _f1 = vec3::load_xmm(f1); + __m128 _f2 = vec3::load_xmm(f2); + __m128 _f = vec3::load_xmm(f); + + const __m128 _1 = vec4::load_xmm(1.0f); + + __m128 t = _mm_div_ps(_mm_sub_ps(_f, _f1), _mm_sub_ps(_f2, _f1)); + return vec3::store_xmm(_mm_add_ps(_mm_mul_ps(_p1, _mm_sub_ps(_1, t)), _mm_mul_ps(_p2, t))); +} + +inline vec4 interpolate_linear_value(const vec4 p1, const vec4 p2, + const vec4 f1, const vec4 f2, const vec4 f) { + if (p1 == p2) + return p1; + + __m128 _p1 = vec4::load_xmm(p1); + __m128 _p2 = vec4::load_xmm(p2); + __m128 _f1 = vec4::load_xmm(f1); + __m128 _f2 = vec4::load_xmm(f2); + __m128 _f = vec4::load_xmm(f); + + const __m128 _1 = vec4::load_xmm(1.0f); + + __m128 t = _mm_div_ps(_mm_sub_ps(_f, _f1), _mm_sub_ps(_f2, _f1)); + return vec4::store_xmm(_mm_add_ps(_mm_mul_ps(_p1, _mm_sub_ps(_1, t)), _mm_mul_ps(_p2, t))); +} + +inline std::vector interpolate_linear(float_t p1, float_t p2, size_t f1, size_t f2) { + size_t length = f2 - f1 + 1; + if (p1 == p2) + return std::vector(length, p1); + + std::vector arr(length); + float_t* a = arr.data(); + for (size_t i = 0, j = length; j; i++, j--, a++) + *a = interpolate_linear_value(p1, p2, + (float_t)f1, (float_t)f2, (float_t)(f1 + i)); + return arr; +} + +inline float_t interpolate_chs_value(const float_t p1, const float_t p2, + const float_t t1, const float_t t2, const float_t f1, const float_t f2, const float_t f) { + if (p1 == p2 && fabsf(t1) == 0.0f && fabsf(t2) == 0.0f) + return p1; + + float_t df = f2 - f1; + float_t t = (f - f1) / df; + float_t t_2 = t * t; + float_t t_3 = t_2 * t; + float_t t_23 = 3.0f * t_2; + float_t t_32 = 2.0f * t_3; + + float_t h00 = t_32 - t_23 + 1.0f; + float_t h01 = t_23 - t_32; + float_t h10 = t_3 - 2.0f * t_2 + t; + float_t h11 = t_3 - t_2; + + return h00 * p1 + h01 * p2 + h10 * (t1 * df) + h11 * (t2 * df); +} + +inline vec2 interpolate_chs_value(const vec2 p1, const vec2 p2, + const vec2 t1, const vec2 t2, const vec2 f1, const vec2 f2, const vec2 f) { + if (p1 == p2 && vec2::abs(t1) == 0.0f && vec2::abs(t2) == 0.0f) + return p1; + + __m128 _p1 = vec2::load_xmm(p1); + __m128 _p2 = vec2::load_xmm(p2); + __m128 _t1 = vec2::load_xmm(t1); + __m128 _t2 = vec2::load_xmm(t2); + __m128 _f1 = vec2::load_xmm(f1); + __m128 _f2 = vec2::load_xmm(f2); + __m128 _f = vec2::load_xmm(f); + + const __m128 _1 = vec4::load_xmm(1.0f); + const __m128 _2 = vec4::load_xmm(2.0f); + const __m128 _3 = vec4::load_xmm(3.0f); + + __m128 df = _mm_sub_ps(_f2, _f1); + __m128 t = _mm_div_ps(_mm_sub_ps(_f, _f1), df); + __m128 t_2 = _mm_mul_ps(t, t); + __m128 t_3 = _mm_mul_ps(t_2, t); + __m128 t_23 = _mm_mul_ps(_3, t_2); + __m128 t_32 = _mm_mul_ps(_2, t_3); + + __m128 h00 = _mm_add_ps(_mm_sub_ps(t_32, t_23), _1); + __m128 h01 = _mm_sub_ps(t_23, t_32); + __m128 h10 = _mm_add_ps(_mm_sub_ps(t_3, _mm_mul_ps(_2, t_2)), t); + __m128 h11 = _mm_sub_ps(t_3, t_2); + + _p1 = _mm_mul_ps(h00, _p1); + _p2 = _mm_mul_ps(h01, _p2); + _t1 = _mm_mul_ps(h10, _mm_mul_ps(_t1, df)); + _t2 = _mm_mul_ps(h11, _mm_mul_ps(_t2, df)); + return vec2::store_xmm(_mm_add_ps(_mm_add_ps(_p1, _p2), _mm_add_ps(_t1, _t2))); +} + +inline vec3 interpolate_chs_value(const vec3 p1, const vec3 p2, + const vec3 t1, const vec3 t2, const vec3 f1, const vec3 f2, const vec3 f) { + if (p1 == p2 && vec3::abs(t1) == 0.0f && vec3::abs(t2) == 0.0f) + return p1; + + __m128 _p1 = vec3::load_xmm(p1); + __m128 _p2 = vec3::load_xmm(p2); + __m128 _t1 = vec3::load_xmm(t1); + __m128 _t2 = vec3::load_xmm(t2); + __m128 _f1 = vec3::load_xmm(f1); + __m128 _f2 = vec3::load_xmm(f2); + __m128 _f = vec3::load_xmm(f); + + const __m128 _1 = vec4::load_xmm(1.0f); + const __m128 _2 = vec4::load_xmm(2.0f); + const __m128 _3 = vec4::load_xmm(3.0f); + + __m128 df = _mm_sub_ps(_f2, _f1); + __m128 t = _mm_div_ps(_mm_sub_ps(_f, _f1), df); + __m128 t_2 = _mm_mul_ps(t, t); + __m128 t_3 = _mm_mul_ps(t_2, t); + __m128 t_23 = _mm_mul_ps(_3, t_2); + __m128 t_32 = _mm_mul_ps(_2, t_3); + + __m128 h00 = _mm_add_ps(_mm_sub_ps(t_32, t_23), _1); + __m128 h01 = _mm_sub_ps(t_23, t_32); + __m128 h10 = _mm_add_ps(_mm_sub_ps(t_3, _mm_mul_ps(_2, t_2)), t); + __m128 h11 = _mm_sub_ps(t_3, t_2); + + _p1 = _mm_mul_ps(h00, _p1); + _p2 = _mm_mul_ps(h01, _p2); + _t1 = _mm_mul_ps(h10, _mm_mul_ps(_t1, df)); + _t2 = _mm_mul_ps(h11, _mm_mul_ps(_t2, df)); + return vec3::store_xmm(_mm_add_ps(_mm_add_ps(_p1, _p2), _mm_add_ps(_t1, _t2))); +} + +inline vec4 interpolate_chs_value(const vec4 p1, const vec4 p2, + const vec4 t1, const vec4 t2, const vec4 f1, const vec4 f2, const vec4 f) { + if (p1 == p2 && vec4::abs(t1) == 0.0f && vec4::abs(t2) == 0.0f) + return p1; + + __m128 _p1 = vec4::load_xmm(p1); + __m128 _p2 = vec4::load_xmm(p2); + __m128 _t1 = vec4::load_xmm(t1); + __m128 _t2 = vec4::load_xmm(t2); + __m128 _f1 = vec4::load_xmm(f1); + __m128 _f2 = vec4::load_xmm(f2); + __m128 _f = vec4::load_xmm(f); + + const __m128 _1 = vec4::load_xmm(1.0f); + const __m128 _2 = vec4::load_xmm(2.0f); + const __m128 _3 = vec4::load_xmm(3.0f); + + __m128 df = _mm_sub_ps(_f2, _f1); + __m128 t = _mm_div_ps(_mm_sub_ps(_f, _f1), df); + __m128 t_2 = _mm_mul_ps(t, t); + __m128 t_3 = _mm_mul_ps(t_2, t); + __m128 t_23 = _mm_mul_ps(_3, t_2); + __m128 t_32 = _mm_mul_ps(_2, t_3); + + __m128 h00 = _mm_add_ps(_mm_sub_ps(t_32, t_23), _1); + __m128 h01 = _mm_sub_ps(t_23, t_32); + __m128 h10 = _mm_add_ps(_mm_sub_ps(t_3, _mm_mul_ps(_2, t_2)), t); + __m128 h11 = _mm_sub_ps(t_3, t_2); + + _p1 = _mm_mul_ps(h00, _p1); + _p2 = _mm_mul_ps(h01, _p2); + _t1 = _mm_mul_ps(h10, _mm_mul_ps(_t1, df)); + _t2 = _mm_mul_ps(h11, _mm_mul_ps(_t2, df)); + return vec4::store_xmm(_mm_add_ps(_mm_add_ps(_p1, _p2), _mm_add_ps(_t1, _t2))); +} + +inline std::vector interpolate_chs(const float_t p1, const float_t p2, + const float_t t1, const float_t t2, const size_t f1, const size_t f2) { + size_t length = f2 - f1 + 1; + if (p1 == p2 && fabsf(t1) == 0.0f && fabsf(t2) == 0.0f) + return std::vector(length, p1); + + std::vector arr(length); + float_t* a = arr.data(); + for (size_t i = 0, j = length; j; i++, j--, a++) + *a = interpolate_chs_value(p1, p2, t1, t2, + (float_t)f1, (float_t)f2, (float_t)(f1 + i)); + return arr; +} + +extern void interpolate_chs_reverse_value(float_t* arr, size_t length, + float_t& t1, float_t& t2, size_t f1, size_t f2, size_t f); +extern void interpolate_chs_reverse(float_t* arr, size_t length, + float_t& t1, float_t& t2, size_t f1, size_t f2); +extern int32_t interpolate_chs_reverse_sequence( + std::vector& values_src, std::vector& values, bool fast = false); diff --git a/src/KKdLib/io/file_stream.cpp b/src/KKdLib/io/file_stream.cpp new file mode 100644 index 0000000..6fac13d --- /dev/null +++ b/src/KKdLib/io/file_stream.cpp @@ -0,0 +1,176 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "file_stream.hpp" +#include + +file_stream::file_stream() : stream() { + +} + +file_stream::~file_stream() { + close(); +} + +int file_stream::flush() { + return fflush(stream); +} + +void file_stream::close() { + if (!this) + return; + + if (stream) { + fflush(stream); + fclose(stream); + stream = 0; + } + stream::close(); +} + +bool file_stream::check_null() { + return !stream; +} + +bool file_stream::check_not_null() { + return !!stream; +} + +void file_stream::align_read(size_t align) { + int64_t position = _ftelli64(stream); + size_t temp_align = align - position % align; + if (align != temp_align) + _fseeki64(stream, position + temp_align, 0); +} + +void file_stream::align_write(size_t align) { + int64_t position = _ftelli64(stream); + size_t temp_align = align - position % align; + if (align != temp_align) { + memset(buf, 0, min_def(sizeof(buf), temp_align)); + size_t i = temp_align; + while (i >= sizeof(buf)) { + fwrite(buf, 1, sizeof(buf), stream); + i -= sizeof(buf); + } + + if (i > 0) + fwrite(buf, 1, i, stream); + } +} + +size_t file_stream::read(size_t count) { + return read((void*)0, count); +} + +size_t file_stream::read(void* buf, size_t count) { + if (!buf) { + int64_t act_count = 0; + while (count > 0) { + act_count += fread(this->buf, 1, min_def(count, sizeof(this->buf)), stream); + count -= sizeof(this->buf); + } + return act_count; + } + else + return fread(buf, 1, count, stream); +} + +size_t file_stream::read(void* buf, size_t size, size_t count) { + if (!buf) { + int64_t act_count = 0; + while (count > 0) { + act_count += fread(this->buf, size, min_def(count, sizeof(this->buf) / size), stream); + count -= sizeof(this->buf); + } + return act_count; + } + else + return fread(buf, size, count, stream); +} + +size_t file_stream::write(size_t count) { + return write((void*)0, count); +} + +size_t file_stream::write(const void* buf, size_t count) { + if (!buf) { + memset(this->buf, 0, sizeof(this->buf)); + int64_t act_count = 0; + while (count > 0) { + act_count += fwrite(this->buf, 1, min_def(count, sizeof(this->buf)), stream); + count -= sizeof(this->buf); + } + return act_count; + } + else + return fwrite(buf, 1, count, stream); +} + +size_t file_stream::write(const void* buf, size_t size, size_t count) { + if (!buf) { + memset(this->buf, 0, sizeof(this->buf)); + int64_t act_count = 0; + while (count > 0) { + act_count += fwrite(this->buf, 1, min_def(count, sizeof(this->buf) / size), stream); + count -= sizeof(this->buf) / size; + } + return act_count; + } + else + return fwrite(buf, size, count, stream); +} + +int32_t file_stream::read_char() { + return fgetc(stream); +} + +int32_t file_stream::write_char(char c) { + return fputc(c, stream); +} + +int64_t file_stream::get_length() { + if (stream) { + size_t temp = _ftelli64(stream); + _fseeki64(stream, 0, SEEK_END); + length = _ftelli64(stream); + _fseeki64(stream, temp, SEEK_SET); + } + else + length = 0; + return length; +} + +int64_t file_stream::get_position() { + return _ftelli64(stream); +} + +int32_t file_stream::set_position(int64_t pos, int32_t seek) { + return _fseeki64(stream, pos, seek); +} + +void file_stream::open(const char* path, const char* mode) { + close(); + + if (!path || !mode) + return; + + wchar_t* temp_path = utf8_to_utf16(path); + wchar_t* temp_mode = utf8_to_utf16(mode); + stream = _wfsopen(temp_path, temp_mode, _SH_DENYNO); + get_length(); + free_def(temp_path); + free_def(temp_mode); +} + +void file_stream::open(const wchar_t* path, const wchar_t* mode) { + close(); + + if (!path || !mode) + return; + + stream = _wfsopen(path, mode, _SH_DENYNO); + get_length(); +} diff --git a/src/KKdLib/io/file_stream.hpp b/src/KKdLib/io/file_stream.hpp new file mode 100644 index 0000000..896ee9e --- /dev/null +++ b/src/KKdLib/io/file_stream.hpp @@ -0,0 +1,49 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "stream.hpp" + +class file_stream : public stream { +private: + FILE* stream; + +public: + file_stream(); + virtual ~file_stream(); + + virtual int flush() override; + virtual void close() override; + virtual bool check_null() override; + virtual bool check_not_null() override; + + virtual void align_read(size_t align) override; + virtual void align_write(size_t align) override; + virtual size_t read(size_t count) override; + virtual size_t read(void* buf, size_t count) override; + virtual size_t read(void* buf, size_t size, size_t count) override; + virtual size_t write(size_t count) override; + virtual size_t write(const void* buf, size_t count) override; + virtual size_t write(const void* buf, size_t size, size_t count) override; + virtual int32_t read_char() override; + virtual int32_t write_char(char c) override; + virtual int64_t get_length() override; + virtual int64_t get_position() override; + virtual int32_t set_position(int64_t pos, int32_t seek) override; + + void open(const char* path, const char* mode); + void open(const wchar_t* path, const wchar_t* mode); + + template + size_t read_data(T& data) { + return read(&data, sizeof(T)); + } + + template + size_t write_data(const T& data) { + return write(&data, sizeof(T)); + } +}; diff --git a/src/KKdLib/io/memory_stream.cpp b/src/KKdLib/io/memory_stream.cpp new file mode 100644 index 0000000..d574f9a --- /dev/null +++ b/src/KKdLib/io/memory_stream.cpp @@ -0,0 +1,255 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "memory_stream.hpp" + +memory_stream::memory_stream() { + data.data = data.vec.begin(); +} + +memory_stream::~memory_stream() { + close(); +} + +int memory_stream::flush() { + return 0; +} + +void memory_stream::close() { + if (!this) + return; + + data.vec.clear(); + data.vec.shrink_to_fit(); + data.data = data.vec.begin(); + stream::close(); +} + +bool memory_stream::check_null() { + return !data.vec.size(); +} + +bool memory_stream::check_not_null() { + return !!data.vec.size(); +} + +void memory_stream::align_read(size_t align) { + size_t position = data.data - data.vec.begin(); + size_t temp_align = align - position % align; + if (align != temp_align) { + size_t pos = data.data - data.vec.begin(); + if (data.vec.size() < (size_t)(pos + temp_align)) { + size_t size = data.vec.size(); + data.vec.resize(pos + temp_align); + data.data = data.vec.begin() + pos; + memset(data.data._Ptr, 0, temp_align); + } + data.data += temp_align; + } +} + +void memory_stream::align_write(size_t align) { + size_t position = data.data - data.vec.begin(); + size_t temp_align = align - position % align; + if (align != temp_align) { + size_t pos = data.data - data.vec.begin(); + if (data.vec.size() < (size_t)(pos + temp_align)) { + size_t size = data.vec.size(); + data.vec.resize(pos + temp_align); + data.data = data.vec.begin() + pos; + memset(data.data._Ptr, 0, temp_align); + } + data.data += temp_align; + } +} + +size_t memory_stream::read(size_t count) { + return read((void*)0, count); +} + +size_t memory_stream::read(void* buf, size_t count) { + if (data.data >= data.vec.end()) + return EOF; + + size_t _count = data.vec.end() - data.data; + if (_count >= count) + _count = count; + if (buf) + memcpy(buf, data.data._Ptr, _count); + data.data += _count; + return _count; +} + +size_t memory_stream::read(void* buf, size_t size, size_t count) { + if (data.data >= data.vec.end()) + return EOF; + + size_t _count = data.vec.end() - data.data; + if (_count >= size * count) + _count = size * count; + if (buf) + memcpy(buf, data.data._Ptr, _count); + data.data += _count; + return _count; +} + +size_t memory_stream::write(size_t count) { + return write((void*)0, count); +} + +size_t memory_stream::write(const void* buf, size_t count) { + size_t pos = data.data - data.vec.begin(); + if (data.vec.size() < pos + count) { + data.vec.resize(pos + count); + data.data = data.vec.begin() + pos; + } + if (buf) + memcpy(data.data._Ptr, buf, count); + else + memset(data.data._Ptr, 0, count); + data.data += count; + return count; +} + +size_t memory_stream::write(const void* buf, size_t size, size_t count) { + size_t pos = data.data - data.vec.begin(); + size_t _count = size * count; + if (data.vec.size() < (size_t)(pos + _count)) { + data.vec.resize(pos + _count); + data.data = data.vec.begin() + pos; + } + if (buf) + memcpy(data.data._Ptr, buf, _count); + else + memset(data.data._Ptr, 0, _count); + data.data += _count; + return _count; +} + +int32_t memory_stream::read_char() { + if (data.data >= data.vec.end()) + return EOF; + return *(data.data++); +} + +int32_t memory_stream::write_char(char c) { + size_t pos = data.data - data.vec.begin(); + if (data.vec.size() < (size_t)(pos + 1)) { + size_t size = data.vec.size(); + data.vec.resize(pos + 1); + data.data = data.vec.begin() + pos; + } + *(data.data++) = c; + return 0; +} + +int64_t memory_stream::get_length() { + length = data.vec.size(); + return length; +} + +int64_t memory_stream::get_position() { + return data.data - data.vec.begin(); +} + +int32_t memory_stream::set_position(int64_t pos, int32_t seek) { + switch (seek) { + case SEEK_SET: { + if (pos < 0) + return EOF; + + if (data.vec.size() < (size_t)pos) { + size_t size = data.vec.size(); + data.vec.resize(pos); + memset(data.vec.data() + size, 0, pos - size); + } + data.data = data.vec.begin() + pos; + } return 0; + case SEEK_CUR: { + if (pos > 0) { + size_t _pos = data.data - data.vec.begin(); + if (data.vec.size() < (size_t)(_pos + pos)) { + size_t size = data.vec.size(); + data.vec.resize(_pos + pos); + memset(data.vec.data() + size, 0, _pos + pos - size); + data.data = data.vec.begin() + _pos; + } + data.data += pos; + } + else if (pos < 0) { + if (data.data - data.vec.begin() < -pos) + return EOF; + else + data.data += pos; + } + } return 0; + case SEEK_END: { + if (pos < 0) + break; + + if (data.vec.size() < (size_t)pos) + break; + + data.data = data.vec.end() - pos; + } return 0; + } + return EOF; +} + +void memory_stream::open() { + close(); + + length = 0; +} + +void memory_stream::open(const void* data, size_t size) { + close(); + + if (!size) { + length = 0; + return; + } + + this->data.vec.clear(); + this->data.vec.resize(size); + if (this->data.vec.data()) + if (data) + memcpy(this->data.vec.data(), data, size); + else + memset(this->data.vec.data(), 0, size); + this->data.data = this->data.vec.begin(); + length = size; +} + +void memory_stream::open(std::vector& data) { + close(); + + if (!data.size()) { + length = 0; + return; + } + + this->data.vec = data; + this->data.data = this->data.vec.begin(); + length = data.size(); +} + +void memory_stream::copy(void** data, size_t* size) { + if (!this || !data || !size) + return; + + *size = this->data.vec.size(); + *data = force_malloc(*size); + memcpy(*data, this->data.vec.data(), *size); +} + +void memory_stream::copy(std::vector& data) { + if (!this) + return; + + size_t length = this->data.vec.size(); + data.resize(length); + memcpy(data.data(), this->data.vec.data(), length); +} diff --git a/src/KKdLib/io/memory_stream.hpp b/src/KKdLib/io/memory_stream.hpp new file mode 100644 index 0000000..cd398c7 --- /dev/null +++ b/src/KKdLib/io/memory_stream.hpp @@ -0,0 +1,55 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "stream.hpp" + +class memory_stream : public stream { +private: + struct data { + std::vector::iterator data; + std::vector vec; + } data; + +public: + memory_stream(); + virtual ~memory_stream(); + + virtual int flush() override; + virtual void close() override; + virtual bool check_null() override; + virtual bool check_not_null() override; + + virtual void align_read(size_t align) override; + virtual void align_write(size_t align) override; + virtual size_t read(size_t count) override; + virtual size_t read(void* buf, size_t count) override; + virtual size_t read(void* buf, size_t size, size_t count) override; + virtual size_t write(size_t count) override; + virtual size_t write(const void* buf, size_t count) override; + virtual size_t write(const void* buf, size_t size, size_t count) override; + virtual int32_t read_char() override; + virtual int32_t write_char(char c) override; + virtual int64_t get_length() override; + virtual int64_t get_position() override; + virtual int32_t set_position(int64_t pos, int32_t seek) override; + + void open(); + void open(const void* data, size_t size); + void open(std::vector& data); + void copy(void** data, size_t* size); + void copy(std::vector& data); + + template + size_t read_data(T& data) { + return read(&data, sizeof(T)); + } + + template + size_t write_data(const T& data) { + return write(&data, sizeof(T)); + } +}; diff --git a/src/KKdLib/io/path.cpp b/src/KKdLib/io/path.cpp new file mode 100644 index 0000000..7c9dd39 --- /dev/null +++ b/src/KKdLib/io/path.cpp @@ -0,0 +1,395 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "path.hpp" +#include "../str_utils.hpp" + +bool path_check_path_exists(const char* path) { + wchar_t* path_temp = utf8_to_utf16(path); + DWORD ftyp = GetFileAttributesW(path_temp); + free_def(path_temp); + if (ftyp == INVALID_FILE_ATTRIBUTES) + return false; + else + return true; +} + +bool path_check_path_exists(const wchar_t* path) { + DWORD ftyp = GetFileAttributesW(path); + if (ftyp == INVALID_FILE_ATTRIBUTES) + return false; + else + return true; +} + +bool path_check_file_exists(const char* path) { + wchar_t* path_temp = utf8_to_utf16(path); + DWORD ftyp = GetFileAttributesW(path_temp); + free_def(path_temp); + if (ftyp == INVALID_FILE_ATTRIBUTES) + return false; + else + return ftyp & FILE_ATTRIBUTE_DIRECTORY ? false : true; +} + +bool path_check_file_exists(const wchar_t* path) { + DWORD ftyp = GetFileAttributesW(path); + if (ftyp == INVALID_FILE_ATTRIBUTES) + return false; + else + return ftyp & FILE_ATTRIBUTE_DIRECTORY ? false : true; +} + +bool path_check_directory_exists(const char* path) { + wchar_t* path_temp = utf8_to_utf16(path); + DWORD ftyp = GetFileAttributesW(path_temp); + free_def(path_temp); + if (ftyp == INVALID_FILE_ATTRIBUTES) + return false; + else + return ftyp & FILE_ATTRIBUTE_DIRECTORY ? true : false; +} + +bool path_check_directory_exists(const wchar_t* path) { + DWORD ftyp = GetFileAttributesW(path); + if (ftyp == INVALID_FILE_ATTRIBUTES) + return false; + else + return ftyp & FILE_ATTRIBUTE_DIRECTORY ? true : false; +} + +std::vector path_get_files(const char* path) { + wchar_t* dir_temp = utf8_to_utf16(path); + size_t dir_len = utf16_length(dir_temp); + if (!dir_temp) + return {}; + + std::wstring dir; + dir.assign(dir_temp, dir_len); + if (dir.size() && dir.back() != L'\\') + dir.push_back(L'\\'); + dir.push_back(L'*'); + free_def(dir_temp); + + WIN32_FIND_DATAW fdata = {}; + HANDLE h = FindFirstFileW(dir.c_str(), &fdata); + if (h == INVALID_HANDLE_VALUE) + return {}; + + std::vector files; + do { + if (fdata.dwFileAttributes & FILE_ATTRIBUTE_DIRECTORY) + continue; + + char* file_temp = utf16_to_utf8(fdata.cFileName); + if (file_temp) { + files.push_back(file_temp); + free(file_temp); + } + } while (FindNextFileW(h, &fdata)); + FindClose(h); + return files; +} + +std::vector path_get_files(const wchar_t* path) { + size_t dir_len = utf16_length(path); + + std::wstring dir; + dir.assign(path, dir_len); + if (dir.size() && dir.back() != L'\\') + dir.push_back(L'\\'); + dir.push_back(L'*'); + + WIN32_FIND_DATAW fdata = {}; + HANDLE h = FindFirstFileW(dir.c_str(), &fdata); + if (h == INVALID_HANDLE_VALUE) + return {}; + + std::vector files; + do { + if (fdata.dwFileAttributes & FILE_ATTRIBUTE_DIRECTORY) + continue; + + files.push_back(fdata.cFileName); + } while (FindNextFileW(h, &fdata)); + FindClose(h); + return files; +} + +std::vector path_get_directories( + const char* path, const char** exclude_list, size_t exclude_count) { + wchar_t* dir_temp = utf8_to_utf16(path); + size_t dir_len = utf16_length(dir_temp); + if (!dir_temp) + return {}; + + std::wstring dir; + std::wstring temp; + dir.assign(dir_temp, dir_len); + if (dir.size() && dir.back() != L'\\') + dir.push_back(L'\\'); + temp.assign(dir); + dir.push_back(L'*'); + free_def(dir_temp); + + WIN32_FIND_DATAW fdata = {}; + HANDLE h = FindFirstFileW(dir.c_str(), &fdata); + if (h == INVALID_HANDLE_VALUE) + return {}; + + std::vector directories; + do { + if (!(fdata.dwFileAttributes & FILE_ATTRIBUTE_DIRECTORY) + || !str_utils_compare(fdata.cFileName, L".") + || !str_utils_compare(fdata.cFileName, L"..")) + continue; + + if (exclude_list && exclude_count) { + size_t len = utf16_length(fdata.cFileName); + temp.append(fdata.cFileName, len); + std::string temp_utf8 = utf16_to_utf8(temp); + + bool exclude = false; + for (size_t i = 0; i < exclude_count; i++) + if (!temp_utf8.compare(exclude_list[i])) { + exclude = true; + break; + } + temp.resize(temp.size() - len); + + if (exclude) + continue; + } + + char* directory_temp = utf16_to_utf8(fdata.cFileName); + if (directory_temp) + directories.push_back(directory_temp); + free_def(directory_temp); + } while (FindNextFileW(h, &fdata)); + FindClose(h); + return directories; +} + +std::vector path_get_directories( + const wchar_t* path, wchar_t** exclude_list, size_t exclude_count) { + size_t dir_len = utf16_length(path); + + std::wstring dir; + std::wstring temp; + dir.assign(path, dir_len); + if (dir.size() && dir.back() != L'\\') + dir.push_back(L'\\'); + temp.assign(dir); + dir.push_back(L'*'); + + WIN32_FIND_DATAW fdata = {}; + HANDLE h = FindFirstFileW(dir.c_str(), &fdata); + if (h == INVALID_HANDLE_VALUE) + return {}; + + std::vector directories; + do { + if (!(fdata.dwFileAttributes & FILE_ATTRIBUTE_DIRECTORY) + || !str_utils_compare(fdata.cFileName, L".") + || !str_utils_compare(fdata.cFileName, L"..")) + continue; + + if (exclude_list && exclude_count) { + size_t len = utf16_length(fdata.cFileName); + temp.append(fdata.cFileName, len); + + bool exclude = false; + for (size_t i = 0; i < exclude_count; i++) + if (!temp.compare(exclude_list[i])) { + exclude = true; + break; + } + temp.resize(temp.size() - len); + + if (exclude) + continue; + } + + directories.push_back(fdata.cFileName); + } while (FindNextFileW(h, &fdata)); + FindClose(h); + return directories; +} + +std::vector path_get_directories_recursive( + const char* path, const char** exclude_list, size_t exclude_count) { + wchar_t* dir_temp = utf8_to_utf16(path); + size_t dir_len = utf16_length(dir_temp); + if (!dir_temp) + return {}; + + std::wstring dir; + std::wstring temp; + dir.assign(dir_temp, dir_len); + if (dir.size() && dir.back() != L'\\') + dir.push_back(L'\\'); + temp.assign(dir); + dir.push_back(L'*'); + free_def(dir_temp); + + WIN32_FIND_DATAW fdata = {}; + HANDLE h = FindFirstFileW(dir.c_str(), &fdata); + if (h == INVALID_HANDLE_VALUE) + return {}; + + std::vector temp_vec; + do { + if (!(fdata.dwFileAttributes & FILE_ATTRIBUTE_DIRECTORY) + || !str_utils_compare(fdata.cFileName, L".") + || !str_utils_compare(fdata.cFileName, L"..")) + continue; + + if (exclude_list && exclude_count) { + size_t len = utf16_length(fdata.cFileName); + temp.append(fdata.cFileName, len); + std::string temp_utf8 = utf16_to_utf8(temp); + + bool exclude = false; + for (size_t i = 0; i < exclude_count; i++) + if (!temp_utf8.compare(exclude_list[i])) { + exclude = true; + break; + } + temp.resize(temp.size() - len); + + if (exclude) + continue; + } + + char* directory_temp = utf16_to_utf8(fdata.cFileName); + temp_vec.push_back(directory_temp); + free_def(directory_temp); + } while (FindNextFileW(h, &fdata)); + FindClose(h); + + size_t max_len = 0; + for (std::string& i : temp_vec) { + size_t len = i.size(); + if (max_len < len) + max_len = len; + } + + std::vector directories; + std::string path_temp; + path_temp.assign(path); + path_temp += '\\'; + for (std::string& i : temp_vec) { + path_temp.append(i); + std::vector temp = path_get_directories_recursive( + path_temp.c_str(), exclude_list, exclude_count); + path_temp.resize(path_temp.size() - i.size()); + + directories.push_back(i); + + if (temp.size() < 1) + continue; + + max_len = 0; + for (std::string& j : temp) { + size_t len = j.size(); + if (max_len < len) + max_len = len; + } + + if (i.size()) { + std::string sub_path_temp; + sub_path_temp.assign(i); + sub_path_temp += '\\'; + for (std::string& j : temp) + directories.push_back(sub_path_temp + j); + } + } + return directories; +} + +std::vector path_get_directories_recursive( + const wchar_t* path, const wchar_t** exclude_list, size_t exclude_count) { + size_t dir_len = utf16_length(path); + + std::wstring dir; + std::wstring temp; + dir.assign(path, dir_len); + if (dir.size() && dir.back() != L'\\') + dir.push_back(L'\\'); + temp.assign(dir); + dir.push_back(L'*'); + + WIN32_FIND_DATAW fdata = {}; + HANDLE h = FindFirstFileW(dir.c_str(), &fdata); + if (h == INVALID_HANDLE_VALUE) + return {}; + + std::vector temp_vec; + do { + if (!(fdata.dwFileAttributes & FILE_ATTRIBUTE_DIRECTORY) + || !str_utils_compare(fdata.cFileName, L".") + || !str_utils_compare(fdata.cFileName, L"..")) + continue; + + if (exclude_list && exclude_count) { + size_t len = utf16_length(fdata.cFileName); + temp.append(fdata.cFileName, len); + + bool exclude = false; + for (size_t i = 0; i < exclude_count; i++) + if (!temp.compare(exclude_list[i])) { + exclude = true; + break; + } + temp.resize(temp.size() - len); + + if (exclude) + continue; + } + + std::wstring directory = std::wstring(fdata.cFileName); + temp_vec.push_back(directory); + } while (FindNextFileW(h, &fdata)); + FindClose(h); + + size_t max_len = 0; + for (std::wstring& i : temp_vec) { + size_t len = i.size(); + if (max_len < len) + max_len = len; + } + + std::vector directories; + std::wstring path_temp; + path_temp.assign(path); + path_temp.push_back(L'\\'); + for (std::wstring& i : temp_vec) { + path_temp.append(i); + std::vector temp = path_get_directories_recursive( + path_temp.c_str(), exclude_list, exclude_count); + path_temp.resize(path_temp.size() - i.size()); + + directories.push_back(i); + + if (temp.size() < 1) + continue; + + max_len = 0; + for (std::wstring& j : temp) { + size_t len = j.size(); + if (max_len < len) + max_len = len; + } + + if (i.size()) { + std::wstring sub_path_temp; + sub_path_temp.assign(i); + sub_path_temp.push_back(L'\\'); + for (std::wstring& j : temp) + directories.push_back(sub_path_temp + j); + } + } + return directories; +} diff --git a/src/KKdLib/io/path.hpp b/src/KKdLib/io/path.hpp new file mode 100644 index 0000000..021ae10 --- /dev/null +++ b/src/KKdLib/io/path.hpp @@ -0,0 +1,27 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include +#include +#include "../default.hpp" + +extern bool path_check_path_exists(const char* path); +extern bool path_check_path_exists(const wchar_t* path); +extern bool path_check_file_exists(const char* path); +extern bool path_check_file_exists(const wchar_t* path); +extern bool path_check_directory_exists(const char* path); +extern bool path_check_directory_exists(const wchar_t* path); +extern std::vector path_get_files(const char* path); +extern std::vector path_get_files(const wchar_t* path); +extern std::vector path_get_directories( + const char* path, const char** exclude_list = 0, size_t exclude_count = 0); +extern std::vector path_get_directories( + const wchar_t* path, const wchar_t** exclude_list = 0, size_t exclude_count = 0); +extern std::vector path_get_directories_recursive( + const char* path, const char** exclude_list = 0, size_t exclude_count = 0); +extern std::vector path_get_directories_recursive( + const wchar_t* path, const wchar_t** exclude_list = 0, size_t exclude_count = 0); \ No newline at end of file diff --git a/src/KKdLib/io/stream.cpp b/src/KKdLib/io/stream.cpp new file mode 100644 index 0000000..b08e92d --- /dev/null +++ b/src/KKdLib/io/stream.cpp @@ -0,0 +1,716 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "stream.hpp" + +stream::stream() : buf(), length(), big_endian() { + +} + +stream::~stream() { + close(); +} + +void stream::close() { + if (!this) + return; + + memset(buf, 0, sizeof(buf)); + length = 0; + big_endian = false; + position_stack.clear(); + position_stack.shrink_to_fit(); +} + +int32_t stream::position_push(int64_t pos, int32_t seek) { + position_stack.push_back(get_position()); + return set_position(pos, seek); +} + +void stream::position_pop() { + if (position_stack.size() < 1) + return; + + int64_t position = position_stack.back(); + position_stack.pop_back(); + set_position(position, SEEK_SET); + flush(); +} + +int8_t stream::read_int8_t() { + int32_t c = read_char(); + if (c != EOF) + return (int8_t)c; + return 0; +} + +uint8_t stream::read_uint8_t() { + int32_t c = read_char(); + if (c != EOF) + return (uint8_t)c; + return 0; +} + +void stream::write_int8_t(int8_t val) { + write_char((char)val); +} + +void stream::write_uint8_t(uint8_t val) { + write_char((char)val); +} + +std::string stream::read_string(size_t length) { + std::string str = std::string(length, 0); + read(&str.front(), sizeof(char) * length); + return str; +} + +std::wstring stream::read_wstring(size_t length) { + std::wstring str = std::wstring(length, 0); + read(&str.front(), sizeof(wchar_t) * length); + return str; +} + +std::string stream::read_string_null_terminated() { + int64_t offset = get_position(); + + size_t length = read_utf8_string_null_terminated_offset_length(offset); + if (length) + return read_string(length); + + std::string str = {}; + return str; +} + +std::wstring stream::read_wstring_null_terminated() { + int64_t offset = get_position(); + + size_t length = read_utf16_string_null_terminated_offset_length(offset); + if (length) + return read_wstring(length); + + std::wstring str = {}; + return str; +} + +std::string stream::read_string_null_terminated_offset(int64_t offset) { + if (offset) { + size_t length = read_utf8_string_null_terminated_offset_length(offset); + if (length) { + position_push(offset, SEEK_SET); + std::string str = read_string(length); + position_pop(); + return str; + } + } + + std::string str = {}; + return str; +} + +std::wstring stream::read_wstring_null_terminated_offset(int64_t offset) { + if (offset) { + size_t length = read_utf16_string_null_terminated_offset_length(offset); + if (length) { + position_push(offset, SEEK_SET); + std::wstring str = read_wstring(length); + position_pop(); + return str; + } + } + + std::wstring str = {}; + return str; +} + +char* stream::read_utf8_string_null_terminated() { + int64_t offset = get_position(); + return read_utf8_string_null_terminated_offset(offset); +} + +wchar_t* stream::read_utf16_string_null_terminated() { + int64_t offset = get_position(); + return read_utf16_string_null_terminated_offset(offset); +} + +char* stream::read_utf8_string_null_terminated_offset(int64_t offset) { + size_t len = read_utf8_string_null_terminated_offset_length(offset); + if (!len) { + return 0; + } + + char* str = force_malloc(len + 1); + position_push(offset, SEEK_SET); + read(str, len); + str[len] = 0; + position_pop(); + return str; +} + +wchar_t* stream::read_utf16_string_null_terminated_offset(int64_t offset) { + size_t len = read_utf16_string_null_terminated_offset_length(offset); + if (!len) { + position_pop(); + return 0; + } + + wchar_t* str = force_malloc(len + 1); + position_push(offset, SEEK_SET); + read(str, sizeof(wchar_t) * len); + str[len] = 0; + position_pop(); + return str; +} + +size_t stream::read_utf8_string_null_terminated_length() { + int64_t offset = get_position(); + return read_utf8_string_null_terminated_offset_length(offset); +} + +size_t stream::read_utf16_string_null_terminated_length() { + int64_t offset = get_position(); + return read_utf16_string_null_terminated_offset_length(offset); +} + +size_t stream::read_utf8_string_null_terminated_offset_length(int64_t offset) { + position_push(offset, SEEK_SET); + + size_t len = 0; + int32_t c; + while ((c = read_char()) != EOF && c != 0) + len++; + + position_pop(); + return len; +} + +size_t stream::read_utf16_string_null_terminated_offset_length(int64_t offset) { + position_push(offset, SEEK_SET); + + size_t len = 0; + int32_t c0, c1; + while ((c0 = read_char()) != EOF && (c1 = read_char()) != EOF + && (((c0 & 0xFF) | ((c1 & 0xFF) << 8)) != 0)) + len++; + + position_pop(); + return len; +} + +int16_t stream::read_int16_t() { + read(buf, sizeof(int16_t)); + return *(int16_t*)buf; +} + +int16_t stream::read_int16_t_reverse_endianness() { + read(buf, sizeof(int16_t)); + int16_t val; + if (big_endian) + val = load_reverse_endianness_int16_t(buf); + else + val = *(int16_t*)buf; + return val; +} + +int16_t stream::read_int16_t_reverse_endianness(bool big_endian) { + read(buf, sizeof(int16_t)); + int16_t val; + if (big_endian) + val = load_reverse_endianness_int16_t(buf); + else + val = *(int16_t*)buf; + return val; +} + +void stream::write_int16_t(int16_t val) { + *(int16_t*)buf = val; + write(buf, sizeof(int16_t)); +} + +void stream::write_int16_t_reverse_endianness(int16_t val) { + if (big_endian) + store_reverse_endianness_int16_t(buf, val); + else + *(int16_t*)buf = val; + write(buf, sizeof(int16_t)); +} + +void stream::write_int16_t_reverse_endianness(int16_t val, bool big_endian) { + if (big_endian) + store_reverse_endianness_int16_t(buf, val); + else + *(int16_t*)buf = val; + write(buf, sizeof(int16_t)); +} + +uint16_t stream::read_uint16_t() { + read(buf, sizeof(uint16_t)); + return *(uint16_t*)buf; +} + +uint16_t stream::read_uint16_t_reverse_endianness() { + read(buf, sizeof(uint16_t)); + uint16_t val; + if (big_endian) + val = load_reverse_endianness_uint16_t(buf); + else + val = *(uint16_t*)buf; + return val; +} + +uint16_t stream::read_uint16_t_reverse_endianness(bool big_endian) { + read(buf, sizeof(uint16_t)); + uint16_t val; + if (big_endian) + val = load_reverse_endianness_uint16_t(buf); + else + val = *(uint16_t*)buf; + return val; +} + +void stream::write_uint16_t(uint16_t val) { + *(uint16_t*)buf = val; + write(buf, sizeof(uint16_t)); +} + +void stream::write_uint16_t_reverse_endianness(uint16_t val) { + if (big_endian) + store_reverse_endianness_uint16_t(buf, val); + else + *(uint16_t*)buf = val; + write(buf, sizeof(uint16_t)); +} + +void stream::write_uint16_t_reverse_endianness(uint16_t val, bool big_endian) { + if (big_endian) + store_reverse_endianness_uint16_t(buf, val); + else + *(uint16_t*)buf = val; + write(buf, sizeof(uint16_t)); +} + +int32_t stream::read_int32_t() { + read(buf, sizeof(int32_t)); + return *(int32_t*)buf; +} + +int32_t stream::read_int32_t_reverse_endianness() { + read(buf, sizeof(int32_t)); + int32_t val; + if (big_endian) + val = load_reverse_endianness_int32_t(buf); + else + val = *(int32_t*)buf; + return val; +} + +int32_t stream::read_int32_t_reverse_endianness(bool big_endian) { + read(buf, sizeof(int32_t)); + int32_t val; + if (big_endian) + val = load_reverse_endianness_int32_t(buf); + else + val = *(int32_t*)buf; + return val; +} + +void stream::write_int32_t(int32_t val) { + *(int32_t*)buf = val; + write(buf, sizeof(int32_t)); +} + +void stream::write_int32_t_reverse_endianness(int32_t val) { + if (big_endian) + store_reverse_endianness_int32_t(buf, val); + else + *(int32_t*)buf = val; + write(buf, sizeof(int32_t)); +} + +void stream::write_int32_t_reverse_endianness(int32_t val, bool big_endian) { + if (big_endian) + store_reverse_endianness_int32_t(buf, val); + else + *(int32_t*)buf = val; + write(buf, sizeof(int32_t)); +} + +uint32_t stream::read_uint32_t() { + read(buf, sizeof(uint32_t)); + return *(uint32_t*)buf; +} + +uint32_t stream::read_uint32_t_reverse_endianness() { + read(buf, sizeof(uint32_t)); + uint32_t val; + if (big_endian) + val = load_reverse_endianness_uint32_t(buf); + else + val = *(uint32_t*)buf; + return val; +} + +uint32_t stream::read_uint32_t_reverse_endianness(bool big_endian) { + read(buf, sizeof(uint32_t)); + uint32_t val; + if (big_endian) + val = load_reverse_endianness_uint32_t(buf); + else + val = *(uint32_t*)buf; + return val; +} + +void stream::write_uint32_t(uint32_t val) { + *(uint32_t*)buf = val; + write(buf, sizeof(uint32_t)); +} + +void stream::write_uint32_t_reverse_endianness(uint32_t val) { + if (big_endian) + store_reverse_endianness_uint32_t(buf, val); + else + *(uint32_t*)buf = val; + write(buf, sizeof(uint32_t)); +} + +void stream::write_uint32_t_reverse_endianness(uint32_t val, bool big_endian) { + if (big_endian) + store_reverse_endianness_uint32_t(buf, val); + else + *(uint32_t*)buf = val; + write(buf, sizeof(uint32_t)); +} + +int64_t stream::read_int64_t() { + read(buf, sizeof(int64_t)); + return *(int64_t*)buf; +} + +int64_t stream::read_int64_t_reverse_endianness() { + read(buf, sizeof(int64_t)); + int64_t val; + if (big_endian) + val = load_reverse_endianness_int64_t(buf); + else + val = *(int64_t*)buf; + return val; +} + +int64_t stream::read_int64_t_reverse_endianness(bool big_endian) { + read(buf, sizeof(int64_t)); + int64_t val; + if (big_endian) + val = load_reverse_endianness_int64_t(buf); + else + val = *(int64_t*)buf; + return val; +} + +void stream::write_int64_t(int64_t val) { + *(int64_t*)buf = val; + write(buf, sizeof(int64_t)); +} + +void stream::write_int64_t_reverse_endianness(int64_t val) { + if (big_endian) + store_reverse_endianness_int64_t(buf, val); + else + *(int64_t*)buf = val; + write(buf, sizeof(int64_t)); +} + +void stream::write_int64_t_reverse_endianness(int64_t val, bool big_endian) { + if (big_endian) + store_reverse_endianness_int64_t(buf, val); + else + *(int64_t*)buf = val; + write(buf, sizeof(int64_t)); +} + +uint64_t stream::read_uint64_t() { + read(buf, sizeof(uint64_t)); + return *(uint64_t*)buf; +} + +uint64_t stream::read_uint64_t_reverse_endianness() { + read(buf, sizeof(uint64_t)); + uint64_t val; + if (big_endian) + val = load_reverse_endianness_uint64_t(buf); + else + val = *(uint64_t*)buf; + return val; +} + +uint64_t stream::read_uint64_t_reverse_endianness(bool big_endian) { + read(buf, sizeof(uint64_t)); + uint64_t val; + if (big_endian) + val = load_reverse_endianness_uint64_t(buf); + else + val = *(uint64_t*)buf; + return val; +} + +void stream::write_uint64_t(uint64_t val) { + *(uint64_t*)buf = val; + write(buf, sizeof(uint64_t)); +} + +void stream::write_uint64_t_reverse_endianness(uint64_t val) { + if (big_endian) + store_reverse_endianness_uint64_t(buf, val); + else + *(uint64_t*)buf = val; + write(buf, sizeof(uint64_t)); +} + +void stream::write_uint64_t_reverse_endianness(uint64_t val, bool big_endian) { + if (big_endian) + store_reverse_endianness_uint64_t(buf, val); + else + *(uint64_t*)buf = val; + write(buf, sizeof(uint64_t)); +} + +half_t stream::read_half_t() { + read(buf, sizeof(half_t)); + return *(half_t*)buf; +} + +half_t stream::read_half_t_reverse_endianness() { + read(buf, sizeof(half_t)); + half_t val; + if (big_endian) + val = load_reverse_endianness_half_t(buf); + else + val = *(half_t*)buf; + return val; +} + +half_t stream::read_half_t_reverse_endianness(bool big_endian) { + read(buf, sizeof(half_t)); + half_t val; + if (big_endian) + val = load_reverse_endianness_half_t(buf); + else + val = *(half_t*)buf; + return val; +} + +void stream::write_half_t(half_t val) { + *(half_t*)buf = val; + write(buf, sizeof(half_t)); +} + +void stream::write_half_t_reverse_endianness(half_t val) { + if (big_endian) + store_reverse_endianness_half_t(buf, val); + else + *(half_t*)buf = val; + write(buf, sizeof(half_t)); +} + +void stream::write_half_t_reverse_endianness(half_t val, bool big_endian) { + if (big_endian) + store_reverse_endianness_half_t(buf, val); + else + *(half_t*)buf = val; + write(buf, sizeof(half_t)); +} + +float_t stream::read_float_t() { + read(buf, sizeof(float_t)); + return *(float_t*)buf; +} + +float_t stream::read_float_t_reverse_endianness() { + read(buf, sizeof(float_t)); + float_t val; + if (big_endian) + val = load_reverse_endianness_float_t(buf); + else + val = *(float_t*)buf; + return val; +} + +float_t stream::read_float_t_reverse_endianness(bool big_endian) { + read(buf, sizeof(float_t)); + float_t val; + if (big_endian) + val = load_reverse_endianness_float_t(buf); + else + val = *(float_t*)buf; + return val; +} + +void stream::write_float_t(float_t val) { + *(float_t*)buf = val; + write(buf, sizeof(float_t)); +} + +void stream::write_float_t_reverse_endianness(float_t val) { + if (big_endian) + store_reverse_endianness_float_t(buf, val); + else + *(float_t*)buf = val; + write(buf, sizeof(float_t)); +} + +void stream::write_float_t_reverse_endianness(float_t val, bool big_endian) { + if (big_endian) + store_reverse_endianness_float_t(buf, val); + else + *(float_t*)buf = val; + write(buf, sizeof(float_t)); +} + +double_t stream::read_double_t() { + read(buf, sizeof(double_t)); + return *(double_t*)buf; +} + +double_t stream::read_double_t_reverse_endianness() { + read(buf, sizeof(double_t)); + double_t val; + if (big_endian) + val = load_reverse_endianness_double_t(buf); + else + val = *(double_t*)buf; + return val; +} + +double_t stream::read_double_t_reverse_endianness(bool big_endian) { + read(buf, sizeof(double_t)); + double_t val; + if (big_endian) + val = load_reverse_endianness_double_t(buf); + else + val = *(double_t*)buf; + return val; +} + +void stream::write_double_t(double_t val) { + *(double_t*)buf = val; + write(buf, sizeof(double_t)); +} + +void stream::write_double_t_reverse_endianness(double_t val) { + if (big_endian) + store_reverse_endianness_double_t(buf, val); + else + *(double_t*)buf = val; + write(buf, sizeof(double_t)); +} + +void stream::write_double_t_reverse_endianness(double_t val, bool big_endian) { + if (big_endian) + store_reverse_endianness_double_t(buf, val); + else + *(double_t*)buf = val; + write(buf, sizeof(double_t)); +} + +void stream::write_string(std::string& str) { + write(str.c_str(), str.size()); +} + +void stream::write_string(std::string&& str) { + write(str.c_str(), str.size()); +} + +void stream::write_wstring(std::wstring& str) { + write(str.c_str(), sizeof(wchar_t) * str.size()); +} + +void stream::write_wstring(std::wstring&& str) { + write(str.c_str(), sizeof(wchar_t) * str.size()); +} + +void stream::write_string_null_terminated(std::string& str) { + write(str.c_str(), str.size()); + write_uint8_t(0); +} + +void stream::write_string_null_terminated(std::string&& str) { + write_string_null_terminated(str); +} + +void stream::write_wstring_null_terminated(std::wstring& str) { + write(str.c_str(), sizeof(wchar_t) * str.size()); + write_uint16_t(0); +} + +void stream::write_wstring_null_terminated(std::wstring&& str) { + write_wstring_null_terminated(*(std::wstring*)&str); +} + +void stream::write_utf8_string(const char* str) { + write(str, utf8_length(str)); +} + +void stream::write_utf16_string(const wchar_t* str) { + write(str, sizeof(wchar_t) * utf16_length(str)); +} + +void stream::write_utf8_string_null_terminated(const char* str) { + write(str, utf8_length(str)); + write_uint8_t(0); +} + +void stream::write_utf16_string_null_terminated(const wchar_t* str) { + write(str, sizeof(wchar_t) * utf16_length(str)); + write_uint16_t(0); +} + +int64_t stream::read_offset(int64_t offset, bool is_x) { + int64_t val; + if (!is_x) { + val = read_uint32_t_reverse_endianness(); + if (val) + val -= offset; + } + else { + align_read(0x08); + val = read_int64_t_reverse_endianness(); + } + return val; +} + +int64_t stream::read_offset_f2(int64_t offset) { + int64_t val = read_uint32_t_reverse_endianness(); + if (val) + val -= offset; + return val; +} + +int64_t stream::read_offset_x() { + align_read(0x08); + int64_t val = read_int64_t_reverse_endianness(); + return val; +} + +void stream::write_offset(int64_t val, int64_t offset, bool is_x) { + if (!is_x) { + if (val) + val += offset; + write_uint32_t_reverse_endianness((uint32_t)val); + } + else { + align_write(0x08); + write_int64_t_reverse_endianness(val); + } +} + +void stream::write_offset_f2(int64_t val, int64_t offset) { + if (val) + val += offset; + write_uint32_t_reverse_endianness((uint32_t)val); +} + +void stream::write_offset_x(int64_t val) { + align_write(0x08); + write_int64_t_reverse_endianness(val); +} diff --git a/src/KKdLib/io/stream.hpp b/src/KKdLib/io/stream.hpp new file mode 100644 index 0000000..55d202d --- /dev/null +++ b/src/KKdLib/io/stream.hpp @@ -0,0 +1,157 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include +#include +#include "../default.hpp" +#include "../half_t.hpp" + +class stream { +public: + uint8_t buf[0x100]; + int64_t length; + bool big_endian; + std::vector position_stack; + + stream(); + virtual ~stream(); + + virtual int flush() = 0; + virtual void close(); + virtual bool check_null() = 0; + virtual bool check_not_null() = 0; + + virtual void align_read(size_t align) = 0; + virtual void align_write(size_t align) = 0; + virtual size_t read(size_t count) = 0; + virtual size_t read(void* buf, size_t count) = 0; + virtual size_t read(void* buf, size_t size, size_t count) = 0; + virtual size_t write(size_t count) = 0; + virtual size_t write(const void* buf, size_t count) = 0; + virtual size_t write(const void* buf, size_t size, size_t count) = 0; + virtual int32_t read_char() = 0; + virtual int32_t write_char(char c) = 0; + virtual int64_t get_length() = 0; + virtual int64_t get_position() = 0; + virtual int32_t set_position(int64_t pos, int32_t seek) = 0; + + int32_t position_push(int64_t pos, int32_t seek); + void position_pop(); + + int8_t read_int8_t(); + uint8_t read_uint8_t(); + void write_int8_t(int8_t val); + void write_uint8_t(uint8_t val); + + int16_t read_int16_t(); + int16_t read_int16_t_reverse_endianness(); + int16_t read_int16_t_reverse_endianness(bool big_endian); + void write_int16_t(int16_t val); + void write_int16_t_reverse_endianness(int16_t val); + void write_int16_t_reverse_endianness(int16_t val, bool big_endian); + + uint16_t read_uint16_t(); + uint16_t read_uint16_t_reverse_endianness(); + uint16_t read_uint16_t_reverse_endianness(bool big_endian); + void write_uint16_t(uint16_t val); + void write_uint16_t_reverse_endianness(uint16_t val); + void write_uint16_t_reverse_endianness(uint16_t val, bool big_endian); + + int32_t read_int32_t(); + int32_t read_int32_t_reverse_endianness(); + int32_t read_int32_t_reverse_endianness(bool big_endian); + void write_int32_t(int32_t val); + void write_int32_t_reverse_endianness(int32_t val); + void write_int32_t_reverse_endianness(int32_t val, bool big_endian); + + uint32_t read_uint32_t(); + uint32_t read_uint32_t_reverse_endianness(); + uint32_t read_uint32_t_reverse_endianness(bool big_endian); + void write_uint32_t(uint32_t val); + void write_uint32_t_reverse_endianness(uint32_t val); + void write_uint32_t_reverse_endianness(uint32_t val, bool big_endian); + + int64_t read_int64_t(); + int64_t read_int64_t_reverse_endianness(); + int64_t read_int64_t_reverse_endianness(bool big_endian); + void write_int64_t(int64_t val); + void write_int64_t_reverse_endianness(int64_t val); + void write_int64_t_reverse_endianness(int64_t val, bool big_endian); + + uint64_t read_uint64_t(); + uint64_t read_uint64_t_reverse_endianness(); + uint64_t read_uint64_t_reverse_endianness(bool big_endian); + void write_uint64_t(uint64_t val); + void write_uint64_t_reverse_endianness(uint64_t val); + void write_uint64_t_reverse_endianness(uint64_t val, bool big_endian); + + half_t read_half_t(); + half_t read_half_t_reverse_endianness(); + half_t read_half_t_reverse_endianness(bool big_endian); + void write_half_t(half_t val); + void write_half_t_reverse_endianness(half_t val); + void write_half_t_reverse_endianness(half_t val, bool big_endian); + + float_t read_float_t(); + float_t read_float_t_reverse_endianness(); + float_t read_float_t_reverse_endianness(bool big_endian); + void write_float_t(float_t val); + void write_float_t_reverse_endianness(float_t val); + void write_float_t_reverse_endianness(float_t val, bool big_endian); + + double_t read_double_t(); + double_t read_double_t_reverse_endianness(); + double_t read_double_t_reverse_endianness(bool big_endian); + void write_double_t(double_t val); + void write_double_t_reverse_endianness(double_t val); + void write_double_t_reverse_endianness(double_t val, bool big_endian); + + std::string read_string(size_t length); + std::wstring read_wstring(size_t length); + std::string read_string_null_terminated(); + std::wstring read_wstring_null_terminated(); + std::string read_string_null_terminated_offset(int64_t offset); + std::wstring read_wstring_null_terminated_offset(int64_t offset); + char* read_utf8_string_null_terminated(); + wchar_t* read_utf16_string_null_terminated(); + char* read_utf8_string_null_terminated_offset(int64_t offset); + wchar_t* read_utf16_string_null_terminated_offset(int64_t offset); + size_t read_utf8_string_null_terminated_length(); + size_t read_utf16_string_null_terminated_length(); + size_t read_utf8_string_null_terminated_offset_length(int64_t offset); + size_t read_utf16_string_null_terminated_offset_length(int64_t offset); + + void write_string(std::string& str); + void write_string(std::string&& str); + void write_wstring(std::wstring& str); + void write_wstring(std::wstring&& str); + void write_string_null_terminated(std::string& str); + void write_string_null_terminated(std::string&& str); + void write_wstring_null_terminated(std::wstring& str); + void write_wstring_null_terminated(std::wstring&& str); + void write_utf8_string(const char* str); + void write_utf16_string(const wchar_t* str); + void write_utf8_string_null_terminated(const char* str); + void write_utf16_string_null_terminated(const wchar_t* str); + + int64_t read_offset(int64_t offset, bool is_x); + int64_t read_offset_f2(int64_t offset); + int64_t read_offset_x(); + void write_offset(int64_t val, int64_t offset, bool is_x); + void write_offset_f2(int64_t val, int64_t offset); + void write_offset_x(int64_t val); + + template + size_t read_data(T& data) { + return read(&data, sizeof(T)); + } + + template + size_t write_data(const T& data) { + return write(&data, sizeof(T)); + } +}; diff --git a/src/KKdLib/kf.cpp b/src/KKdLib/kf.cpp new file mode 100644 index 0000000..0f58210 --- /dev/null +++ b/src/KKdLib/kf.cpp @@ -0,0 +1,81 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "kf.hpp" + +void kft_check(void* src_key, kf_type src_type, void* dst_key, kf_type* dst_type) { + switch (src_type) { + case KEY_FRAME_TYPE_0: { + kft0* sk = (kft0*)src_key; + kft0* dk = (kft0*)dst_key; + dk->frame = sk->frame; + *dst_type = KEY_FRAME_TYPE_1; + } break; + case KEY_FRAME_TYPE_1: { + kft1* sk = (kft1*)src_key; + if (*(uint32_t*)&sk->value != 0) { + kft1* dk = (kft1*)dst_key; + dk->frame = sk->frame; + dk->value = sk->value; + *dst_type = KEY_FRAME_TYPE_1; + } + else { + kft0* dk = (kft0*)dst_key; + dk->frame = sk->frame; + *dst_type = KEY_FRAME_TYPE_0; + } + } break; + case KEY_FRAME_TYPE_2: { + kft2* sk = (kft2*)src_key; + if (*(uint32_t*)&sk->tangent != 0) { + kft2* dk = (kft2*)dst_key; + dk->frame = sk->frame; + dk->value = sk->value; + dk->tangent = sk->tangent; + *dst_type = KEY_FRAME_TYPE_2; + } + else if (*(uint32_t*)&sk->value != 0) { + kft1* dk = (kft1*)dst_key; + dk->frame = sk->frame; + dk->value = sk->value; + *dst_type = KEY_FRAME_TYPE_1; + } + else { + kft0* dk = (kft0*)dst_key; + dk->frame = sk->frame; + *dst_type = KEY_FRAME_TYPE_0; + } + } break; + case KEY_FRAME_TYPE_3: { + kft3* sk = (kft3*)src_key; + if (*(uint32_t*)&sk->tangent1 != *(uint32_t*)&sk->tangent2) { + kft3* dk = (kft3*)dst_key; + dk->frame = sk->frame; + dk->value = sk->value; + dk->tangent1 = sk->tangent1; + dk->tangent2 = sk->tangent2; + *dst_type = KEY_FRAME_TYPE_3; + } + else if (*(uint32_t*)&sk->tangent1 != 0) { + kft2* dk = (kft2*)dst_key; + dk->frame = sk->frame; + dk->value = sk->value; + dk->tangent = sk->tangent1; + *dst_type = KEY_FRAME_TYPE_2; + } + else if (*(uint32_t*)&sk->value != 0) { + kft1* dk = (kft1*)dst_key; + dk->frame = sk->frame; + dk->value = sk->value; + *dst_type = KEY_FRAME_TYPE_1; + } + else { + kft0* dk = (kft0*)dst_key; + dk->frame = sk->frame; + *dst_type = KEY_FRAME_TYPE_0; + } + } break; + } +} diff --git a/src/KKdLib/kf.hpp b/src/KKdLib/kf.hpp new file mode 100644 index 0000000..e66658d --- /dev/null +++ b/src/KKdLib/kf.hpp @@ -0,0 +1,39 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "default.hpp" + +enum kf_type { + KEY_FRAME_TYPE_0 = 0, + KEY_FRAME_TYPE_1 = 1, + KEY_FRAME_TYPE_2 = 2, + KEY_FRAME_TYPE_3 = 3, +}; + +struct kft0 { + float_t frame; +}; + +struct kft1 { + float_t frame; + float_t value; +}; + +struct kft2 { + float_t frame; + float_t value; + float_t tangent; +}; + +struct kft3 { + float_t frame; + float_t value; + float_t tangent1; + float_t tangent2; +}; + +extern void kft_check(void* src_key, kf_type src_type, void* dst_key, kf_type* dst_type); diff --git a/src/KKdLib/mat.cpp b/src/KKdLib/mat.cpp new file mode 100644 index 0000000..969b331 --- /dev/null +++ b/src/KKdLib/mat.cpp @@ -0,0 +1,2008 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder + Matrix Inverse algo: https://github.com/niswegmann/small-matrix-inverse +*/ + +#include "mat.hpp" + +const mat3 mat3_identity = { + { 1.0f, 0.0f, 0.0f }, + { 0.0f, 1.0f, 0.0f }, + { 0.0f, 0.0f, 1.0f }, +}; + +const mat3 mat3_null = { + { 0.0f, 0.0f, 0.0f }, + { 0.0f, 0.0f, 0.0f }, + { 0.0f, 0.0f, 0.0f }, +}; + +const mat4 mat4_identity = { + { 1.0f, 0.0f, 0.0f, 0.0f }, + { 0.0f, 1.0f, 0.0f, 0.0f }, + { 0.0f, 0.0f, 1.0f, 0.0f }, + { 0.0f, 0.0f, 0.0f, 1.0f }, +}; + +const mat4 mat4_null = { + { 0.0f, 0.0f, 0.0f, 0.0f }, + { 0.0f, 0.0f, 0.0f, 0.0f }, + { 0.0f, 0.0f, 0.0f, 0.0f }, + { 0.0f, 0.0f, 0.0f, 0.0f }, +}; + +inline void mat3_add(const mat3* in_m1, const float_t value, mat3* out_m) { + out_m->row0 = in_m1->row0 + value; + out_m->row1 = in_m1->row1 + value; + out_m->row2 = in_m1->row2 + value; +} + +inline void mat3_add(const mat3* in_m1, const mat3* in_m2, mat3* out_m) { + out_m->row0 = in_m1->row0 + in_m2->row0; + out_m->row1 = in_m1->row1 + in_m2->row1; + out_m->row2 = in_m1->row2 + in_m2->row2; +} + +inline void mat3_sub(const mat3* in_m1, const float_t value, mat3* out_m) { + out_m->row0 = in_m1->row0 - value; + out_m->row1 = in_m1->row1 - value; + out_m->row2 = in_m1->row2 - value; +} + +inline void mat3_sub(const mat3* in_m1, const mat3* in_m2, mat3* out_m) { + out_m->row0 = in_m1->row0 - in_m2->row0; + out_m->row1 = in_m1->row1 - in_m2->row1; + out_m->row2 = in_m1->row2 - in_m2->row2; +} + +inline void mat3_mul(const mat3* in_m1, const float_t value, mat3* out_m) { + out_m->row0 = in_m1->row0 * value; + out_m->row1 = in_m1->row1 * value; + out_m->row2 = in_m1->row2 * value; +} + +inline void mat3_mul(const mat3* in_m1, const mat3* in_m2, mat3* out_m) { + __m128 t0; + __m128 t1; + __m128 t2; + __m128 yt; + __m128 xt0; + __m128 xt1; + __m128 xt2; + xt0 = vec3::load_xmm(in_m1->row0); + xt1 = vec3::load_xmm(in_m1->row1); + xt2 = vec3::load_xmm(in_m1->row2); + yt = vec3::load_xmm(in_m2->row0); + t0 = _mm_mul_ps(xt0, _mm_shuffle_ps(yt, yt, 0x00)); + t1 = _mm_mul_ps(xt1, _mm_shuffle_ps(yt, yt, 0x55)); + t2 = _mm_mul_ps(xt2, _mm_shuffle_ps(yt, yt, 0xAA)); + out_m->row0 = vec3::store_xmm(_mm_add_ps(_mm_add_ps(t0, t1), t2)); + yt = vec3::load_xmm(in_m2->row1); + t0 = _mm_mul_ps(xt0, _mm_shuffle_ps(yt, yt, 0x00)); + t1 = _mm_mul_ps(xt1, _mm_shuffle_ps(yt, yt, 0x55)); + t2 = _mm_mul_ps(xt2, _mm_shuffle_ps(yt, yt, 0xAA)); + out_m->row1 = vec3::store_xmm(_mm_add_ps(_mm_add_ps(t0, t1), t2)); + yt = vec3::load_xmm(in_m2->row2); + t0 = _mm_mul_ps(xt0, _mm_shuffle_ps(yt, yt, 0x00)); + t1 = _mm_mul_ps(xt1, _mm_shuffle_ps(yt, yt, 0x55)); + t2 = _mm_mul_ps(xt2, _mm_shuffle_ps(yt, yt, 0xAA)); + out_m->row2 = vec3::store_xmm(_mm_add_ps(_mm_add_ps(t0, t1), t2)); +} + +inline void mat3_mul(const mat3* in_m1, const vec3* in_axis, const float_t in_angle, mat3* out_m) { + quat q1 = quat(in_m1->row0.x, in_m1->row1.x, in_m1->row2.x, in_m1->row0.y, + in_m1->row1.y, in_m1->row2.y, in_m1->row0.z, in_m1->row1.z, in_m1->row2.z); + quat q2 = quat(*in_axis, in_angle); + quat q3 = quat::mul(q2, q1); + mat3_set(&q3, out_m); +} + +inline void mat3_transform_vector(const mat3* in_m1, const vec2* normal, vec2* normalOut) { + __m128 yt; + __m128 zt0; + __m128 zt1; + yt = vec2::load_xmm(*normal); + zt0 = _mm_mul_ps(vec3::load_xmm(in_m1->row0), _mm_shuffle_ps(yt, yt, 0x00)); + zt1 = _mm_mul_ps(vec3::load_xmm(in_m1->row1), _mm_shuffle_ps(yt, yt, 0x55)); + *normalOut = vec2::store_xmm(_mm_add_ps(zt0, zt1)); +} + +inline void mat3_transform_vector(const mat3* in_m1, const vec3* normal, vec3* normalOut) { + __m128 yt; + __m128 zt0; + __m128 zt1; + __m128 zt2; + yt = vec3::load_xmm(*normal); + zt0 = _mm_mul_ps(vec3::load_xmm(in_m1->row0), _mm_shuffle_ps(yt, yt, 0x00)); + zt1 = _mm_mul_ps(vec3::load_xmm(in_m1->row1), _mm_shuffle_ps(yt, yt, 0x55)); + zt2 = _mm_mul_ps(vec3::load_xmm(in_m1->row2), _mm_shuffle_ps(yt, yt, 0xAA)); + *normalOut = vec3::store_xmm(_mm_add_ps(_mm_add_ps(zt0, zt1), zt2)); +} + +inline void mat3_inverse_transform_vector(const mat3* in_m1, const vec2* normal, vec2* normalOut) { + __m128 yt; + __m128 zt; + yt = vec2::load_xmm(*normal); + zt = _mm_mul_ps(yt, vec3::load_xmm(in_m1->row0)); + normalOut->x = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); + zt = _mm_mul_ps(yt, vec3::load_xmm(in_m1->row1)); + normalOut->y = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline void mat3_inverse_transform_vector(const mat3* in_m1, const vec3* normal, vec3* normalOut) { + __m128 yt; + __m128 zt; + yt = vec3::load_xmm(*normal); + zt = _mm_mul_ps(yt, vec3::load_xmm(in_m1->row0)); + zt = _mm_hadd_ps(zt, zt); + normalOut->x = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); + zt = _mm_mul_ps(yt, vec3::load_xmm(in_m1->row1)); + zt = _mm_hadd_ps(zt, zt); + normalOut->y = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); + zt = _mm_mul_ps(yt, vec3::load_xmm(in_m1->row2)); + zt = _mm_hadd_ps(zt, zt); + normalOut->z = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline void mat3_transpose(const mat3* in_m1, mat3* out_m) { + __m128 xt0; + __m128 xt1; + __m128 xt2; + __m128 xt3; + __m128 yt0; + __m128 yt1; + __m128 yt2; + __m128 yt3; + xt0 = vec3::load_xmm(in_m1->row0); + xt1 = vec3::load_xmm(in_m1->row1); + xt2 = vec3::load_xmm(in_m1->row2); + xt3 = vec4::load_xmm(0.0f); + yt0 = _mm_unpacklo_ps(xt0, xt1); + yt1 = _mm_unpackhi_ps(xt0, xt1); + yt2 = _mm_unpacklo_ps(xt2, xt3); + yt3 = _mm_unpackhi_ps(xt2, xt3); + out_m->row0 = vec3::store_xmm(_mm_movelh_ps(yt0, yt2)); + out_m->row1 = vec3::store_xmm(_mm_movehl_ps(yt2, yt0)); + out_m->row2 = vec3::store_xmm(_mm_movelh_ps(yt1, yt3)); +} + +void mat3_invert(const mat3* in_m1, mat3* out_m) { + vec3 xt0; + vec3 xt1; + vec3 xt2; + __m128 zt0; + __m128 zt1; + __m128 zt2; + __m128 wt; + xt0 = in_m1->row0; + xt1 = in_m1->row1; + xt2 = in_m1->row2; + zt0 = _mm_sub_ps( + _mm_mul_ps( + _mm_set_ps(0.0f, xt0.y, xt0.z, xt1.y), + _mm_set_ps(0.0f, xt1.z, xt2.y, xt2.z) + ), + _mm_mul_ps( + _mm_set_ps(0.0f, xt0.z, xt0.y, xt1.z), + _mm_set_ps(0.0f, xt1.y, xt2.z, xt2.y) + ) + ); + zt1 = _mm_sub_ps( + _mm_mul_ps( + _mm_set_ps(0.0f, xt0.z, xt0.x, xt1.z), + _mm_set_ps(0.0f, xt1.x, xt2.z, xt2.x) + ), + _mm_mul_ps( + _mm_set_ps(0.0f, xt0.x, xt0.z, xt1.x), + _mm_set_ps(0.0f, xt1.z, xt2.x, xt2.z) + ) + ); + zt2 = _mm_sub_ps( + _mm_mul_ps( + _mm_set_ps(0.0f, xt0.x, xt0.y, xt1.x), + _mm_set_ps(0.0f, xt1.y, xt2.x, xt2.y) + ), + _mm_mul_ps( + _mm_set_ps(0.0f, xt0.y, xt0.x, xt1.y), + _mm_set_ps(0.0f, xt1.x, xt2.y, xt2.x) + ) + ); + + wt = _mm_movelh_ps(_mm_unpacklo_ps(zt0, zt1), zt2); + wt = _mm_mul_ps(vec3::load_xmm(xt0), wt); + wt = _mm_hadd_ps(wt, wt); + wt = _mm_hadd_ps(wt, wt); + if (_mm_cvtss_f32(wt) != 0.0f) + wt = _mm_div_ss(_mm_set_ss(1.0f), wt); + wt = _mm_shuffle_ps(wt, wt, 0); + out_m->row0 = vec3::store_xmm(_mm_mul_ps(zt0, wt)); + out_m->row1 = vec3::store_xmm(_mm_mul_ps(zt1, wt)); + out_m->row2 = vec3::store_xmm(_mm_mul_ps(zt2, wt)); +} + +inline void mat3_invert_fast(const mat3* in_m1, mat3* out_m) { + mat3_transpose(in_m1, out_m); +} + +inline void mat3_normalize(const mat3* in_m1, mat3* out_m) { + float_t det = mat3_determinant(in_m1); + if (det != 0.0f) + det = 1.0f / det; + out_m->row0 = in_m1->row0 * det; + out_m->row1 = in_m1->row1 * det; + out_m->row2 = in_m1->row2 * det; +} + +inline void mat3_normalize_rotation(const mat3* in_m1, mat3* out_m) { + out_m->row0 = vec3::normalize(in_m1->row0); + out_m->row1 = vec3::normalize(in_m1->row1); + out_m->row2 = vec3::normalize(in_m1->row2); +} + +inline float_t mat3_determinant(const mat3* in_m1) { + vec3 xt0; + vec3 xt1; + vec3 xt2; + xt0 = in_m1->row0; + xt1 = in_m1->row1; + xt2 = in_m1->row2; + float_t b00 = xt0.x * xt1.y * xt2.z; + float_t b01 = xt0.y * xt1.z * xt2.x; + float_t b02 = xt0.z * xt1.x * xt2.y; + float_t b03 = xt0.z * xt1.y * xt2.x; + float_t b04 = xt0.x * xt1.z * xt2.y; + float_t b05 = xt0.y * xt1.x * xt2.z; + return b00 + b01 + b02 - b03 - b04 - b05; +} + +inline void mat3_rotate_x(float_t rad, mat3* out_m) { + float_t s = sinf(rad); + float_t c = cosf(rad); + *out_m = mat3_identity; + out_m->row1.y = c; + out_m->row1.z = s; + out_m->row2.y = -s; + out_m->row2.z = c; +} + +inline void mat3_rotate_y(float_t rad, mat3* out_m) { + float_t s = sinf(rad); + float_t c = cosf(rad); + *out_m = mat3_identity; + out_m->row0.x = c; + out_m->row0.z = -s; + out_m->row2.x = s; + out_m->row2.z = c; +} + +inline void mat3_rotate_z(float_t rad, mat3* out_m) { + float_t s = sinf(rad); + float_t c = cosf(rad); + *out_m = mat3_identity; + out_m->row0.x = c; + out_m->row0.y = s; + out_m->row1.x = -s; + out_m->row1.y = c; +} + +inline void mat3_rotate_x(float_t s, float_t c, mat3* out_m) { + *out_m = mat3_identity; + out_m->row1.y = c; + out_m->row1.z = s; + out_m->row2.y = -s; + out_m->row2.z = c; +} + +inline void mat3_rotate_y(float_t s, float_t c, mat3* out_m) { + *out_m = mat3_identity; + out_m->row0.x = c; + out_m->row0.z = -s; + out_m->row2.x = s; + out_m->row2.z = c; +} + +inline void mat3_rotate_z(float_t s, float_t c, mat3* out_m) { + *out_m = mat3_identity; + out_m->row0.x = c; + out_m->row0.y = s; + out_m->row1.x = -s; + out_m->row1.y = c; +} + +inline void mat3_rotate_xyz(float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m) { + mat3 dt; + dt = mat3_identity; + if (rad_x != 0.0f) + mat3_mul_rotate_x(&dt, rad_x, &dt); + if (rad_y != 0.0f) + mat3_mul_rotate_y(&dt, rad_y, &dt); + if (rad_z != 0.0f) + mat3_mul_rotate_z(&dt, rad_z, &dt); + *out_m = dt; +} + +inline void mat3_rotate_xzy(float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m) { + mat3 dt; + dt = mat3_identity; + if (rad_x != 0.0f) + mat3_mul_rotate_x(&dt, rad_x, &dt); + if (rad_z != 0.0f) + mat3_mul_rotate_z(&dt, rad_z, &dt); + if (rad_y != 0.0f) + mat3_mul_rotate_y(&dt, rad_y, &dt); + *out_m = dt; +} + +inline void mat3_rotate_yxz(float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m) { + mat3 dt; + dt = mat3_identity; + if (rad_y != 0.0f) + mat3_mul_rotate_y(&dt, rad_y, &dt); + if (rad_x != 0.0f) + mat3_mul_rotate_x(&dt, rad_x, &dt); + if (rad_z != 0.0f) + mat3_mul_rotate_z(&dt, rad_z, &dt); + *out_m = dt; +} + +inline void mat3_rotate_yzx(float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m) { + mat3 dt; + dt = mat3_identity; + if (rad_y != 0.0f) + mat3_mul_rotate_y(&dt, rad_y, &dt); + if (rad_z != 0.0f) + mat3_mul_rotate_z(&dt, rad_z, &dt); + if (rad_x != 0.0f) + mat3_mul_rotate_x(&dt, rad_x, &dt); + *out_m = dt; +} + +inline void mat3_rotate_zxy(float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m) { + mat3 dt; + dt = mat3_identity; + if (rad_z != 0.0f) + mat3_mul_rotate_z(&dt, rad_z, &dt); + if (rad_x != 0.0f) + mat3_mul_rotate_x(&dt, rad_x, &dt); + if (rad_y != 0.0f) + mat3_mul_rotate_y(&dt, rad_y, &dt); + *out_m = dt; +} + +inline void mat3_rotate_zyx(float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m) { + mat3 dt; + dt = mat3_identity; + if (rad_z != 0.0f) + mat3_mul_rotate_z(&dt, rad_z, &dt); + if (rad_y != 0.0f) + mat3_mul_rotate_y(&dt, rad_y, &dt); + if (rad_x != 0.0f) + mat3_mul_rotate_x(&dt, rad_x, &dt); + *out_m = dt; +} + +inline void mat3_mul_rotate_x(const mat3* in_m1, float_t rad, mat3* out_m) { + __m128 t1; + __m128 t2; + __m128 y0; + __m128 y1; + __m128 y2; + float_t s = sinf(rad); + float_t c = cosf(rad); + y0 = vec3::load_xmm(in_m1->row0); + y1 = vec3::load_xmm(in_m1->row1); + y2 = vec3::load_xmm(in_m1->row2); + out_m->row0 = vec3::store_xmm(y0); + t1 = _mm_mul_ps(y1, vec4::load_xmm(c)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(s)); + out_m->row1 = vec3::store_xmm(_mm_add_ps(t1, t2)); + t1 = _mm_mul_ps(y1, vec4::load_xmm(-s)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(c)); + out_m->row2 = vec3::store_xmm(_mm_add_ps(t1, t2)); +} + +inline void mat3_mul_rotate_y(const mat3* in_m1, float_t rad, mat3* out_m) { + __m128 t0; + __m128 t2; + __m128 y0; + __m128 y1; + __m128 y2; + float_t s = sinf(rad); + float_t c = cosf(rad); + y0 = vec3::load_xmm(in_m1->row0); + y1 = vec3::load_xmm(in_m1->row1); + y2 = vec3::load_xmm(in_m1->row2); + t0 = _mm_mul_ps(y0, vec4::load_xmm(c)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(-s)); + out_m->row0 = vec3::store_xmm(_mm_add_ps(t0, t2)); + out_m->row1 = vec3::store_xmm(y1); + t0 = _mm_mul_ps(y0, vec4::load_xmm(s)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(c)); + out_m->row2 = vec3::store_xmm(_mm_add_ps(t0, t2)); +} + +inline void mat3_mul_rotate_z(const mat3* in_m1, float_t rad, mat3* out_m) { + __m128 t0; + __m128 t1; + __m128 y0; + __m128 y1; + __m128 y2; + float_t s = sinf(rad); + float_t c = cosf(rad); + y0 = vec3::load_xmm(in_m1->row0); + y1 = vec3::load_xmm(in_m1->row1); + y2 = vec3::load_xmm(in_m1->row2); + t0 = _mm_mul_ps(y0, vec4::load_xmm(c)); + t1 = _mm_mul_ps(y1, vec4::load_xmm(s)); + out_m->row0 = vec3::store_xmm(_mm_add_ps(t0, t1)); + t0 = _mm_mul_ps(y0, vec4::load_xmm(-s)); + t1 = _mm_mul_ps(y1, vec4::load_xmm(c)); + out_m->row1 = vec3::store_xmm(_mm_add_ps(t0, t1)); + out_m->row2 = vec3::store_xmm(y2); +} + +inline void mat3_mul_rotate_x(const mat3* in_m1, float_t s, float_t c, mat3* out_m) { + __m128 t1; + __m128 t2; + __m128 y0; + __m128 y1; + __m128 y2; + y0 = vec3::load_xmm(in_m1->row0); + y1 = vec3::load_xmm(in_m1->row1); + y2 = vec3::load_xmm(in_m1->row2); + out_m->row0 = vec3::store_xmm(y0); + t1 = _mm_mul_ps(y1, vec4::load_xmm(c)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(s)); + out_m->row1 = vec3::store_xmm(_mm_add_ps(t1, t2)); + t1 = _mm_mul_ps(y1, vec4::load_xmm(-s)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(c)); + out_m->row2 = vec3::store_xmm(_mm_add_ps(t1, t2)); +} + +inline void mat3_mul_rotate_y(const mat3* in_m1, float_t s, float_t c, mat3* out_m) { + __m128 t0; + __m128 t2; + __m128 y0; + __m128 y1; + __m128 y2; + y0 = vec3::load_xmm(in_m1->row0); + y1 = vec3::load_xmm(in_m1->row1); + y2 = vec3::load_xmm(in_m1->row2); + t0 = _mm_mul_ps(y0, vec4::load_xmm(c)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(-s)); + out_m->row0 = vec3::store_xmm(_mm_add_ps(t0, t2)); + out_m->row1 = vec3::store_xmm(y1); + t0 = _mm_mul_ps(y0, vec4::load_xmm(s)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(c)); + out_m->row2 = vec3::store_xmm(_mm_add_ps(t0, t2)); +} + +inline void mat3_mul_rotate_z(const mat3* in_m1, float_t s, float_t c, mat3* out_m) { + __m128 t0; + __m128 t1; + __m128 y0; + __m128 y1; + __m128 y2; + y0 = vec3::load_xmm(in_m1->row0); + y1 = vec3::load_xmm(in_m1->row1); + y2 = vec3::load_xmm(in_m1->row2); + t0 = _mm_mul_ps(y0, vec4::load_xmm(c)); + t1 = _mm_mul_ps(y1, vec4::load_xmm(s)); + out_m->row0 = vec3::store_xmm(_mm_add_ps(t0, t1)); + t0 = _mm_mul_ps(y0, vec4::load_xmm(-s)); + t1 = _mm_mul_ps(y1, vec4::load_xmm(c)); + out_m->row1 = vec3::store_xmm(_mm_add_ps(t0, t1)); + out_m->row2 = vec3::store_xmm(y2); +} + +inline void mat3_mul_rotate_xyz(const mat3* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m) { + mat3 dt; + dt = *in_m1; + if (rad_x != 0.0f) + mat3_mul_rotate_x(&dt, rad_x, &dt); + if (rad_y != 0.0f) + mat3_mul_rotate_y(&dt, rad_y, &dt); + if (rad_z != 0.0f) + mat3_mul_rotate_z(&dt, rad_z, &dt); + *out_m = dt; +} + +inline void mat3_mul_rotate_xzy(const mat3* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m) { + mat3 dt; + dt = *in_m1; + if (rad_x != 0.0f) + mat3_mul_rotate_x(&dt, rad_x, &dt); + if (rad_z != 0.0f) + mat3_mul_rotate_z(&dt, rad_z, &dt); + if (rad_y != 0.0f) + mat3_mul_rotate_y(&dt, rad_y, &dt); + *out_m = dt; +} + +inline void mat3_mul_rotate_yxz(const mat3* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m) { + mat3 dt; + dt = *in_m1; + if (rad_y != 0.0f) + mat3_mul_rotate_y(&dt, rad_y, &dt); + if (rad_x != 0.0f) + mat3_mul_rotate_x(&dt, rad_x, &dt); + if (rad_z != 0.0f) + mat3_mul_rotate_z(&dt, rad_z, &dt); + *out_m = dt; +} + +inline void mat3_mul_rotate_yzx(const mat3* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m) { + mat3 dt; + dt = *in_m1; + if (rad_y != 0.0f) + mat3_mul_rotate_y(&dt, rad_y, &dt); + if (rad_z != 0.0f) + mat3_mul_rotate_z(&dt, rad_z, &dt); + if (rad_x != 0.0f) + mat3_mul_rotate_x(&dt, rad_x, &dt); + *out_m = dt; +} + +inline void mat3_mul_rotate_zxy(const mat3* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m) { + mat3 dt; + dt = *in_m1; + if (rad_z != 0.0f) + mat3_mul_rotate_z(&dt, rad_z, &dt); + if (rad_x != 0.0f) + mat3_mul_rotate_x(&dt, rad_x, &dt); + if (rad_y != 0.0f) + mat3_mul_rotate_y(&dt, rad_y, &dt); + *out_m = dt; +} + +inline void mat3_mul_rotate_zyx(const mat3* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m) { + mat3 dt; + dt = *in_m1; + if (rad_z != 0.0f) + mat3_mul_rotate_z(&dt, rad_z, &dt); + if (rad_y != 0.0f) + mat3_mul_rotate_y(&dt, rad_y, &dt); + if (rad_x != 0.0f) + mat3_mul_rotate_x(&dt, rad_x, &dt); + *out_m = dt; +} + +inline void mat3_scale(float_t sx, float_t sy, float_t sz, mat3* out_m) { + *out_m = mat3_identity; + out_m->row0.x = sx; + out_m->row1.y = sy; + out_m->row2.z = sz; +} + +inline void mat3_scale_x(float_t s, mat3* out_m) { + *out_m = mat3_identity; + out_m->row0.x = s; +} + +inline void mat3_scale_y(float_t s, mat3* out_m) { + *out_m = mat3_identity; + out_m->row1.y = s; +} + +inline void mat3_scale_z(float_t s, mat3* out_m) { + *out_m = mat3_identity; + out_m->row2.z = s; +} + +inline void mat3_mul_scale(const mat3* in_m1, float_t sx, float_t sy, float_t sz, mat3* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + if (sx != 1.0f || sy != 1.0f || sz != 1.0f) { + out_m->row0 *= sx; + out_m->row1 *= sy; + out_m->row2 *= sz; + } +} + +inline void mat3_mul_scale_x(const mat3* in_m1, float_t s, mat3* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + out_m->row0 *= s; +} + +inline void mat3_mul_scale_y(const mat3* in_m1, float_t s, mat3* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + out_m->row1 *= s; +} + +inline void mat3_mul_scale_z(const mat3* in_m1, float_t s, mat3* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + out_m->row2 *= s; +} + +inline void mat3_set(const quat* in_q1, mat3* out_m) { + float_t x = in_q1->x; + float_t y = in_q1->y; + float_t z = in_q1->z; + float_t w = in_q1->w; + float_t len = quat::length_squared(*in_q1); + len = len > 0.0f ? 2.0f / len : 0.0f; + float_t xx = x * x * len; + float_t xy = x * y * len; + float_t xz = x * z * len; + float_t yy = y * y * len; + float_t zz = z * z * len; + float_t yz = y * z * len; + float_t wx = w * x * len; + float_t wy = w * y * len; + float_t wz = w * z * len; + out_m->row0.x = 1.0f - zz - yy; + out_m->row0.y = xy + wz; + out_m->row0.z = xz - wy; + out_m->row1.x = xy - wz; + out_m->row1.y = 1.0f - zz - xx; + out_m->row1.z = yz + wx; + out_m->row2.x = xz + wy; + out_m->row2.y = yz - wx; + out_m->row2.z = 1.0f - yy - xx; +} + +inline void mat3_set(const vec3* in_axis, float_t in_angle, mat3* out_m) { + mat3_set(in_axis, sinf(in_angle), cosf(in_angle), out_m); +} + +inline void mat3_set(const vec3* in_axis, float_t s, float_t c, mat3* out_m) { + float_t c_1 = 1.0f - c; + + vec3 axis = vec3::normalize(*in_axis); + vec3 axis_s = axis * s; + + vec3 temp; + temp = axis * axis.x * c_1; + out_m->row0.x = temp.x + c; + out_m->row1.x = temp.y - axis_s.z; + out_m->row2.x = temp.z + axis_s.y; + temp = axis * axis.y * c_1; + out_m->row0.y = temp.x + axis_s.z; + out_m->row1.y = temp.y + c; + out_m->row2.y = temp.z - axis_s.x; + temp = axis * axis.z * c_1; + out_m->row0.z = temp.x - axis_s.y; + out_m->row1.z = temp.y + axis_s.x; + out_m->row2.z = temp.z + c; +} + +inline void mat4_to_mat3(const mat4* in_m1, mat3* out_m) { + out_m->row0 = *(vec3*)&in_m1->row0; + out_m->row1 = *(vec3*)&in_m1->row1; + out_m->row2 = *(vec3*)&in_m1->row2; +} + +inline void mat4_to_mat3_inverse(const mat4* in_m1, mat3* out_m) { + mat4 yt; + + mat4_invert(in_m1, &yt); + out_m->row0 = *(vec3*)&yt.row0; + out_m->row1 = *(vec3*)&yt.row1; + out_m->row2 = *(vec3*)&yt.row2; +} + +inline void mat3_get_rotation(const mat3* in_m1, vec3* out_rad) { + if (-in_m1->row0.z >= 1.0f) + out_rad->y = (float_t)M_PI_2; + else if (-in_m1->row0.z <= -1.0f) + out_rad->y = (float_t)-M_PI_2; + else + out_rad->y = asinf(-in_m1->row0.z); + + if (fabsf(in_m1->row0.z) < 0.99999899f) { + out_rad->x = atan2f(in_m1->row1.z, in_m1->row2.z); + out_rad->z = atan2f(in_m1->row0.y, in_m1->row0.x); + } + else { + out_rad->x = 0.0f; + out_rad->z = atan2f(in_m1->row2.y, in_m1->row1.y); + if (in_m1->row0.z > 0.0f) + out_rad->z = -out_rad->z; + } +} + +inline void mat3_get_scale(const mat3* in_m1, vec3* out_s) { + out_s->x = vec3::length(in_m1->row0); + out_s->y = vec3::length(in_m1->row1); + out_s->z = vec3::length(in_m1->row2); +} + +inline float_t mat3_get_max_scale(const mat3* in_m1) { + mat3 mat; + mat3_transpose(in_m1, &mat); + + float_t length; + float_t max = 0.0f; + length = vec3::length(mat.row0); + if (max < length) + max = length; + length = vec3::length(mat.row1); + if (max < length) + max = length; + length = vec3::length(mat.row2); + if (max < length) + max = length; + return max; +} + +inline void mat4_set(const quat* in_q1, mat4* out_m) { + mat4_set_rotation(out_m, in_q1); + out_m->row0.w = 0.0f; + out_m->row1.w = 0.0f; + out_m->row2.w = 0.0f; + out_m->row3 = { 0.0f, 0.0f, 0.0f, 1.0f }; +} + +static float_t vec3_angle_between_two_vectors(const vec3& in_v1, const vec3& in_v2) { + vec3 z_t = vec3::cross(in_v1, in_v2); + float_t v2 = vec3::length(z_t); + float_t v3 = vec3::dot(in_v1, in_v2); + return fabsf(atan2f(v2, v3)); +} + +void mat4_set(const vec3* in_v1, const vec3* in_v2, mat4* out_m) { + *out_m = mat4_identity; + if (*in_v1 == *in_v2) + return; + + if (fabsf(1.0f - vec3::dot(*in_v1, *in_v2)) <= 0.000001f) + return; + + vec3 axis = vec3::cross(*in_v1, *in_v2); + float_t axis_length = vec3::length(axis); + if (axis_length > 0.000001f) { + float_t angle = vec3_angle_between_two_vectors(*in_v1, *in_v2); + if (axis_length != 0.0) + axis *= 1.0f / axis_length; + mat4_set(&axis, angle, out_m); + } +} + +inline void mat4_set(const vec3* in_axis, float_t in_angle, mat4* out_m) { + mat4_set_rotation(out_m, in_axis, sinf(in_angle), cosf(in_angle)); + out_m->row0.w = 0.0f; + out_m->row1.w = 0.0f; + out_m->row2.w = 0.0f; + out_m->row3 = { 0.0f, 0.0f, 0.0f, 1.0f }; +} + +inline void mat4_set(const vec3* in_axis, float_t s, float_t c, mat4* out_m) { + mat4_set_rotation(out_m, in_axis, s, c); + out_m->row0.w = 0.0f; + out_m->row1.w = 0.0f; + out_m->row2.w = 0.0f; + out_m->row3 = { 0.0f, 0.0f, 0.0f, 1.0f }; +} + +inline void mat4_set_rotation(mat4* in_m1, const quat* in_q1) { + float_t x = in_q1->x; + float_t y = in_q1->y; + float_t z = in_q1->z; + float_t w = in_q1->w; + float_t len = quat::length_squared(*in_q1); + len = len > 0.0f ? 2.0f / len : 0.0f; + float_t xx = x * x * len; + float_t xy = x * y * len; + float_t xz = x * z * len; + float_t yy = y * y * len; + float_t zz = z * z * len; + float_t yz = y * z * len; + float_t wx = w * x * len; + float_t wy = w * y * len; + float_t wz = w * z * len; + in_m1->row0.x = 1.0f - zz - yy; + in_m1->row0.y = xy + wz; + in_m1->row0.z = xz - wy; + in_m1->row1.x = xy - wz; + in_m1->row1.y = 1.0f - zz - xx; + in_m1->row1.z = yz + wx; + in_m1->row2.x = xz + wy; + in_m1->row2.y = yz - wx; + in_m1->row2.z = 1.0f - yy - xx; +} + +inline void mat4_set_rotation(mat4* in_m1, const vec3* in_axis, const float_t in_angle) { + mat4_set_rotation(in_m1, in_axis, sinf(in_angle), cosf(in_angle)); +} + +inline void mat4_set_rotation(mat4* in_m1, const vec3* in_axis, const float_t s, const float_t c) { + float_t c_1 = 1.0f - c; + + vec3 axis = vec3::normalize(*in_axis); + vec3 axis_s = axis * s; + + vec3 temp; + temp = axis * axis.x * c_1; + in_m1->row0.x = temp.x + c; + in_m1->row1.x = temp.y - axis_s.z; + in_m1->row2.x = temp.z + axis_s.y; + temp = axis * axis.y * c_1; + in_m1->row0.y = temp.x + axis_s.z; + in_m1->row1.y = temp.y + c; + in_m1->row2.y = temp.z - axis_s.x; + temp = axis * axis.z * c_1; + in_m1->row0.z = temp.x - axis_s.y; + in_m1->row1.z = temp.y + axis_s.x; + in_m1->row2.z = temp.z + c; +} + +inline void mat4_add(const mat4* in_m1, const float_t value, mat4* out_m) { + out_m->row0 = in_m1->row0 + value; + out_m->row1 = in_m1->row1 + value; + out_m->row2 = in_m1->row2 + value; + out_m->row3 = in_m1->row3 + value; +} + +inline void mat4_add(const mat4* in_m1, const mat4* in_m2, mat4* out_m) { + out_m->row0 = in_m1->row0 + in_m2->row0; + out_m->row1 = in_m1->row1 + in_m2->row1; + out_m->row2 = in_m1->row2 + in_m2->row2; + out_m->row3 = in_m1->row3 + in_m2->row3; +} + +inline void mat4_sub(const mat4* in_m1, const float_t value, mat4* out_m) { + out_m->row0 = in_m1->row0 - value; + out_m->row1 = in_m1->row1 - value; + out_m->row2 = in_m1->row2 - value; + out_m->row3 = in_m1->row3 - value; +} + +inline void mat4_sub(const mat4* in_m1, const mat4* in_m2, mat4* out_m) { + out_m->row0 = in_m1->row0 - in_m2->row0; + out_m->row1 = in_m1->row1 - in_m2->row1; + out_m->row2 = in_m1->row2 - in_m2->row2; + out_m->row3 = in_m1->row3 - in_m2->row3; +} + +inline void mat4_mul(const mat4* in_m1, const float_t value, mat4* out_m) { + out_m->row0 = in_m1->row0 * value; + out_m->row1 = in_m1->row1 * value; + out_m->row2 = in_m1->row2 * value; + out_m->row3 = in_m1->row3 * value; +} + +inline void mat4_mul(const mat4* in_m1, const mat4* in_m2, mat4* out_m) { + __m128 t0; + __m128 t1; + __m128 t2; + __m128 t3; + __m128 xt; + __m128 y0; + __m128 y1; + __m128 y2; + __m128 y3; + y0 = vec4::load_xmm(in_m2->row0); + y1 = vec4::load_xmm(in_m2->row1); + y2 = vec4::load_xmm(in_m2->row2); + y3 = vec4::load_xmm(in_m2->row3); + xt = vec4::load_xmm(in_m1->row0); + t0 = _mm_mul_ps(y0, _mm_shuffle_ps(xt, xt, 0x00)); + t1 = _mm_mul_ps(y1, _mm_shuffle_ps(xt, xt, 0x55)); + t2 = _mm_mul_ps(y2, _mm_shuffle_ps(xt, xt, 0xAA)); + t3 = _mm_mul_ps(y3, _mm_shuffle_ps(xt, xt, 0xFF)); + out_m->row0 = vec4::store_xmm(_mm_add_ps(_mm_add_ps(t0, t1), _mm_add_ps(t2, t3))); + xt = vec4::load_xmm(in_m1->row1); + t0 = _mm_mul_ps(y0, _mm_shuffle_ps(xt, xt, 0x00)); + t1 = _mm_mul_ps(y1, _mm_shuffle_ps(xt, xt, 0x55)); + t2 = _mm_mul_ps(y2, _mm_shuffle_ps(xt, xt, 0xAA)); + t3 = _mm_mul_ps(y3, _mm_shuffle_ps(xt, xt, 0xFF)); + out_m->row1 = vec4::store_xmm(_mm_add_ps(_mm_add_ps(t0, t1), _mm_add_ps(t2, t3))); + xt = vec4::load_xmm(in_m1->row2); + t0 = _mm_mul_ps(y0, _mm_shuffle_ps(xt, xt, 0x00)); + t1 = _mm_mul_ps(y1, _mm_shuffle_ps(xt, xt, 0x55)); + t2 = _mm_mul_ps(y2, _mm_shuffle_ps(xt, xt, 0xAA)); + t3 = _mm_mul_ps(y3, _mm_shuffle_ps(xt, xt, 0xFF)); + out_m->row2 = vec4::store_xmm(_mm_add_ps(_mm_add_ps(t0, t1), _mm_add_ps(t2, t3))); + xt = vec4::load_xmm(in_m1->row3); + t0 = _mm_mul_ps(y0, _mm_shuffle_ps(xt, xt, 0x00)); + t1 = _mm_mul_ps(y1, _mm_shuffle_ps(xt, xt, 0x55)); + t2 = _mm_mul_ps(y2, _mm_shuffle_ps(xt, xt, 0xAA)); + t3 = _mm_mul_ps(y3, _mm_shuffle_ps(xt, xt, 0xFF)); + out_m->row3 = vec4::store_xmm(_mm_add_ps(_mm_add_ps(t0, t1), _mm_add_ps(t2, t3))); +} + +inline void mat4_mul_rotation(const mat4* in_m1, const vec3* in_axis, const float_t in_angle, mat4* out_m) { + quat q1 = quat(in_m1->row0.x, in_m1->row1.x, in_m1->row2.x, in_m1->row0.y, + in_m1->row1.y, in_m1->row2.y, in_m1->row0.z, in_m1->row1.z, in_m1->row2.z); + quat q2 = quat(*in_axis, in_angle); + quat q3 = quat::mul(q2, q1); + mat4_set(&q3, out_m); +} + +inline void mat4_transform_vector(const mat4* in_m1, const vec2* normal, vec2* normalOut) { + __m128 yt; + __m128 zt0; + __m128 zt1; + yt = vec2::load_xmm(*normal); + zt0 = _mm_mul_ps(vec4::load_xmm(in_m1->row0), _mm_shuffle_ps(yt, yt, 0x00)); + zt1 = _mm_mul_ps(vec4::load_xmm(in_m1->row1), _mm_shuffle_ps(yt, yt, 0x55)); + *normalOut = vec2::store_xmm(_mm_add_ps(zt0, zt1)); +} + +inline void mat4_transform_vector(const mat4* in_m1, const vec3* normal, vec3* normalOut) { + __m128 yt; + __m128 zt0; + __m128 zt1; + __m128 zt2; + yt = vec3::load_xmm(*normal); + zt0 = _mm_mul_ps(vec4::load_xmm(in_m1->row0), _mm_shuffle_ps(yt, yt, 0x00)); + zt1 = _mm_mul_ps(vec4::load_xmm(in_m1->row1), _mm_shuffle_ps(yt, yt, 0x55)); + zt2 = _mm_mul_ps(vec4::load_xmm(in_m1->row2), _mm_shuffle_ps(yt, yt, 0xAA)); + *normalOut = vec3::store_xmm(_mm_add_ps(_mm_add_ps(zt0, zt1), zt2)); +} + +inline void mat4_transform_vector(const mat4* in_m1, const vec4* normal, vec4* normalOut) { + __m128 yt; + __m128 zt0; + __m128 zt1; + __m128 zt2; + __m128 zt3; + yt = vec4::load_xmm(*normal); + zt0 = _mm_mul_ps(vec4::load_xmm(in_m1->row0), _mm_shuffle_ps(yt, yt, 0x00)); + zt1 = _mm_mul_ps(vec4::load_xmm(in_m1->row1), _mm_shuffle_ps(yt, yt, 0x55)); + zt2 = _mm_mul_ps(vec4::load_xmm(in_m1->row2), _mm_shuffle_ps(yt, yt, 0xAA)); + zt3 = _mm_mul_ps(vec4::load_xmm(in_m1->row3), _mm_shuffle_ps(yt, yt, 0xFF)); + *normalOut = vec4::store_xmm(_mm_add_ps(_mm_add_ps(zt0, zt1), _mm_add_ps(zt2, zt3))); +} + +inline void mat4_transform_point(const mat4* in_m1, const vec2* point, vec2* normalOut) { + __m128 yt; + __m128 zt0; + __m128 zt1; + __m128 zt2; + yt = vec2::load_xmm(*point); + zt0 = _mm_mul_ps(vec4::load_xmm(in_m1->row0), _mm_shuffle_ps(yt, yt, 0x00)); + zt1 = _mm_mul_ps(vec4::load_xmm(in_m1->row1), _mm_shuffle_ps(yt, yt, 0x55)); + zt2 = vec4::load_xmm(in_m1->row3); + *normalOut = vec2::store_xmm(_mm_add_ps(_mm_add_ps(zt0, zt1), zt2)); +} + +inline void mat4_transform_point(const mat4* in_m1, const vec3* point, vec3* normalOut) { + __m128 yt; + __m128 zt0; + __m128 zt1; + __m128 zt2; + __m128 zt3; + yt = vec3::load_xmm(*point); + zt0 = _mm_mul_ps(vec4::load_xmm(in_m1->row0), _mm_shuffle_ps(yt, yt, 0x00)); + zt1 = _mm_mul_ps(vec4::load_xmm(in_m1->row1), _mm_shuffle_ps(yt, yt, 0x55)); + zt2 = _mm_mul_ps(vec4::load_xmm(in_m1->row2), _mm_shuffle_ps(yt, yt, 0xAA)); + zt3 = vec4::load_xmm(in_m1->row3); + *normalOut = vec3::store_xmm(_mm_add_ps(_mm_add_ps(zt0, zt1), _mm_add_ps(zt2, zt3))); +} + +inline void mat4_inverse_transform_vector(const mat4* in_m1, const vec2* normal, vec2* normalOut) { + __m128 yt; + __m128 zt; + yt = vec2::load_xmm(*normal); + zt = _mm_mul_ps(yt, vec4::load_xmm(in_m1->row0)); + normalOut->x = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); + zt = _mm_mul_ps(yt, vec4::load_xmm(in_m1->row1)); + normalOut->y = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline void mat4_inverse_transform_vector(const mat4* in_m1, const vec3* normal, vec3* normalOut) { + __m128 yt; + __m128 zt; + yt = vec3::load_xmm(*normal); + zt = _mm_mul_ps(yt, vec4::load_xmm(in_m1->row0)); + zt = _mm_hadd_ps(zt, zt); + normalOut->x = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); + zt = _mm_mul_ps(yt, vec4::load_xmm(in_m1->row1)); + zt = _mm_hadd_ps(zt, zt); + normalOut->y = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); + zt = _mm_mul_ps(yt, vec4::load_xmm(in_m1->row2)); + zt = _mm_hadd_ps(zt, zt); + normalOut->z = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline void mat4_inverse_transform_vector(const mat4* in_m1, const vec4* normal, vec4* normalOut) { + __m128 yt; + __m128 zt; + yt = vec4::load_xmm(*normal); + zt = _mm_mul_ps(yt, vec4::load_xmm(in_m1->row0)); + zt = _mm_hadd_ps(zt, zt); + normalOut->x = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); + zt = _mm_mul_ps(yt, vec4::load_xmm(in_m1->row1)); + zt = _mm_hadd_ps(zt, zt); + normalOut->y = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); + zt = _mm_mul_ps(yt, vec4::load_xmm(in_m1->row2)); + zt = _mm_hadd_ps(zt, zt); + normalOut->z = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); + zt = _mm_mul_ps(yt, vec4::load_xmm(in_m1->row3)); + zt = _mm_hadd_ps(zt, zt); + normalOut->w = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline void mat4_inverse_transform_point(const mat4* in_m1, const vec2* point, vec2* pointOut) { + __m128 yt; + __m128 zt; + yt = vec2::load_xmm(*point); + yt = _mm_sub_ps(yt, vec4::load_xmm(in_m1->row3)); + zt = _mm_mul_ps(yt, vec4::load_xmm(in_m1->row0)); + pointOut->x = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); + zt = _mm_mul_ps(yt, vec4::load_xmm(in_m1->row1)); + pointOut->y = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline void mat4_inverse_transform_point(const mat4* in_m1, const vec3* point, vec3* pointOut) { + __m128 yt; + __m128 zt; + yt = vec3::load_xmm(*point); + yt = _mm_sub_ps(yt, vec4::load_xmm(in_m1->row3)); + zt = _mm_mul_ps(yt, vec4::load_xmm(in_m1->row0)); + zt = _mm_hadd_ps(zt, zt); + pointOut->x = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); + zt = _mm_mul_ps(yt, vec4::load_xmm(in_m1->row1)); + zt = _mm_hadd_ps(zt, zt); + pointOut->y = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); + zt = _mm_mul_ps(yt, vec4::load_xmm(in_m1->row2)); + zt = _mm_hadd_ps(zt, zt); + pointOut->z = _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline void mat4_transpose(const mat4* in_m1, mat4* out_m) { + __m128 xt0; + __m128 xt1; + __m128 xt2; + __m128 xt3; + __m128 yt0; + __m128 yt1; + __m128 yt2; + __m128 yt3; + xt0 = vec4::load_xmm(in_m1->row0); + xt1 = vec4::load_xmm(in_m1->row1); + xt2 = vec4::load_xmm(in_m1->row2); + xt3 = vec4::load_xmm(in_m1->row3); + yt0 = _mm_unpacklo_ps(xt0, xt1); + yt1 = _mm_unpackhi_ps(xt0, xt1); + yt2 = _mm_unpacklo_ps(xt2, xt3); + yt3 = _mm_unpackhi_ps(xt2, xt3); + out_m->row0 = vec4::store_xmm(_mm_movelh_ps(yt0, yt2)); + out_m->row1 = vec4::store_xmm(_mm_movehl_ps(yt2, yt0)); + out_m->row2 = vec4::store_xmm(_mm_movelh_ps(yt1, yt3)); + out_m->row3 = vec4::store_xmm(_mm_movehl_ps(yt3, yt1)); +} + +void mat4_invert(const mat4* in_m1, mat4* out_m) { + static const __m128 xor0 = { -0.0f, 0.0f, -0.0f, 0.0f }; + static const __m128 xor1 = { 0.0f, -0.0f, 0.0f, -0.0f }; + + __m128 xt0, xt1, xt2, xt3; + __m128 xt0x, xt1x, xt2x, xt3x; + __m128 xt0y, xt1y, xt2y, xt3y; + __m128 xt0z, xt1z, xt2z, xt3z; + __m128 xt0w, xt1w, xt2w, xt3w; + __m128 yt00, yt01, yt02, yt03, yt04, yt05, yt06, yt07, yt08, yt09, yt10, yt11; + __m128 zt0, zt1, zt2, zt3; + __m128 wt; + __m128 wt0, wt1, wt2; + __m128 t0, t1, t2, t3, t4, t5; + xt0 = vec4::load_xmm(in_m1->row0); + xt1 = vec4::load_xmm(in_m1->row1); + xt2 = vec4::load_xmm(in_m1->row2); + xt3 = vec4::load_xmm(in_m1->row3); + xt0x = _mm_shuffle_ps(xt0, xt0, 0x00); + xt0y = _mm_shuffle_ps(xt0, xt0, 0x55); + xt0z = _mm_shuffle_ps(xt0, xt0, 0xAA); + xt0w = _mm_shuffle_ps(xt0, xt0, 0xFF); + xt1x = _mm_shuffle_ps(xt1, xt1, 0x00); + xt1y = _mm_shuffle_ps(xt1, xt1, 0x55); + xt1z = _mm_shuffle_ps(xt1, xt1, 0xAA); + xt1w = _mm_shuffle_ps(xt1, xt1, 0xFF); + xt2x = _mm_shuffle_ps(xt2, xt2, 0x00); + xt2y = _mm_shuffle_ps(xt2, xt2, 0x55); + xt2z = _mm_shuffle_ps(xt2, xt2, 0xAA); + xt2w = _mm_shuffle_ps(xt2, xt2, 0xFF); + xt3x = _mm_shuffle_ps(xt3, xt3, 0x00); + xt3y = _mm_shuffle_ps(xt3, xt3, 0x55); + xt3z = _mm_shuffle_ps(xt3, xt3, 0xAA); + xt3w = _mm_shuffle_ps(xt3, xt3, 0xFF); + yt00 = _mm_movelh_ps(_mm_unpacklo_ps(xt1y, xt0y), xt0y); + yt01 = _mm_movelh_ps(xt2y, xt1y); + yt02 = _mm_movelh_ps(xt3y, _mm_unpacklo_ps(xt3y, xt2y)); + yt03 = _mm_movelh_ps(xt2z, xt1z); + yt04 = _mm_movelh_ps(xt3w, _mm_unpacklo_ps(xt3w, xt2w)); + yt05 = _mm_movelh_ps(xt2w, xt1w); + yt06 = _mm_movelh_ps(_mm_unpacklo_ps(xt3z, xt3z), _mm_unpacklo_ps(xt3z, xt2z)); + yt07 = _mm_movelh_ps(_mm_unpacklo_ps(xt1z, xt0z), xt0z); + yt08 = _mm_movelh_ps(_mm_unpacklo_ps(xt1w, xt0w), xt0w); + yt09 = _mm_movelh_ps(_mm_unpacklo_ps(xt1x, xt0x), xt0x); + yt10 = _mm_movelh_ps(xt2x, xt1x); + yt11 = _mm_movelh_ps(xt3x, _mm_unpacklo_ps(xt3x, xt2x)); + + t0 = _mm_sub_ps(_mm_mul_ps(yt03, yt04), _mm_mul_ps(yt05, yt06)); + t1 = _mm_sub_ps(_mm_mul_ps(yt07, yt04), _mm_mul_ps(yt08, yt06)); + t2 = _mm_sub_ps(_mm_mul_ps(yt07, yt05), _mm_mul_ps(yt08, yt03)); + + t3 = _mm_xor_ps(yt09, xor0); + t4 = _mm_xor_ps(yt10, xor1); + t5 = _mm_xor_ps(yt11, xor0); + + wt0 = _mm_mul_ps(_mm_xor_ps(yt00, xor1), t0); + wt1 = _mm_mul_ps(_mm_xor_ps(yt01, xor0), t1); + wt2 = _mm_mul_ps(_mm_xor_ps(yt02, xor1), t2); + zt0 = _mm_add_ps(_mm_add_ps(wt0, wt1), wt2); + + wt0 = _mm_mul_ps(t3, t0); + wt1 = _mm_mul_ps(t4, t1); + wt2 = _mm_mul_ps(t5, t2); + zt1 = _mm_add_ps(_mm_add_ps(wt0, wt1), wt2); + + wt0 = _mm_mul_ps(_mm_xor_ps(yt09, xor1), _mm_sub_ps(_mm_mul_ps(yt01, yt04), _mm_mul_ps(yt05, yt02))); + wt1 = _mm_mul_ps(_mm_xor_ps(yt10, xor0), _mm_sub_ps(_mm_mul_ps(yt00, yt04), _mm_mul_ps(yt08, yt02))); + wt2 = _mm_mul_ps(_mm_xor_ps(yt11, xor1), _mm_sub_ps(_mm_mul_ps(yt00, yt05), _mm_mul_ps(yt08, yt01))); + zt2 = _mm_add_ps(_mm_add_ps(wt0, wt1), wt2); + + wt0 = _mm_mul_ps(t3, _mm_sub_ps(_mm_mul_ps(yt01, yt06), _mm_mul_ps(yt03, yt02))); + wt1 = _mm_mul_ps(t4, _mm_sub_ps(_mm_mul_ps(yt00, yt06), _mm_mul_ps(yt07, yt02))); + wt2 = _mm_mul_ps(t5, _mm_sub_ps(_mm_mul_ps(yt00, yt03), _mm_mul_ps(yt07, yt01))); + zt3 = _mm_add_ps(_mm_add_ps(wt0, wt1), wt2); + + wt = _mm_movelh_ps(_mm_unpacklo_ps(zt0, zt1), _mm_unpacklo_ps(zt2, zt3)); + wt = _mm_mul_ps(xt0, wt); + wt = _mm_hadd_ps(wt, wt); + wt = _mm_hadd_ps(wt, wt); + if (_mm_cvtss_f32(wt) != 0.0f) + wt = _mm_div_ss(_mm_set_ss(1.0f), wt); + wt = _mm_shuffle_ps(wt, wt, 0); + out_m->row0 = vec4::store_xmm(_mm_mul_ps(zt0, wt)); + out_m->row1 = vec4::store_xmm(_mm_mul_ps(zt1, wt)); + out_m->row2 = vec4::store_xmm(_mm_mul_ps(zt2, wt)); + out_m->row3 = vec4::store_xmm(_mm_mul_ps(zt3, wt)); +} + +inline void mat4_invert_rotation(const mat4* in_m1, mat4* out_m) { + mat3 yt; + mat4_to_mat3(in_m1, &yt); + mat3_invert(&yt, &yt); + mat4_from_mat3(&yt, out_m); + out_m->row0.w = in_m1->row0.w; + out_m->row1.w = in_m1->row1.w; + out_m->row2.w = in_m1->row2.w; + out_m->row3 = in_m1->row3; +} + +inline void mat4_invert_fast(const mat4* in_m1, mat4* out_m) { + mat3 yt; + mat4_to_mat3(in_m1, &yt); + vec3 row3; + row3.x = vec3::dot(*(vec3*)&in_m1->row0, *(vec3*)&in_m1->row3); + row3.y = vec3::dot(*(vec3*)&in_m1->row1, *(vec3*)&in_m1->row3); + row3.z = vec3::dot(*(vec3*)&in_m1->row2, *(vec3*)&in_m1->row3); + mat3_transpose(&yt, &yt); + mat4_from_mat3(&yt, out_m); + *(vec3*)&out_m->row3 = -row3; + out_m->row0.w = 0.0f; + out_m->row1.w = 0.0f; + out_m->row2.w = 0.0f; + out_m->row3.w = 1.0f; +} + +inline void mat4_invert_rotation_fast(const mat4* in_m1, mat4* out_m) { + mat3 yt; + mat4_to_mat3(in_m1, &yt); + mat3_transpose(&yt, &yt); + mat4_from_mat3(&yt, out_m); + out_m->row0.w = in_m1->row0.w; + out_m->row1.w = in_m1->row1.w; + out_m->row2.w = in_m1->row2.w; + out_m->row3 = in_m1->row3; +} + +inline void mat4_normalize(const mat4* in_m1, mat4* out_m) { + __m128 det; + det = _mm_set_ss(mat4_determinant(in_m1)); + if (_mm_cvtss_f32(det) != 0.0f) + det = _mm_div_ss(_mm_set_ss(1.0f), det); + det = _mm_shuffle_ps(det, det, 0); + out_m->row0 = vec4::store_xmm(_mm_mul_ps(vec4::load_xmm(in_m1->row0), det)); + out_m->row1 = vec4::store_xmm(_mm_mul_ps(vec4::load_xmm(in_m1->row1), det)); + out_m->row2 = vec4::store_xmm(_mm_mul_ps(vec4::load_xmm(in_m1->row2), det)); + out_m->row3 = vec4::store_xmm(_mm_mul_ps(vec4::load_xmm(in_m1->row3), det)); +} + +inline void mat4_normalize_rotation(const mat4* in_m1, mat4* out_m) { + *(vec3*)&out_m->row0 = vec3::normalize(*(vec3*)&in_m1->row0); + *(vec3*)&out_m->row1 = vec3::normalize(*(vec3*)&in_m1->row1); + *(vec3*)&out_m->row2 = vec3::normalize(*(vec3*)&in_m1->row2); + out_m->row0.w = in_m1->row0.w; + out_m->row1.w = in_m1->row1.w; + out_m->row2.w = in_m1->row2.w; + out_m->row3 = in_m1->row3; +} + +inline float_t mat4_determinant(const mat4* in_m1) { + vec4 xt0; + vec4 xt1; + vec4 xt2; + vec4 xt3; + xt0 = in_m1->row0; + xt1 = in_m1->row1; + xt2 = in_m1->row2; + xt3 = in_m1->row3; + float_t b00 = xt0.x * xt1.y - xt0.y * xt1.x; + float_t b01 = xt0.x * xt1.z - xt0.z * xt1.x; + float_t b02 = xt0.x * xt1.w - xt0.w * xt1.x; + float_t b03 = xt0.y * xt1.z - xt0.z * xt1.y; + float_t b04 = xt0.y * xt1.w - xt0.w * xt1.y; + float_t b05 = xt0.z * xt1.w - xt0.w * xt1.z; + float_t b06 = xt2.x * xt3.y - xt2.y * xt3.x; + float_t b07 = xt2.x * xt3.z - xt2.z * xt3.x; + float_t b08 = xt2.x * xt3.w - xt2.w * xt3.x; + float_t b09 = xt2.y * xt3.z - xt2.z * xt3.y; + float_t b10 = xt2.y * xt3.w - xt2.w * xt3.y; + float_t b11 = xt2.z * xt3.w - xt2.w * xt3.z; + return b00 * b11 - b01 * b10 + b02 * b09 + b03 * b08 - b04 * b07 + b05 * b06; +} + +inline void mat4_rotate_x(float_t rad, mat4* out_m) { + float_t s = sinf(rad); + float_t c = cosf(rad); + *out_m = mat4_identity; + out_m->row1.y = c; + out_m->row1.z = s; + out_m->row2.y = -s; + out_m->row2.z = c; +} + +inline void mat4_rotate_y(float_t rad, mat4* out_m) { + float_t s = sinf(rad); + float_t c = cosf(rad); + *out_m = mat4_identity; + out_m->row0.x = c; + out_m->row0.z = -s; + out_m->row2.x = s; + out_m->row2.z = c; +} + +inline void mat4_rotate_z(float_t rad, mat4* out_m) { + float_t s = sinf(rad); + float_t c = cosf(rad); + *out_m = mat4_identity; + out_m->row0.x = c; + out_m->row0.y = s; + out_m->row1.x = -s; + out_m->row1.y = c; +} + +inline void mat4_rotate_x(float_t s, float_t c, mat4* out_m) { + *out_m = mat4_identity; + out_m->row1.y = c; + out_m->row1.z = s; + out_m->row2.y = -s; + out_m->row2.z = c; +} + +inline void mat4_rotate_y(float_t s, float_t c, mat4* out_m) { + *out_m = mat4_identity; + out_m->row0.x = c; + out_m->row0.z = -s; + out_m->row2.x = s; + out_m->row2.z = c; +} + +inline void mat4_rotate_z(float_t s, float_t c, mat4* out_m) { + *out_m = mat4_identity; + out_m->row0.x = c; + out_m->row0.y = s; + out_m->row1.x = -s; + out_m->row1.y = c; +} + +inline void mat4_rotate_xyz(float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m) { + mat4 dt; + dt = mat4_identity; + if (rad_x != 0.0f) + mat4_mul_rotate_x(&dt, rad_x, &dt); + if (rad_y != 0.0f) + mat4_mul_rotate_y(&dt, rad_y, &dt); + if (rad_z != 0.0f) + mat4_mul_rotate_z(&dt, rad_z, &dt); + *out_m = dt; +} + +inline void mat4_rotate_xzy(float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m) { + mat4 dt; + dt = mat4_identity; + if (rad_x != 0.0f) + mat4_mul_rotate_x(&dt, rad_x, &dt); + if (rad_z != 0.0f) + mat4_mul_rotate_z(&dt, rad_z, &dt); + if (rad_y != 0.0f) + mat4_mul_rotate_y(&dt, rad_y, &dt); + *out_m = dt; +} + +inline void mat4_rotate_yxz(float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m) { + mat4 dt; + dt = mat4_identity; + if (rad_y != 0.0f) + mat4_mul_rotate_y(&dt, rad_y, &dt); + if (rad_x != 0.0f) + mat4_mul_rotate_x(&dt, rad_x, &dt); + if (rad_z != 0.0f) + mat4_mul_rotate_z(&dt, rad_z, &dt); + *out_m = dt; +} + +inline void mat4_rotate_yzx(float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m) { + mat4 dt; + dt = mat4_identity; + if (rad_y != 0.0f) + mat4_mul_rotate_y(&dt, rad_y, &dt); + if (rad_z != 0.0f) + mat4_mul_rotate_z(&dt, rad_z, &dt); + if (rad_x != 0.0f) + mat4_mul_rotate_x(&dt, rad_x, &dt); + *out_m = dt; +} + +inline void mat4_rotate_zxy(float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m) { + mat4 dt; + dt = mat4_identity; + if (rad_z != 0.0f) + mat4_mul_rotate_z(&dt, rad_z, &dt); + if (rad_x != 0.0f) + mat4_mul_rotate_x(&dt, rad_x, &dt); + if (rad_y != 0.0f) + mat4_mul_rotate_y(&dt, rad_y, &dt); + *out_m = dt; +} + +inline void mat4_rotate_zyx(float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m) { + mat4 dt; + dt = mat4_identity; + if (rad_z != 0.0f) + mat4_mul_rotate_z(&dt, rad_z, &dt); + if (rad_y != 0.0f) + mat4_mul_rotate_y(&dt, rad_y, &dt); + if (rad_x != 0.0f) + mat4_mul_rotate_x(&dt, rad_x, &dt); + *out_m = dt; +} + +inline void mat4_mul_rotate_x(const mat4* in_m1, float_t rad, mat4* out_m) { + __m128 t1; + __m128 t2; + __m128 y0; + __m128 y1; + __m128 y2; + __m128 y3; + float_t s = sinf(rad); + float_t c = cosf(rad); + y0 = vec4::load_xmm(in_m1->row0); + y1 = vec4::load_xmm(in_m1->row1); + y2 = vec4::load_xmm(in_m1->row2); + y3 = vec4::load_xmm(in_m1->row3); + out_m->row0 = vec4::store_xmm(y0); + t1 = _mm_mul_ps(y1, vec4::load_xmm(c)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(s)); + out_m->row1 = vec4::store_xmm(_mm_add_ps(t1, t2)); + t1 = _mm_mul_ps(y1, vec4::load_xmm(-s)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(c)); + out_m->row2 = vec4::store_xmm(_mm_add_ps(t1, t2)); + out_m->row3 = vec4::store_xmm(y3); +} + +inline void mat4_mul_rotate_y(const mat4* in_m1, float_t rad, mat4* out_m) { + __m128 t0; + __m128 t2; + __m128 y0; + __m128 y1; + __m128 y2; + __m128 y3; + float_t s = sinf(rad); + float_t c = cosf(rad); + y0 = vec4::load_xmm(in_m1->row0); + y1 = vec4::load_xmm(in_m1->row1); + y2 = vec4::load_xmm(in_m1->row2); + y3 = vec4::load_xmm(in_m1->row3); + t0 = _mm_mul_ps(y0, vec4::load_xmm(c)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(-s)); + out_m->row0 = vec4::store_xmm(_mm_add_ps(t0, t2)); + out_m->row1 = vec4::store_xmm(y1); + t0 = _mm_mul_ps(y0, vec4::load_xmm(s)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(c)); + out_m->row2 = vec4::store_xmm(_mm_add_ps(t0, t2)); + out_m->row3 = vec4::store_xmm(y3); +} + +inline void mat4_mul_rotate_z(const mat4* in_m1, float_t rad, mat4* out_m) { + __m128 t0; + __m128 t1; + __m128 y0; + __m128 y1; + __m128 y2; + __m128 y3; + float_t s = sinf(rad); + float_t c = cosf(rad); + y0 = vec4::load_xmm(in_m1->row0); + y1 = vec4::load_xmm(in_m1->row1); + y2 = vec4::load_xmm(in_m1->row2); + y3 = vec4::load_xmm(in_m1->row3); + t0 = _mm_mul_ps(y0, vec4::load_xmm(c)); + t1 = _mm_mul_ps(y1, vec4::load_xmm(s)); + out_m->row0 = vec4::store_xmm(_mm_add_ps(t0, t1)); + t0 = _mm_mul_ps(y0, vec4::load_xmm(-s)); + t1 = _mm_mul_ps(y1, vec4::load_xmm(c)); + out_m->row1 = vec4::store_xmm(_mm_add_ps(t0, t1)); + out_m->row2 = vec4::store_xmm(y2); + out_m->row3 = vec4::store_xmm(y3); +} + +inline void mat4_mul_rotate_x(const mat4* in_m1, float_t s, float_t c, mat4* out_m) { + __m128 t1; + __m128 t2; + __m128 y0; + __m128 y1; + __m128 y2; + __m128 y3; + y0 = vec4::load_xmm(in_m1->row0); + y1 = vec4::load_xmm(in_m1->row1); + y2 = vec4::load_xmm(in_m1->row2); + y3 = vec4::load_xmm(in_m1->row3); + out_m->row0 = vec4::store_xmm(y0); + t1 = _mm_mul_ps(y1, vec4::load_xmm(c)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(s)); + out_m->row1 = vec4::store_xmm(_mm_add_ps(t1, t2)); + t1 = _mm_mul_ps(y1, vec4::load_xmm(-s)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(c)); + out_m->row2 = vec4::store_xmm(_mm_add_ps(t1, t2)); + out_m->row3 = vec4::store_xmm(y3); +} + +inline void mat4_mul_rotate_y(const mat4* in_m1, float_t s, float_t c, mat4* out_m) { + __m128 t0; + __m128 t2; + __m128 y0; + __m128 y1; + __m128 y2; + __m128 y3; + y0 = vec4::load_xmm(in_m1->row0); + y1 = vec4::load_xmm(in_m1->row1); + y2 = vec4::load_xmm(in_m1->row2); + y3 = vec4::load_xmm(in_m1->row3); + t0 = _mm_mul_ps(y0, vec4::load_xmm(c)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(-s)); + out_m->row0 = vec4::store_xmm(_mm_add_ps(t0, t2)); + out_m->row1 = vec4::store_xmm(y1); + t0 = _mm_mul_ps(y0, vec4::load_xmm(s)); + t2 = _mm_mul_ps(y2, vec4::load_xmm(c)); + out_m->row2 = vec4::store_xmm(_mm_add_ps(t0, t2)); + out_m->row3 = vec4::store_xmm(y3); +} + +inline void mat4_mul_rotate_z(const mat4* in_m1, float_t s, float_t c, mat4* out_m) { + __m128 t0; + __m128 t1; + __m128 y0; + __m128 y1; + __m128 y2; + __m128 y3; + y0 = vec4::load_xmm(in_m1->row0); + y1 = vec4::load_xmm(in_m1->row1); + y2 = vec4::load_xmm(in_m1->row2); + y3 = vec4::load_xmm(in_m1->row3); + t0 = _mm_mul_ps(y0, vec4::load_xmm(c)); + t1 = _mm_mul_ps(y1, vec4::load_xmm(s)); + out_m->row0 = vec4::store_xmm(_mm_add_ps(t0, t1)); + t0 = _mm_mul_ps(y0, vec4::load_xmm(-s)); + t1 = _mm_mul_ps(y1, vec4::load_xmm(c)); + out_m->row1 = vec4::store_xmm(_mm_add_ps(t0, t1)); + out_m->row2 = vec4::store_xmm(y2); + out_m->row3 = vec4::store_xmm(y3); +} + +inline void mat4_mul_rotate_xyz(const mat4* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m) { + mat4 dt; + dt = *in_m1; + if (rad_x != 0.0f) + mat4_mul_rotate_x(&dt, rad_x, &dt); + if (rad_y != 0.0f) + mat4_mul_rotate_y(&dt, rad_y, &dt); + if (rad_z != 0.0f) + mat4_mul_rotate_z(&dt, rad_z, &dt); + *out_m = dt; +} + +inline void mat4_mul_rotate_xzy(const mat4* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m) { + mat4 dt; + dt = *in_m1; + if (rad_x != 0.0f) + mat4_mul_rotate_x(&dt, rad_x, &dt); + if (rad_z != 0.0f) + mat4_mul_rotate_z(&dt, rad_z, &dt); + if (rad_y != 0.0f) + mat4_mul_rotate_y(&dt, rad_y, &dt); + *out_m = dt; +} + +inline void mat4_mul_rotate_yxz(const mat4* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m) { + mat4 dt; + dt = *in_m1; + if (rad_y != 0.0f) + mat4_mul_rotate_y(&dt, rad_y, &dt); + if (rad_x != 0.0f) + mat4_mul_rotate_x(&dt, rad_x, &dt); + if (rad_z != 0.0f) + mat4_mul_rotate_z(&dt, rad_z, &dt); + *out_m = dt; +} + +inline void mat4_mul_rotate_yzx(const mat4* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m) { + mat4 dt; + dt = *in_m1; + if (rad_y != 0.0f) + mat4_mul_rotate_y(&dt, rad_y, &dt); + if (rad_z != 0.0f) + mat4_mul_rotate_z(&dt, rad_z, &dt); + if (rad_x != 0.0f) + mat4_mul_rotate_x(&dt, rad_x, &dt); + *out_m = dt; +} + +inline void mat4_mul_rotate_zxy(const mat4* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m) { + mat4 dt; + dt = *in_m1; + if (rad_z != 0.0f) + mat4_mul_rotate_z(&dt, rad_z, &dt); + if (rad_x != 0.0f) + mat4_mul_rotate_x(&dt, rad_x, &dt); + if (rad_y != 0.0f) + mat4_mul_rotate_y(&dt, rad_y, &dt); + *out_m = dt; +} + +inline void mat4_mul_rotate_zyx(const mat4* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m) { + mat4 dt; + dt = *in_m1; + if (rad_z != 0.0f) + mat4_mul_rotate_z(&dt, rad_z, &dt); + if (rad_y != 0.0f) + mat4_mul_rotate_y(&dt, rad_y, &dt); + if (rad_x != 0.0f) + mat4_mul_rotate_x(&dt, rad_x, &dt); + *out_m = dt; +} + +inline void mat4_scale(float_t sx, float_t sy, float_t sz, mat4* out_m) { + *out_m = mat4_identity; + out_m->row0.x = sx; + out_m->row1.y = sy; + out_m->row2.z = sz; +} + +inline void mat4_scale_x(float_t s, mat4* out_m) { + *out_m = mat4_identity; + out_m->row0.x = s; +} + +inline void mat4_scale_y(float_t s, mat4* out_m) { + *out_m = mat4_identity; + out_m->row1.y = s; +} + +inline void mat4_scale_z(float_t s, mat4* out_m) { + *out_m = mat4_identity; + out_m->row2.z = s; +} + +inline void mat4_mul_scale(const mat4* in_m1, float_t sx, float_t sy, float_t sz, float_t sw, mat4* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + if (sx != 1.0f || sy != 1.0f || sz != 1.0f || sw != 1.0f) { + out_m->row0 *= sx; + out_m->row1 *= sy; + out_m->row2 *= sz; + out_m->row3 *= sw; + } +} + +inline void mat4_mul_scale_x(const mat4* in_m1, float_t s, mat4* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + out_m->row0 *= s; +} + +inline void mat4_mul_scale_y(const mat4* in_m1, float_t s, mat4* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + out_m->row1 *= s; +} + +inline void mat4_mul_scale_z(const mat4* in_m1, float_t s, mat4* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + out_m->row2 *= s; +} + +inline void mat4_scale_w_mult(const mat4* in_m1, float_t s, mat4* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + out_m->row3 *= s; +} + +inline void mat4_scale_rot(const mat4* in_m1, float_t sx, float_t sy, float_t sz, mat4* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + if (sx != 1.0f || sy != 1.0f || sz != 1.0f) { + *(vec3*)&out_m->row0 *= sx; + *(vec3*)&out_m->row1 *= sy; + *(vec3*)&out_m->row2 *= sz; + } + else + *out_m = *in_m1; +} + +inline void mat4_scale_x_rot(const mat4* in_m1, float_t s, mat4* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + *(vec3*)&out_m->row0 *= s; +} + +inline void mat4_scale_y_rot(const mat4* in_m1, float_t s, mat4* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + *(vec3*)&out_m->row1 *= s; +} + +inline void mat4_scale_z_rot(const mat4* in_m1, float_t s, mat4* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + *(vec3*)&out_m->row2 *= s; +} + +inline void mat4_translate(float_t tx, float_t ty, float_t tz, mat4* out_m) { + *out_m = mat4_identity; + out_m->row3.x = tx; + out_m->row3.y = ty; + out_m->row3.z = tz; +} + +inline void mat4_translate_x(float_t t, mat4* out_m) { + *out_m = mat4_identity; + out_m->row3.x = t; +} + +inline void mat4_translate_y(float_t t, mat4* out_m) { + *out_m = mat4_identity; + out_m->row3.y = t; +} + +inline void mat4_translate_z(float_t t, mat4* out_m) { + *out_m = mat4_identity; + out_m->row3.z = t; +} + +inline void mat4_mul_translate(const mat4* in_m1, float_t tx, float_t ty, float_t tz, mat4* out_m) { + __m128 yt; + __m128 yt0; + __m128 yt1; + __m128 yt2; + __m128 yt3; + if (in_m1 != out_m) + *out_m = *in_m1; + if (tx != 0.0f || ty != 0.0f || tz != 0.0f) { + yt0 = _mm_mul_ps(vec4::load_xmm(in_m1->row0), vec4::load_xmm(tx)); + yt1 = _mm_mul_ps(vec4::load_xmm(in_m1->row1), vec4::load_xmm(ty)); + yt2 = _mm_mul_ps(vec4::load_xmm(in_m1->row2), vec4::load_xmm(tz)); + yt3 = vec4::load_xmm(in_m1->row3); + yt = _mm_add_ps(_mm_add_ps(yt0, yt1), _mm_add_ps(yt2, yt3)); + *(vec3*)&out_m->row3 = vec3::store_xmm(yt); + } +} + +inline void mat4_mul_translate_x(const mat4* in_m1, float_t t, mat4* out_m) { + __m128 yt0; + __m128 yt1; + if (in_m1 != out_m) + *out_m = *in_m1; + if (t != 0.0f) { + yt0 = vec4::load_xmm(in_m1->row0); + yt1 = vec4::load_xmm(in_m1->row3); + yt0 = _mm_add_ps(_mm_mul_ps(yt0, vec4::load_xmm(t)), yt1); + *(vec3*)&out_m->row3 = vec3::store_xmm(yt0); + } +} + +inline void mat4_mul_translate_y(const mat4* in_m1, float_t t, mat4* out_m) { + __m128 yt0; + __m128 yt1; + if (in_m1 != out_m) + *out_m = *in_m1; + if (t != 0.0f) { + yt0 = vec4::load_xmm(in_m1->row1); + yt1 = vec4::load_xmm(in_m1->row3); + yt0 = _mm_add_ps(_mm_mul_ps(yt0, vec4::load_xmm(t)), yt1); + *(vec3*)&out_m->row3 = vec3::store_xmm(yt0); + } +} + +inline void mat4_mul_translate_z(const mat4* in_m1, float_t t, mat4* out_m) { + __m128 yt0; + __m128 yt1; + if (in_m1 != out_m) + *out_m = *in_m1; + if (t != 0.0f) { + yt0 = vec4::load_xmm(in_m1->row2); + yt1 = vec4::load_xmm(in_m1->row3); + yt0 = _mm_add_ps(_mm_mul_ps(yt0, vec4::load_xmm(t)), yt1); + *(vec3*)&out_m->row3 = vec3::store_xmm(yt0); + } +} + +inline void mat4_add_translate(const mat4* in_m1, float_t tx, float_t ty, float_t tz, mat4* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + if (tx != 0.0f || ty != 0.0f || tz != 0.0f) + out_m->row3 = vec4::store_xmm(_mm_add_ps(vec4::load_xmm(in_m1->row3), vec4::load_xmm(vec4(tx, ty, tz, 0.0f)))); +} + +inline void mat4_add_translate_x(const mat4* in_m1, float_t t, mat4* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + if (t != 0.0f) + out_m->row3.x += t; +} + +inline void mat4_add_translate_y(const mat4* in_m1, float_t t, mat4* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + if (t != 0.0f) + out_m->row3.y += t; +} + +inline void mat4_add_translate_z(const mat4* in_m1, float_t t, mat4* out_m) { + if (in_m1 != out_m) + *out_m = *in_m1; + if (t != 0.0f) + out_m->row3.z += t; +} + +inline void mat4_from_mat3(const mat3* in_m1, mat4* out_m) { + *(vec3*)&out_m->row0 = in_m1->row0; + out_m->row0.w = 0.0f; + *(vec3*)&out_m->row1 = in_m1->row1; + out_m->row1.w = 0.0f; + *(vec3*)&out_m->row2 = in_m1->row2; + out_m->row2.w = 0.0f; + out_m->row3 = { 0.0f, 0.0f, 0.0f, 1.0f }; +} + +inline void mat4_from_mat3_inverse(const mat3* in_m1, mat4* out_m) { + mat3 yt; + + mat3_invert(in_m1, &yt); + *(vec3*)&out_m->row0 = yt.row0; + out_m->row0.w = 0.0f; + *(vec3*)&out_m->row1 = yt.row1; + out_m->row1.w = 0.0f; + *(vec3*)&out_m->row2 = yt.row2; + out_m->row2.w = 0.0f; + out_m->row3 = { 0.0f, 0.0f, 0.0f, 1.0f }; +} + +inline void mat4_clear_rot(const mat4* in_m1, mat4* out_m) { + out_m->row0 = mat4_identity.row0; + out_m->row1 = mat4_identity.row1; + out_m->row2 = mat4_identity.row2; + out_m->row3 = in_m1->row3; +} + +inline void mat4_clear_trans(const mat4* in_m1, mat4* out_m) { + if (in_m1 != out_m) { + out_m->row0 = in_m1->row0; + out_m->row1 = in_m1->row1; + out_m->row2 = in_m1->row2; + } + out_m->row3 = { 0.0f, 0.0f, 0.0f, 1.0f }; +} + +inline void mat4_get_rotation(const mat4* in_m1, vec3* out_rad) { + if (-in_m1->row0.z >= 1.0f) + out_rad->y = (float_t)M_PI_2; + else if (-in_m1->row0.z <= -1.0f) + out_rad->y = (float_t)-M_PI_2; + else + out_rad->y = asinf(-in_m1->row0.z); + + if (fabsf(in_m1->row0.z) < 0.99999899f) { + out_rad->x = atan2f(in_m1->row1.z, in_m1->row2.z); + out_rad->z = atan2f(in_m1->row0.y, in_m1->row0.x); + } + else { + out_rad->x = 0.0f; + out_rad->z = atan2f(in_m1->row2.y, in_m1->row1.y); + if (in_m1->row0.z > 0.0f) + out_rad->z = -out_rad->z; + } +} + +inline void mat4_get_scale(const mat4* in_m1, vec3* out_s) { + out_s->x = vec4::length(in_m1->row0); + out_s->y = vec4::length(in_m1->row1); + out_s->z = vec4::length(in_m1->row2); +} + +inline void mat4_get_translation(const mat4* in_m1, vec3* out_t) { + *out_t = *(vec3*)&in_m1->row3; +} + +inline void mat4_set_translation(mat4* in_m1, const vec3* in_t) { + *(vec3*)&in_m1->row3 = *in_t; +} + +inline float_t mat4_get_max_scale(const mat4* in_m1) { + mat4 mat; + mat4_transpose(in_m1, &mat); + + float_t length; + float_t max = 0.0f; + length = vec3::length(*(vec3*)&mat.row0); + if (max < length) + max = length; + length = vec3::length(*(vec3*)&mat.row1); + if (max < length) + max = length; + length = vec3::length(*(vec3*)&mat.row2); + if (max < length) + max = length; + return max; +} + +inline void mat4_blend(const mat4* in_m1, const mat4* in_m2, mat4* out_m, float_t blend) { + quat q0; + quat q1; + quat q2; + + q0 = quat(in_m1->row0.x, in_m1->row1.x, in_m1->row2.x, in_m1->row0.y, + in_m1->row1.y, in_m1->row2.y, in_m1->row0.z, in_m1->row1.z, in_m1->row2.z); + q0 = quat::normalize(q0); + q1 = quat(in_m2->row0.x, in_m2->row1.x, in_m2->row2.x, in_m2->row0.y, + in_m2->row1.y, in_m2->row2.y, in_m2->row0.z, in_m2->row1.z, in_m2->row2.z); + q1 = quat::normalize(q1); + + vec3 t0; + vec3 t1; + vec3 t2; + mat4_get_translation(in_m1, &t0); + mat4_get_translation(in_m2, &t1); + + q2 = quat::lerp(q0, q1, blend); + t2 = vec3::lerp(t0, t1, blend); + + mat4_set(&q2, out_m); + mat4_set_translation(out_m, &t2); +} + +inline void mat4_blend_rotation(const mat4* in_m1, const mat4* in_m2, mat4* out_m, float_t blend) { + quat q1 = quat(in_m1->row0.x, in_m1->row1.x, in_m1->row2.x, in_m1->row0.y, + in_m1->row1.y, in_m1->row2.y, in_m1->row0.z, in_m1->row1.z, in_m1->row2.z); + quat q2 = quat(in_m2->row0.x, in_m2->row1.x, in_m2->row2.x, in_m2->row0.y, + in_m2->row1.y, in_m2->row2.y, in_m2->row0.z, in_m2->row1.z, in_m2->row2.z); + quat q3 = quat::slerp(q1, q2, blend); + mat4_set(&q3, out_m); +} + +void mat4_lerp_rotation(const mat4* in_m1, const mat4* in_m2, mat4* out_m, float_t blend) { + vec3 m0; + vec3 m1; + m0 = vec3::lerp(*(vec3*)&in_m1->row0, *(vec3*)&in_m2->row0, blend); + m1 = vec3::lerp(*(vec3*)&in_m1->row1, *(vec3*)&in_m2->row1, blend); + + float_t m0_len_sq = vec3::length_squared(m0); + float_t m1_len_sq = vec3::length_squared(m1); + + if (m0_len_sq <= 0.000001f || m1_len_sq <= 0.000001f) { + *out_m = *in_m2; + return; + } + + vec3 m2; + m2 = vec3::cross(m0, m1); + m1 = vec3::cross(m2, m0); + + float_t m2_len_sq; + m1_len_sq = vec3::length_squared(m1); + m2_len_sq = vec3::length_squared(m2); + if (m2_len_sq <= 0.000001f || m1_len_sq <= 0.000001) { + *out_m = *in_m2; + return; + } + + float_t m0_len = sqrtf(m0_len_sq); + if (m0_len != 0.0f) + m0 *= 1.0f / m0_len; + + float_t m1_len = sqrtf(m1_len_sq); + if (m1_len != 0.0f) + m1 *= 1.0f / m1_len; + + float_t m2_len = sqrtf(m2_len_sq); + if (m2_len != 0.0f) + m2 *= 1.0f / m2_len; + + *out_m = mat4_identity; + *(vec3*)&out_m->row0 = m0; + *(vec3*)&out_m->row1 = m1; + *(vec3*)&out_m->row2 = m2; +} + +inline void mat4_frustrum(double_t left, double_t right, + double_t bottom, double_t top, double_t z_near, double_t z_far, mat4* out_m) { + *out_m = mat4_null; + out_m->row0.x = (float_t)((2.0 * z_near) / (right - left)); + out_m->row1.y = (float_t)((2.0 * z_near) / (top - bottom)); + out_m->row2.x = (float_t)((right + left) / (right - left)); + out_m->row2.y = (float_t)((top + bottom) / (top - bottom)); + out_m->row2.z = -(float_t)((z_far + z_near) / (z_far - z_near)); + out_m->row2.w = -1.0f; + out_m->row3.z = -(float_t)((2.0 * z_far * z_near) / (z_far - z_near)); +} + +inline void mat4_ortho(double_t left, double_t right, + double_t bottom, double_t top, double_t z_near, double_t z_far, mat4* out_m) { + *out_m = mat4_null; + out_m->row0.x = (float_t)(2.0 / (right - left)); + out_m->row1.y = (float_t)(2.0 / (top - bottom)); + out_m->row2.z = (float_t)(-2.0 / (z_far - z_near)); + out_m->row3.x = -(float_t)((right + left) / (right - left)); + out_m->row3.y = -(float_t)((top + bottom) / (top - bottom)); + out_m->row3.z = -(float_t)((z_far + z_near) / (z_far - z_near)); + out_m->row3.w = 1.0f; +} + +inline void mat4_persp(double_t fov_y, double_t aspect, double_t z_near, double_t z_far, mat4* out_m) { + double_t tan_fov = tan(fov_y * 0.5); + + *out_m = mat4_null; + out_m->row0.x = (float_t)(1.0 / (aspect * tan_fov)); + out_m->row1.y = (float_t)(1.0 / tan_fov); + out_m->row2.z = -(float_t)((z_far + z_near) / (z_far - z_near)); + out_m->row2.w = -1.0f; + out_m->row3.z = -(float_t)((2.0 * z_far * z_near) / (z_far - z_near)); +} + +inline void mat4_look_at(const vec3* eye, const vec3* target, const vec3* up, mat4* out_m) { + vec3 x_axis, y_axis, z_axis; + vec3 xyz; + + z_axis = vec3::normalize(*eye - *target); + + x_axis = vec3::normalize(vec3::cross(*up, z_axis)); + if (vec3::length(x_axis) == 0.0f) + x_axis = { 1.0f, 0.0f, 0.0f }; + + y_axis = vec3::normalize(vec3::cross(z_axis, x_axis)); + + xyz.x = vec3::dot(x_axis, *eye); + xyz.y = vec3::dot(y_axis, *eye); + xyz.z = vec3::dot(z_axis, *eye); + + out_m->row0 = { x_axis.x, y_axis.x, z_axis.x, 0.0f }; + out_m->row1 = { x_axis.y, y_axis.y, z_axis.y, 0.0f }; + out_m->row2 = { x_axis.z, y_axis.z, z_axis.z, 0.0f }; + *(vec3*)&out_m->row3 = -xyz; + out_m->row3.w = 1.0f; +} + +inline void mat4_look_at(const vec3* eye, const vec3* target, mat4* out_m) { + vec3 up = { 0.0f, 1.0f, 0.0f }; + vec3 dir; + dir = *target - *eye; + if (vec3::length_squared(dir) <= 0.000001f) { + up.x = 0.0f; + up.y = 0.0f; + if (dir.z < 0.0f) + up.z = 1.0f; + else + up.z = -1.0f; + } + + mat4_look_at(eye, target, &up, out_m); +} diff --git a/src/KKdLib/mat.hpp b/src/KKdLib/mat.hpp new file mode 100644 index 0000000..4f0cde1 --- /dev/null +++ b/src/KKdLib/mat.hpp @@ -0,0 +1,345 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "default.hpp" +#include "vec.hpp" +#include "quat.hpp" + +struct mat3 { + vec3 row0; + vec3 row1; + vec3 row2; + + inline mat3() : row0(), row1(), row2() { + + } + + inline mat3(vec3 row0, vec3 row1, vec3 row2) : + row0(row0), row1(row1), row2(row2) { + + } +}; + +struct mat4 { + vec4 row0; + vec4 row1; + vec4 row2; + vec4 row3; + + inline mat4() : row0(), row1(), row2(), row3() { + + } + + inline mat4(vec4 row0, vec4 row1, vec4 row2, vec4 row3) : + row0(row0), row1(row1), row2(row2), row3(row3) { + + } +}; + +extern const mat3 mat3_identity; +extern const mat3 mat3_null; +extern const mat4 mat4_identity; +extern const mat4 mat4_null; + +extern void mat3_set(const quat* in_q1, mat3* out_m); +extern void mat3_set(const vec3* in_axis, const float_t in_angle, mat3* out_m); +extern void mat3_set(const vec3* in_axis, const float_t s, const float_t c, mat3* out_m); +extern void mat3_add(const mat3* in_m1, const float_t value, mat3* out_m); +extern void mat3_add(const mat3* in_m1, const mat3* in_m2, mat3* out_m); +extern void mat3_sub(const mat3* in_m1, const float_t value, mat3* out_m); +extern void mat3_sub(const mat3* in_m1, const mat3* in_m2, mat3* out_m); +extern void mat3_mul(const mat3* in_m1, const float_t value, mat3* out_m); +extern void mat3_mul(const mat3* in_m1, const mat3* in_m2, mat3* out_m); +extern void mat3_mul(const mat3* in_m1, const vec3* in_axis, const float_t in_angle, mat3* out_m); +extern void mat3_transform_vector(const mat3* in_m1, const vec2* normal, vec2* normalOut); +extern void mat3_transform_vector(const mat3* in_m1, const vec3* normal, vec3* normalOut); +extern void mat3_inverse_transform_vector(const mat3* in_m1, const vec2* normal, vec2* normalOut); +extern void mat3_inverse_transform_vector(const mat3* in_m1, const vec3* normal, vec3* normalOut); +extern void mat3_transpose(const mat3* in_m1, mat3* out_m); +extern void mat3_invert(const mat3* in_m1, mat3* out_m); +extern void mat3_invert_fast(const mat3* in_m1, mat3* out_m); +extern void mat3_normalize(const mat3* in_m1, mat3* out_m); +extern void mat3_normalize_rotation(const mat3* in_m1, mat3* out_m); +extern float_t mat3_determinant(const mat3* in_m1); +extern void mat3_rotate_x(float_t rad, mat3* out_m); +extern void mat3_rotate_y(float_t rad, mat3* out_m); +extern void mat3_rotate_z(float_t rad, mat3* out_m); +extern void mat3_rotate_x(float_t s, float_t c, mat3* out_m); +extern void mat3_rotate_y(float_t s, float_t c, mat3* out_m); +extern void mat3_rotate_z(float_t s, float_t c, mat3* out_m); +extern void mat3_rotate_xyz(float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m); +extern void mat3_rotate_xzy(float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m); +extern void mat3_rotate_yxz(float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m); +extern void mat3_rotate_yzx(float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m); +extern void mat3_rotate_zxy(float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m); +extern void mat3_rotate_zyx(float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m); +extern void mat3_mul_rotate_x(const mat3* in_m1, float_t rad, mat3* out_m); +extern void mat3_mul_rotate_y(const mat3* in_m1, float_t rad, mat3* out_m); +extern void mat3_mul_rotate_z(const mat3* in_m1, float_t rad, mat3* out_m); +extern void mat3_mul_rotate_x(const mat3* in_m1, float_t s, float_t c, mat3* out_m); +extern void mat3_mul_rotate_y(const mat3* in_m1, float_t s, float_t c, mat3* out_m); +extern void mat3_mul_rotate_z(const mat3* in_m1, float_t s, float_t c, mat3* out_m); +extern void mat3_mul_rotate_xyz(const mat3* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m); +extern void mat3_mul_rotate_xzy(const mat3* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m); +extern void mat3_mul_rotate_yxz(const mat3* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m); +extern void mat3_mul_rotate_yzx(const mat3* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m); +extern void mat3_mul_rotate_zxy(const mat3* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m); +extern void mat3_mul_rotate_zyx(const mat3* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat3* out_m); +extern void mat3_scale(float_t sx, float_t sy, float_t sz, mat3* out_m); +extern void mat3_scale_x(float_t s, mat3* out_m); +extern void mat3_scale_y(float_t s, mat3* out_m); +extern void mat3_scale_z(float_t s, mat3* out_m); +extern void mat3_mul_scale(const mat3* in_m1, float_t sx, float_t sy, float_t sz, mat3* out_m); +extern void mat3_mul_scale_x(const mat3* in_m1, float_t s, mat3* out_m); +extern void mat3_mul_scale_y(const mat3* in_m1, float_t s, mat3* out_m); +extern void mat3_mul_scale_z(const mat3* in_m1, float_t s, mat3* out_m); +extern void mat3_get_rotation(const mat3* in_m1, vec3* out_rad); +extern void mat3_get_scale(const mat3* in_m1, vec3* out_s); +extern float_t mat3_get_max_scale(const mat3* in_m1); + +extern void mat4_set(const quat* in_q1, mat4* out_m); +extern void mat4_set(const vec3* in_v1, const vec3* in_v2, mat4* out_m); +extern void mat4_set(const vec3* in_axis, const float_t in_angle, mat4* out_m); +extern void mat4_set(const vec3* in_axis, const float_t s, const float_t c, mat4* out_m); +extern void mat4_set_rotation(mat4* in_m1, const quat* in_q1); +extern void mat4_set_rotation(mat4* in_m1, const vec3* in_axis, const float_t in_angle); +extern void mat4_set_rotation(mat4* in_m1, const vec3* in_axis, const float_t s, const float_t c); +extern void mat4_add(const mat4* in_m1, const float_t value, mat4* out_m); +extern void mat4_add(const mat4* in_m1, const mat4* in_m2, mat4* out_m); +extern void mat4_sub(const mat4* in_m1, const float_t value, mat4* out_m); +extern void mat4_sub(const mat4* in_m1, const mat4* in_m2, mat4* out_m); +extern void mat4_mul(const mat4* in_m1, const float_t value, mat4* out_m); +extern void mat4_mul(const mat4* in_m1, const mat4* in_m2, mat4* out_m); +extern void mat4_mul_rotation(const mat4* in_m1, const vec3* in_axis, const float_t angle, mat4* out_m); +extern void mat4_transform_vector(const mat4* in_m1, const vec2* normal, vec2* normalOut); +extern void mat4_transform_vector(const mat4* in_m1, const vec3* normal, vec3* normalOut); +extern void mat4_transform_vector(const mat4* in_m1, const vec4* normal, vec4* normalOut); +extern void mat4_transform_point(const mat4* in_m1, const vec2* point, vec2* pointOut); +extern void mat4_transform_point(const mat4* in_m1, const vec3* point, vec3* pointOut); +extern void mat4_inverse_transform_vector(const mat4* in_m1, const vec2* normal, vec2* normalOut); +extern void mat4_inverse_transform_vector(const mat4* in_m1, const vec3* normal, vec3* normalOut); +extern void mat4_inverse_transform_vector(const mat4* in_m1, const vec4* normal, vec4* normalOut); +extern void mat4_inverse_transform_point(const mat4* in_m1, const vec2* point, vec2* pointOut); +extern void mat4_inverse_transform_point(const mat4* in_m1, const vec3* point, vec3* pointOut); +extern void mat4_transpose(const mat4* in_m1, mat4* out_m); +extern void mat4_invert(const mat4* in_m1, mat4* out_m); +extern void mat4_invert_rotation(const mat4* in_m1, mat4* out_m); +extern void mat4_invert_fast(const mat4* in_m1, mat4* out_m); +extern void mat4_invert_rotation_fast(const mat4* in_m1, mat4* out_m); +extern void mat4_normalize(const mat4* in_m1, mat4* out_m); +extern void mat4_normalize_rotation(const mat4* in_m1, mat4* out_m); +extern float_t mat4_determinant(const mat4* in_m1); +extern void mat4_rotate_x(float_t rad, mat4* out_m); +extern void mat4_rotate_y(float_t rad, mat4* out_m); +extern void mat4_rotate_z(float_t rad, mat4* out_m); +extern void mat4_rotate_x(float_t s, float_t c, mat4* out_m); +extern void mat4_rotate_y(float_t s, float_t c, mat4* out_m); +extern void mat4_rotate_z(float_t s, float_t c, mat4* out_m); +extern void mat4_rotate_xyz(float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m); +extern void mat4_rotate_xzy(float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m); +extern void mat4_rotate_yxz(float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m); +extern void mat4_rotate_yzx(float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m); +extern void mat4_rotate_zxy(float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m); +extern void mat4_rotate_zyx(float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m); +extern void mat4_mul_rotate_x(const mat4* in_m1, float_t rad, mat4* out_m); +extern void mat4_mul_rotate_y(const mat4* in_m1, float_t rad, mat4* out_m); +extern void mat4_mul_rotate_z(const mat4* in_m1, float_t rad, mat4* out_m); +extern void mat4_mul_rotate_x(const mat4* in_m1, float_t s, float_t c, mat4* out_m); +extern void mat4_mul_rotate_y(const mat4* in_m1, float_t s, float_t c, mat4* out_m); +extern void mat4_mul_rotate_z(const mat4* in_m1, float_t s, float_t c, mat4* out_m); +extern void mat4_mul_rotate_xyz(const mat4* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m); +extern void mat4_mul_rotate_xzy(const mat4* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m); +extern void mat4_mul_rotate_yxz(const mat4* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m); +extern void mat4_mul_rotate_yzx(const mat4* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m); +extern void mat4_mul_rotate_zxy(const mat4* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m); +extern void mat4_mul_rotate_zyx(const mat4* in_m1, float_t rad_x, float_t rad_y, float_t rad_z, mat4* out_m); +extern void mat4_scale(float_t sx, float_t sy, float_t sz, mat4* out_m); +extern void mat4_scale_x(float_t s, mat4* out_m); +extern void mat4_scale_y(float_t s, mat4* out_m); +extern void mat4_scale_z(float_t s, mat4* out_m); +extern void mat4_mul_scale(const mat4* in_m1, float_t sx, float_t sy, float_t sz, float_t sw, mat4* out_m); +extern void mat4_mul_scale_x(const mat4* in_m1, float_t s, mat4* out_m); +extern void mat4_mul_scale_y(const mat4* in_m1, float_t s, mat4* out_m); +extern void mat4_mul_scale_z(const mat4* in_m1, float_t s, mat4* out_m); +extern void mat4_scale_w_mult(const mat4* in_m1, float_t s, mat4* out_m); +extern void mat4_scale_rot(const mat4* in_m1, float_t sx, float_t sy, float_t sz, mat4* out_m); +extern void mat4_scale_x_rot(const mat4* in_m1, float_t s, mat4* out_m); +extern void mat4_scale_y_rot(const mat4* in_m1, float_t s, mat4* out_m); +extern void mat4_scale_z_rot(const mat4* in_m1, float_t s, mat4* out_m); +extern void mat4_translate(float_t tx, float_t ty, float_t tz, mat4* out_m); +extern void mat4_translate_x(float_t t, mat4* out_m); +extern void mat4_translate_y(float_t t, mat4* out_m); +extern void mat4_translate_z(float_t t, mat4* out_m); +extern void mat4_mul_translate(const mat4* in_m1, float_t tx, float_t ty, float_t tz, mat4* out_m); +extern void mat4_mul_translate_x(const mat4* in_m1, float_t t, mat4* out_m); +extern void mat4_mul_translate_y(const mat4* in_m1, float_t t, mat4* out_m); +extern void mat4_mul_translate_z(const mat4* in_m1, float_t t, mat4* out_m); +extern void mat4_add_translate(const mat4* in_m1, float_t tx, float_t ty, float_t tz, mat4* out_m); +extern void mat4_add_translate_x(const mat4* in_m1, float_t t, mat4* out_m); +extern void mat4_add_translate_y(const mat4* in_m1, float_t t, mat4* out_m); +extern void mat4_add_translate_z(const mat4* in_m1, float_t t, mat4* out_m); +extern void mat4_to_mat3(const mat4* in_m1, mat3* out_m); +extern void mat4_to_mat3_inverse(const mat4* in_m1, mat3* out_m); +extern void mat4_from_mat3(const mat3* in_m1, mat4* out_m); +extern void mat4_from_mat3_inverse(const mat3* in_m1, mat4* out_m); +extern void mat4_clear_rot(const mat4* in_m1, mat4* out_m); +extern void mat4_clear_trans(const mat4* in_m1, mat4* out_m); +extern void mat4_get_scale(const mat4* in_m1, vec3* out_s); +extern void mat4_get_rotation(const mat4* in_m1, vec3* out_rad); +extern void mat4_get_translation(const mat4* in_m1, vec3* out_t); +extern void mat4_set_translation(mat4* in_m1, const vec3* in_t); +extern float_t mat4_get_max_scale(const mat4* in_m1); +extern void mat4_blend(const mat4* in_m1, const mat4* in_m2, mat4* out_m, float_t blend); +extern void mat4_blend_rotation(const mat4* in_m1, const mat4* in_m2, mat4* out_m, float_t blend); +extern void mat4_lerp_rotation(const mat4* in_m1, const mat4* in_m2, mat4* out_m, float_t blend); +extern void mat4_frustrum(double_t left, double_t right, + double_t bottom, double_t top, double_t z_near, double_t z_far, mat4* out_m); +extern void mat4_ortho(double_t left, double_t right, + double_t bottom, double_t top, double_t z_near, double_t z_far, mat4* out_m); +extern void mat4_persp(double_t fov_y, double_t aspect, double_t z_near, double_t z_far, mat4* out_m); +extern void mat4_look_at(const vec3* eye, const vec3* target, const vec3* up, mat4* out_m); +extern void mat4_look_at(const vec3* eye, const vec3* target, mat4* out_m); + +inline void mat3_rotate_xyz(const vec3* rad, mat3* out_m) { + mat3_rotate_xyz(rad->x, rad->y, rad->z, out_m); +} + +inline void mat3_rotate_xzy(const vec3* rad, mat3* out_m) { + mat3_rotate_xzy(rad->x, rad->y, rad->z, out_m); +} + +inline void mat3_rotate_yxz(const vec3* rad, mat3* out_m) { + mat3_rotate_yxz(rad->x, rad->y, rad->z, out_m); +} + +inline void mat3_rotate_yzx(const vec3* rad, mat3* out_m) { + mat3_rotate_yzx(rad->x, rad->y, rad->z, out_m); +} + +inline void mat3_rotate_zxy(const vec3* rad, mat3* out_m) { + mat3_rotate_zxy(rad->x, rad->y, rad->z, out_m); +} + +inline void mat3_rotate_zyx(const vec3* rad, mat3* out_m) { + mat3_rotate_zyx(rad->x, rad->y, rad->z, out_m); +} + +inline void mat3_mul_rotate_xyz(const mat3* in_m1, const vec3* rad, mat3* out_m) { + mat3_mul_rotate_xyz(in_m1, rad->x, rad->y, rad->z, out_m); +} + +inline void mat3_mul_rotate_xzy(const mat3* in_m1, const vec3* rad, mat3* out_m) { + mat3_mul_rotate_xzy(in_m1, rad->x, rad->y, rad->z, out_m); +} + +inline void mat3_mul_rotate_yxz(const mat3* in_m1, const vec3* rad, mat3* out_m) { + mat3_mul_rotate_yxz(in_m1, rad->x, rad->y, rad->z, out_m); +} + +inline void mat3_mul_rotate_yzx(const mat3* in_m1, const vec3* rad, mat3* out_m) { + mat3_mul_rotate_yzx(in_m1, rad->x, rad->y, rad->z, out_m); +} + +inline void mat3_mul_rotate_zxy(const mat3* in_m1, const vec3* rad, mat3* out_m) { + mat3_mul_rotate_zxy(in_m1, rad->x, rad->y, rad->z, out_m); +} + +inline void mat3_mul_rotate_zyx(const mat3* in_m1, const vec3* rad, mat3* out_m) { + mat3_mul_rotate_zyx(in_m1, rad->x, rad->y, rad->z, out_m); +} + +inline void mat3_scale(const vec3* s, mat3* out_m) { + mat3_scale(s->x, s->y, s->z, out_m); +} + +inline void mat3_mul_scale(const mat3* in_m1, float_t s, mat3* out_m) { + mat3_mul_scale(in_m1, s, s, s, out_m); +} + +inline void mat3_mul_scale(const mat3* in_m1, const vec3* s, mat3* out_m) { + mat3_mul_scale(in_m1, s->x, s->y, s->z, out_m); +} + +inline void mat4_rotate_xyz(const vec3* rad, mat4* out_m) { + mat4_rotate_xyz(rad->x, rad->y, rad->z, out_m); +} + +inline void mat4_rotate_xzy(const vec3* rad, mat4* out_m) { + mat4_rotate_xzy(rad->x, rad->y, rad->z, out_m); +} + +inline void mat4_rotate_yxz(const vec3* rad, mat4* out_m) { + mat4_rotate_yxz(rad->x, rad->y, rad->z, out_m); +} + +inline void mat4_rotate_yzx(const vec3* rad, mat4* out_m) { + mat4_rotate_yzx(rad->x, rad->y, rad->z, out_m); +} + +inline void mat4_rotate_zxy(const vec3* rad, mat4* out_m) { + mat4_rotate_zxy(rad->x, rad->y, rad->z, out_m); +} + +inline void mat4_rotate_zyx(const vec3* rad, mat4* out_m) { + mat4_rotate_zyx(rad->x, rad->y, rad->z, out_m); +} + +inline void mat4_mul_rotate_xyz(const mat4* in_m1, const vec3* rad, mat4* out_m) { + mat4_mul_rotate_xyz(in_m1, rad->x, rad->y, rad->z, out_m); +} + +inline void mat4_mul_rotate_xzy(const mat4* in_m1, const vec3* rad, mat4* out_m) { + mat4_mul_rotate_xzy(in_m1, rad->x, rad->y, rad->z, out_m); +} + +inline void mat4_mul_rotate_yxz(const mat4* in_m1, const vec3* rad, mat4* out_m) { + mat4_mul_rotate_yxz(in_m1, rad->x, rad->y, rad->z, out_m); +} + +inline void mat4_mul_rotate_yzx(const mat4* in_m1, const vec3* rad, mat4* out_m) { + mat4_mul_rotate_yzx(in_m1, rad->x, rad->y, rad->z, out_m); +} + +inline void mat4_mul_rotate_zxy(const mat4* in_m1, const vec3* rad, mat4* out_m) { + mat4_mul_rotate_zxy(in_m1, rad->x, rad->y, rad->z, out_m); +} + +inline void mat4_mul_rotate_zyx(const mat4* in_m1, const vec3* rad, mat4* out_m) { + mat4_mul_rotate_zyx(in_m1, rad->x, rad->y, rad->z, out_m); +} + +inline void mat4_scale(const vec3* s, mat4* out_m) { + mat4_scale(s->x, s->y, s->z, out_m); +} + +inline void mat4_mul_scale(const mat4* in_m1, float_t s, mat4* out_m) { + mat4_mul_scale(in_m1, s, s, s, s, out_m); +} + +inline void mat4_mul_scale(const mat4* in_m1, vec4* s, mat4* out_m) { + mat4_mul_scale(in_m1, s->x, s->y, s->z, s->w, out_m); +} + +inline void mat4_scale_rot(const mat4* in_m1, const float_t s, mat4* out_m) { + mat4_scale_rot(in_m1, s, s, s, out_m); +} + +inline void mat4_scale_rot(const mat4* in_m1, const vec3* s, mat4* out_m) { + mat4_scale_rot(in_m1, s->x, s->y, s->z, out_m); +} + +inline void mat4_translate(const vec3* s, mat4* out_m) { + mat4_translate(s->x, s->y, s->z, out_m); +} + +inline void mat4_mul_translate(const mat4* in_m1, const vec3* t, mat4* out_m) { + mat4_mul_translate(in_m1, t->x, t->y, t->z, out_m); +} + +inline void mat4_add_translate(const mat4* in_m1, const vec3* t, mat4* out_m) { + mat4_add_translate(in_m1, t->x, t->y, t->z, out_m); +} diff --git a/src/KKdLib/msgpack.cpp b/src/KKdLib/msgpack.cpp new file mode 100644 index 0000000..2367fe1 --- /dev/null +++ b/src/KKdLib/msgpack.cpp @@ -0,0 +1,540 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "msgpack.hpp" + +msgpack* msgpack::get_by_index(size_t index) { + if (type != MSGPACK_ARRAY) + return 0; + + msgpack_array* ptr = data.arr; + if (index < ptr->size()) + return &ptr->data()[index]; + return 0; +} + +void msgpack::set_by_index(msgpack* m, size_t index) { + if (type != MSGPACK_ARRAY) + return; + + msgpack_array* ptr = data.arr; + if (index < ptr->size()) { + msgpack& msg = ptr->data()[index]; + msg = *m; + } +} + +msgpack* msgpack::get_by_name(const char* name) { + if (type != MSGPACK_MAP) + return 0; + + msgpack_map* ptr = data.map; + for (auto& i : *ptr) + if (!i.first.compare(name)) + return &i.second; + + return 0; +} + +void msgpack::set_by_name(const char* name, msgpack* m) { + if (type != MSGPACK_MAP) + return; + + msgpack_map* ptr = data.map; + for (auto& i : *ptr) + if (!i.first.compare(name)) { + i.second = *m; + return; + } + + append(name, m); +} + +msgpack* msgpack::append(const char* name, msgpack* m) { + if (type != MSGPACK_MAP) + return 0; + + msgpack* tm = get_by_name(name); + if (tm) { + *tm = *m; + m->clear(); + return tm; + } + else { + data.map->push_back(name, *m); + m->clear(); + return &data.map->back().second; + } +} + +msgpack* msgpack::append(const char* name, msgpack& m) { + if (type != MSGPACK_MAP) + return 0; + + msgpack* tm = get_by_name(name); + if (tm) { + *tm = m; + m.clear(); + return tm; + } + else { + data.map->push_back(name, m); + m.clear(); + return &data.map->back().second; + } +} + +msgpack* msgpack::append(const char* name, msgpack&& m) { + if (type != MSGPACK_MAP) + return 0; + + msgpack* tm = get_by_name(name); + if (tm) { + *tm = m; + m.clear(); + return tm; + } + else { + data.map->push_back(name, m); + m.clear(); + return &data.map->back().second; + } +} + +msgpack* msgpack::read(const char* name) { + if (!this) + return 0; + + return name ? get_by_name(name) : this; +} + +msgpack* msgpack::read(const char* name, msgpack_type type) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (m) { + if (m->type == type) + return m; + else if (m->type >= MSGPACK_INT8 && m->type <= MSGPACK_FLOAT64 && m->type <= type) + return m; + return m; + } + return 0; +} + +msgpack* msgpack::read_array(const char* name) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (m && m->type == MSGPACK_ARRAY) + return m; + return 0; +} + +msgpack* msgpack::read_map(const char* name) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (m && m->type == MSGPACK_MAP) + return m; + return 0; +} + +bool msgpack::read_bool(const char* name) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (m && m->type == MSGPACK_BOOL) + return m->data.b; + return 0; +} + +int8_t msgpack::read_int8_t(const char* name) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (!m) + return 0; + + switch (m->type) { + case MSGPACK_INT8: + return m->data.i8; + case MSGPACK_UINT8: + return (int8_t)m->data.u8; + } + return 0; +} + +uint8_t msgpack::read_uint8_t(const char* name) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (!m) + return 0; + + switch (m->type) { + case MSGPACK_INT8: + return (uint8_t)m->data.i8; + case MSGPACK_UINT8: + return m->data.u8; + } + return 0; +} + +int16_t msgpack::read_int16_t(const char* name) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (!m) + return 0; + + switch (m->type) { + case MSGPACK_INT8: + return m->data.i8; + case MSGPACK_UINT8: + return m->data.u8; + case MSGPACK_INT16: + return m->data.i16; + case MSGPACK_UINT16: + return (int16_t)m->data.u16; + } + return 0; +} + +uint16_t msgpack::read_uint16_t(const char* name) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (!m) + return 0; + + switch (m->type) { + case MSGPACK_INT8: + return (uint16_t)(int16_t)m->data.i8; + case MSGPACK_UINT8: + return m->data.u8; + case MSGPACK_INT16: + return (uint16_t)m->data.i16; + case MSGPACK_UINT16: + return m->data.u16; + } + return 0; +} + +int32_t msgpack::read_int32_t(const char* name) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (!m) + return 0; + + switch (m->type) { + case MSGPACK_INT8: + return m->data.i8; + case MSGPACK_UINT8: + return m->data.u8; + case MSGPACK_INT16: + return m->data.i16; + case MSGPACK_UINT16: + return m->data.u16; + case MSGPACK_INT32: + return m->data.i32; + case MSGPACK_UINT32: + return (int32_t)m->data.u32; + } + return 0; +} + +uint32_t msgpack::read_uint32_t(const char* name) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (!m) + return 0; + + switch (m->type) { + case MSGPACK_INT8: + return (uint32_t)(int32_t)m->data.i8; + case MSGPACK_UINT8: + return m->data.u8; + case MSGPACK_INT16: + return (uint32_t)(int32_t)m->data.i16; + case MSGPACK_UINT16: + return m->data.u16; + case MSGPACK_INT32: + return (uint32_t)(int32_t)m->data.i32; + case MSGPACK_UINT32: + return m->data.u32; + } + return 0; +} + +int64_t msgpack::read_int64_t(const char* name) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (!m) + return 0; + + switch (m->type) { + case MSGPACK_INT8: + return m->data.i8; + case MSGPACK_UINT8: + return m->data.u8; + case MSGPACK_INT16: + return m->data.i16; + case MSGPACK_UINT16: + return m->data.u16; + case MSGPACK_INT32: + return m->data.i32; + case MSGPACK_UINT32: + return m->data.u32; + case MSGPACK_INT64: + return m->data.i64; + case MSGPACK_UINT64: + return (int64_t)m->data.u64; + } + return 0; +} + +uint64_t msgpack::read_uint64_t(const char* name) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (!m) + return 0; + + switch (m->type) { + case MSGPACK_INT8: + return (uint64_t)(int64_t)m->data.i8; + case MSGPACK_UINT8: + return m->data.u8; + case MSGPACK_INT16: + return (uint64_t)(int64_t)m->data.i16; + case MSGPACK_UINT16: + return m->data.u16; + case MSGPACK_INT32: + return (uint64_t)(int64_t)m->data.i32; + case MSGPACK_UINT32: + return m->data.u32; + case MSGPACK_INT64: + return (uint64_t)m->data.i64; + case MSGPACK_UINT64: + return m->data.u64; + } + return 0; +} + +float_t msgpack::read_float_t(const char* name) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (!m) + return 0; + + switch (m->type) { + case MSGPACK_INT8: + return m->data.i8; + case MSGPACK_UINT8: + return m->data.u8; + case MSGPACK_INT16: + return m->data.i16; + case MSGPACK_UINT16: + return m->data.u16; + case MSGPACK_INT32: + return (float_t)m->data.i32; + case MSGPACK_UINT32: + return (float_t)m->data.u32; + case MSGPACK_INT64: + return (float_t)m->data.i64; + case MSGPACK_UINT64: + return (float_t)m->data.u64; + case MSGPACK_FLOAT32: + return m->data.f32; + case MSGPACK_FLOAT64: + return (float_t)m->data.f64; + } + return 0; +} + +double_t msgpack::read_double_t(const char* name) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (!m) + return 0; + + switch (m->type) { + case MSGPACK_INT8: + return m->data.i8; + case MSGPACK_UINT8: + return m->data.u8; + case MSGPACK_INT16: + return m->data.i16; + case MSGPACK_UINT16: + return m->data.u16; + case MSGPACK_INT32: + return m->data.i32; + case MSGPACK_UINT32: + return m->data.u32; + case MSGPACK_INT64: + return (double_t)m->data.i64; + case MSGPACK_UINT64: + return (double_t)m->data.u64; + case MSGPACK_FLOAT32: + return m->data.f32; + case MSGPACK_FLOAT64: + return m->data.f64; + } + return 0; +} + +char* msgpack::read_utf8_string(const char* name) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (m && m->type == MSGPACK_STRING) { + size_t length = m->data.str->size(); + char* val = force_malloc(length + 1); + memcpy(val, m->data.str->c_str(), length); + val[length] = 0; + return val; + } + return 0; +} + +wchar_t* msgpack::read_utf16_string(const char* name) { + if (!this) + return 0; + + msgpack* m = name ? get_by_name(name) : this; + if (m && m->type == MSGPACK_STRING) + return utf8_to_utf16(m->data.str->c_str()); + return 0; +} + +std::string msgpack::read_string(const char* name) { + if (!this) + return {}; + + msgpack* m = name ? get_by_name(name) : this; + if (m && m->type == MSGPACK_STRING) + return *m->data.str; + return {}; +} + +std::wstring msgpack::read_wstring(const char* name) { + if (!this) + return {}; + + msgpack* m = name ? get_by_name(name) : this; + if (m && m->type == MSGPACK_STRING) + return utf8_to_utf16(*m->data.str); + return {}; +} + +msgpack& msgpack::operator=(const msgpack& m) { + switch (m.type) { + case MSGPACK_BOOL: + case MSGPACK_INT8: + case MSGPACK_UINT8: + case MSGPACK_INT16: + case MSGPACK_UINT16: + case MSGPACK_INT32: + case MSGPACK_UINT32: + case MSGPACK_INT64: + case MSGPACK_UINT64: + case MSGPACK_FLOAT32: + case MSGPACK_FLOAT64: + switch (type) { + case MSGPACK_STRING: + delete data.str; + break; + case MSGPACK_ARRAY: + delete data.arr; + break; + case MSGPACK_MAP: + delete data.map; + break; + } + + memcpy(&data, &m.data, sizeof(data)); + break; + case MSGPACK_STRING: + switch (type) { + case MSGPACK_STRING: + break; + case MSGPACK_ARRAY: + delete data.arr; + data.str = new std::string; + break; + case MSGPACK_MAP: + delete data.map; + data.str = new std::string; + break; + default: + data.str = new std::string; + break; + } + data.str->assign(*m.data.str); + break; + case MSGPACK_ARRAY: + switch (type) { + case MSGPACK_STRING: + delete data.str; + data.arr = new msgpack_array; + break; + case MSGPACK_ARRAY: + break; + case MSGPACK_MAP: + delete data.map; + data.arr = new msgpack_array; + break; + default: + data.arr = new msgpack_array; + break; + } + data.arr->assign(m.data.arr->begin(), m.data.arr->end()); + break; + case MSGPACK_MAP: + switch (type) { + case MSGPACK_STRING: + delete data.str; + data.map = new msgpack_map; + break; + case MSGPACK_ARRAY: + delete data.arr; + data.map = new msgpack_map; + break; + case MSGPACK_MAP: + break; + default: + data.map = new msgpack_map; + break; + } + data.map->assign(m.data.map->begin(), m.data.map->end()); + break; + default: + data = {}; + break; + } + type = m.type; + return *this; +} diff --git a/src/KKdLib/msgpack.hpp b/src/KKdLib/msgpack.hpp new file mode 100644 index 0000000..03c1a75 --- /dev/null +++ b/src/KKdLib/msgpack.hpp @@ -0,0 +1,277 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include +#include +#include "default.hpp" +#include "prj/vector_pair.hpp" + +enum msgpack_type : uint32_t { + MSGPACK_NONE = 0, + MSGPACK_NULL, + MSGPACK_BOOL, + MSGPACK_INT8, + MSGPACK_UINT8, + MSGPACK_INT16, + MSGPACK_UINT16, + MSGPACK_INT32, + MSGPACK_UINT32, + MSGPACK_INT64, + MSGPACK_UINT64, + MSGPACK_FLOAT32, + MSGPACK_FLOAT64, + MSGPACK_STRING, + MSGPACK_ARRAY, + MSGPACK_MAP, +}; + +struct msgpack; + +typedef std::vector msgpack_array; +typedef prj::vector_pair msgpack_map; + +union msgpack_data { + bool b; + int8_t i8; + uint8_t u8; + int16_t i16; + uint16_t u16; + int32_t i32; + uint32_t u32; + int64_t i64; + uint64_t u64; + float_t f32; + double_t f64; + std::string* str; + msgpack_array* arr; + msgpack_map* map; +}; + +struct msgpack { + msgpack_type type; + msgpack_data data; + + msgpack* get_by_index(size_t index); + void set_by_index(msgpack* m, size_t index); + msgpack* get_by_name(const char* name); + void set_by_name(const char* name, msgpack* m); + msgpack* append(const char* name, msgpack* m); + msgpack* append(const char* name, msgpack& m); + msgpack* append(const char* name, msgpack&& m); + msgpack* read(const char* name); + msgpack* read(const char* name, msgpack_type type); + msgpack* read_array(const char* name = 0); + msgpack* read_map(const char* name = 0); + bool read_bool(const char* name = 0); + int8_t read_int8_t(const char* name = 0); + uint8_t read_uint8_t(const char* name = 0); + int16_t read_int16_t(const char* name = 0); + uint16_t read_uint16_t(const char* name = 0); + int32_t read_int32_t(const char* name = 0); + uint32_t read_uint32_t(const char* name = 0); + int64_t read_int64_t(const char* name = 0); + uint64_t read_uint64_t(const char* name = 0); + float_t read_float_t(const char* name = 0); + double_t read_double_t(const char* name = 0); + char* read_utf8_string(const char* name = 0); + wchar_t* read_utf16_string(const char* name = 0); + std::string read_string(const char* name = 0); + std::wstring read_wstring(const char* name = 0); + + msgpack& operator=(const msgpack& m); + + inline msgpack() : data() { + type = MSGPACK_NULL; + } + + inline msgpack(const msgpack& m) : type(), data() { + *this = m; + } + + inline msgpack(msgpack_array& val) : data() { + type = MSGPACK_ARRAY; + data.arr = new msgpack_array; + data.arr->assign(val.begin(), val.end()); + val.clear(); + val.shrink_to_fit(); + } + + inline msgpack(msgpack_array&& val) : data() { + type = MSGPACK_ARRAY; + data.arr = new msgpack_array; + data.arr->assign(val.begin(), val.end()); + val.clear(); + val.shrink_to_fit(); + } + + inline msgpack(msgpack_map& val) : data() { + type = MSGPACK_MAP; + data.map = new msgpack_map; + data.map->assign(val.begin(), val.end()); + val.clear(); + val.shrink_to_fit(); + } + inline msgpack(msgpack_map&& val) : data() { + type = MSGPACK_MAP; + data.map = new msgpack_map; + data.map->assign(val.begin(), val.end()); + val.clear(); + val.shrink_to_fit(); + } + + inline msgpack(bool val) : data() { + type = MSGPACK_BOOL; + data.b = val; + } + + inline msgpack(int8_t val) : data() { + type = MSGPACK_INT8; + data.i64 = val; + } + + inline msgpack(uint8_t val) : data() { + type = MSGPACK_UINT8; + data.u64 = val; + } + + inline msgpack(int16_t val) : data() { + type = MSGPACK_INT16; + data.i64 = val; + } + + inline msgpack(uint16_t val) : data() { + type = MSGPACK_UINT16; + data.u64 = val; + } + + inline msgpack(int32_t val) : data() { + type = MSGPACK_INT32; + data.i64 = val; + } + + inline msgpack(uint32_t val) : data() { + type = MSGPACK_UINT32; + data.u64 = val; + } + + inline msgpack(int64_t val) : data() { + type = MSGPACK_INT64; + data.i64 = val; + } + + inline msgpack(uint64_t val) : data() { + type = MSGPACK_UINT64; + data.u64 = val; + } + + inline msgpack(float_t val) : data() { + type = MSGPACK_FLOAT32; + data.f32 = val; + } + + inline msgpack(double_t val) : data() { + type = MSGPACK_FLOAT64; + data.f64 = val; + } + + inline msgpack(const char* val) : data() { + type = MSGPACK_STRING; + data.str = new std::string; + if (val) + data.str->assign(val); + } + + inline msgpack(const wchar_t* val) : data() { + type = MSGPACK_STRING; + data.str = new std::string; + if (val) { + char* temp = utf16_to_utf8(val); + data.str->assign(temp); + free_def(temp); + } + } + + inline msgpack(std::string& val) : data() { + type = MSGPACK_STRING; + data.str = new std::string; + if (val.size()) + data.str->assign(val); + } + + inline msgpack(std::string&& val) : data() { + type = MSGPACK_STRING; + data.str = new std::string; + if (val.size()) + data.str->assign(val); + } + + inline msgpack(std::wstring& val) : data() { + type = MSGPACK_STRING; + data.str = new std::string; + if (val.size()) { + char* temp = utf16_to_utf8(val.c_str()); + data.str->assign(temp); + free_def(temp); + } + } + + inline msgpack(std::wstring&& val) : data() { + type = MSGPACK_STRING; + data.str = new std::string; + if (val.size()) { + char* temp = utf16_to_utf8(val.c_str()); + data.str->assign(temp); + free_def(temp); + } + } + + inline ~msgpack() { + switch (type) { + case MSGPACK_STRING: + delete data.str; + break; + case MSGPACK_ARRAY: + delete data.arr; + break; + case MSGPACK_MAP: + delete data.map; + break; + } + } + + inline bool check_null() { + if (type == MSGPACK_ARRAY) + return !data.arr->size(); + else if (type == MSGPACK_MAP) + return !data.map->size(); + return type == MSGPACK_NONE; + } + + inline bool check_not_null() { + if (type == MSGPACK_ARRAY) + return !!data.arr->size(); + else if (type == MSGPACK_MAP) + return !!data.map->size(); + return type != MSGPACK_NONE; + } + + inline void clear() { + switch (type) { + case MSGPACK_STRING: + delete data.str; + break; + case MSGPACK_ARRAY: + delete data.arr; + break; + case MSGPACK_MAP: + delete data.map; + break; + } + type = MSGPACK_NONE; + data = {}; + } +}; diff --git a/src/KKdLib/prj/algorithm.hpp b/src/KKdLib/prj/algorithm.hpp new file mode 100644 index 0000000..d4fcde2 --- /dev/null +++ b/src/KKdLib/prj/algorithm.hpp @@ -0,0 +1,56 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder + + Taken from MSVC's VC/include/memory +*/ + +#pragma once + +#include "../default.hpp" +#include +#include + +namespace prj { + template + bool find(std::vector& vec, T& value) { + auto begin = vec.begin(); + auto end = vec.end(); + for (auto i = begin; i != end; i++) + if (*i == value) + return true; + return false; + } + + template + void sort(std::vector& vec) { + std::sort(vec.begin(), vec.end()); + } + + template + void unique(std::vector& vec) { + if (vec.size() <= 1) + return; + + auto begin = vec.begin(); + auto end = vec.end(); + for (auto i = begin, j = begin + 1; i != end && j != end; ) + if (*i == *j) { + std::move(j + 1, end, j); + end--; + } + else { + i++; + j++; + } + + if (vec.size() != end - begin) + vec.resize(end - begin); + } + + template + void sort_unique(std::vector& vec) { + sort(vec); + unique(vec); + } +} diff --git a/src/KKdLib/prj/math.hpp b/src/KKdLib/prj/math.hpp new file mode 100644 index 0000000..b16bc44 --- /dev/null +++ b/src/KKdLib/prj/math.hpp @@ -0,0 +1,43 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "../default.hpp" +#include + +namespace prj { + inline int32_t extract_sign(float_t x) { + return _mm_movemask_ps(_mm_set_ss(x)) & 0x01; + } + + inline float_t ceilf(float_t x) { + int32_t x_int = (int32_t)x; + if (x_int != 0x80000000 && (float_t)x_int != x) + x = (float_t)(x_int + !extract_sign(x)); + return x; + } + + inline float_t floorf(float_t x) { + int32_t x_int = (int32_t)x; + if (x_int != 0x80000000 && (float_t)x_int != x) + x = (float_t)(x_int - extract_sign(x)); + return x; + } + + inline float_t roundf(float_t x) { + if (x >= 0.0f) + return floorf(x + 0.5f); + else + return ceilf(x - 0.5f); + } + + inline float_t truncf(float_t x) { + if (x >= 0.0f) + return floorf(x); + else + return ceilf(x); + } +} diff --git a/src/KKdLib/prj/shared_ptr.hpp b/src/KKdLib/prj/shared_ptr.hpp new file mode 100644 index 0000000..ca3b6e1 --- /dev/null +++ b/src/KKdLib/prj/shared_ptr.hpp @@ -0,0 +1,284 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder + + Taken from MSVC's VC/include/memory +*/ + +#pragma once + +#include "../default.hpp" + +namespace prj { + template + class ref_count { + private: + uint32_t uses; + uint32_t weaks; + T* ptr; + void(*delete_this_func)(ref_count* ref); + void(*destroy_func)(ref_count* ref, T* ptr); + + protected: + ref_count() { + uses = 1; + weaks = 1; + ptr = nullptr; + delete_this_func = delete_this; + destroy_func = destroy; + } + + public: + ~ref_count() { + + } + + ref_count(T* ptr) { + uses = 1; + weaks = 1; + this->ptr = ptr; + delete_this_func = delete_this; + destroy_func = destroy; + } + + bool expired() const { + return !use_count(); + } + + void decrement() { + if (!--uses) { + destroy_func(this, ptr); + decrement_weaks(); + } + } + + void decrement_weaks() { + if (!--weaks) + delete_this_func(this); + } + + uint32_t use_count() const { + return uses; + } + + void increment() { + uses++; + } + + void increment_weaks() { + weaks++; + } + + private: + static void delete_this(ref_count* ref) { + delete ref; + } + + static void destroy(ref_count* ref, T* ptr) { + delete ptr; + } + }; + + template + class shared_ptr; + + template + class ptr_base { + public: + typedef ptr_base my_t; + + ptr_base() : ptr(0), ref(0) { + + } + + ptr_base(my_t&& right) noexcept : ptr(0), ref(0) { + assign(std::forward(right)); + } + + template + ptr_base(ptr_base&& right) + : ptr(right.ptr), ref(right.ref) { + right.ptr = 0; + right.ref = 0; + } + + my_t& operator=(my_t&& right) { + assign(std::forward(right)); + return *this; + } + + void assign(my_t&& right) { + swap(right); + } + + uint32_t use_count() const { + return ref ? ref->use_count() : 0; + } + + void swap(ptr_base& right) { + std::swap(ref, right.ref); + std::swap(ptr, right.ptr); + } + + template + bool owner_before(const ptr_base& right) const { + return ref < right.ref; + } + + T* get() const { + return ptr; + } + + bool expired() const { + return !ref || ref->expired(); + } + + void decrement() { + if (ref) + ref->decrement(); + } + + void reset() { + reset(0, 0); + } + + template + void reset(const ptr_base& other) { + reset(other.ptr, other.ref); + } + + template + void reset(T* ptr, const ptr_base& other) { + reset(ptr, other.ref); + } + + void reset(T* other_ptr, ref_count* other_ref) { + if (other_ref) + other_ref->increment(); + reset_base(other_ptr, other_ref); + } + + void reset_base(T* other_ptr, ref_count* other_ref) { + if (ref) + ref->decrement(); + ref = other_ref; + ptr = other_ptr; + } + + void decrement_weaks() { + if (ref) + ref->decrement_weaks(); + } + + private: + T* ptr; + ref_count* ref; + template + friend class ptr_base; + }; + + template + class shared_ptr : public ptr_base { + public: + typedef shared_ptr my_t; + typedef ptr_base my_base; + + shared_ptr() { + + } + + template + explicit shared_ptr(U* ptr) { + reset_ptr(ptr); + } + + template + shared_ptr(nullptr_t) { + reset_ptr((T*)0); + } + + shared_ptr(const my_t& other) { + ptr_base::reset(other); + } + + shared_ptr(my_t&& right) noexcept : my_base(std::forward(right)) { + + } + + my_t& operator=(my_t&& right) noexcept { + shared_ptr(std::move(right)).swap(*this); + return *this; + } + + template + my_t& operator=(shared_ptr&& right) { + shared_ptr(std::move(right)).swap(*this); + return *this; + } + + ~shared_ptr() { + this->decrement(); + } + + my_t& operator=(const my_t& right) { + shared_ptr(right).swap(*this); + return *this; + } + + template + my_t& operator=(const shared_ptr& right) { + shared_ptr(right).swap(*this); + return *this; + } + + void reset() { + shared_ptr().swap(*this); + } + + template + void reset(U* ptr) { + shared_ptr(ptr).swap(*this); + } + + T* operator->() const { + return this->get(); + } + + bool unique() const { + return this->use_count() == 1; + } + + operator bool() const { + return !!this->get(); + } + + private: + template + void reset_ptr(U* ptr) { + try { + this->reset_base(ptr, new ref_count(ptr)); + } + catch (...) { + delete ptr; + throw; + } + } + }; + + template + bool operator==(const shared_ptr& left, + const shared_ptr& right) { + return left.get() == right.get(); + } + + template + bool operator!=(const shared_ptr& left, + const shared_ptr& right) { + return !(left == right); + } + + template + void swap(shared_ptr& left, shared_ptr& right) { + left.swap(right); + } +} diff --git a/src/KKdLib/prj/stack_allocator.cpp b/src/KKdLib/prj/stack_allocator.cpp new file mode 100644 index 0000000..764f29c --- /dev/null +++ b/src/KKdLib/prj/stack_allocator.cpp @@ -0,0 +1,104 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "stack_allocator.hpp" + +namespace prj { + void* stack_allocator::allocate(size_t size) { +#if PRJ_STACK_ALLOCATOR_ORIGINAL_CODE + stack_allocator_node* node = (stack_allocator_node*)begin; +#else + stack_allocator_node* node = next; +#endif + size = max_def(size, 1); + size = align_val(size, 8); + +#if PRJ_STACK_ALLOCATOR_ORIGINAL_CODE + bool allocate = !node || (capacity_end - end) < (ssize_t)size; +#else + bool allocate = true; + if (node) { + stack_allocator_node* last_node = 0; + while (node) { + if ((size_t)(node->capacity - node->size) >= size) + last_node = node; + node = node->next; + } + + if (last_node) { + node = last_node; + allocate = false; + } + else + node = next; + } +#endif + + if (allocate) { +#if PRJ_STACK_ALLOCATOR_ORIGINAL_CODE + size_t data_size = this->size; +#else + size_t data_size = node ? (node->capacity + sizeof(stack_allocator_node)) * 2 : this->size; +#endif + size_t _size = size + sizeof(stack_allocator_node); + size_t mults = 0; + while (data_size < _size) { + data_size *= 2; + if (data_size >= _size) + break; + + mults++; + data_size *= 2; + + if (mults >= 16) { + if (data_size < _size) + return 0; + break; + } + } + + stack_allocator_node* new_node = (stack_allocator_node*)malloc(data_size); + if (!new_node) + return 0; + +#if PRJ_STACK_ALLOCATOR_ORIGINAL_CODE + new_node->next = node; + begin = (uint8_t*)new_node; + end = (uint8_t*)new_node + sizeof(stack_allocator_node); + capacity_end = (uint8_t*)new_node + data_size; +#else + new_node->next = node; + new_node->size = 0; + new_node->capacity = data_size - sizeof(stack_allocator_node); + next = new_node; +#endif + node = new_node; + } + +#if PRJ_STACK_ALLOCATOR_ORIGINAL_CODE + uint8_t* data = end; + end += size; +#else + uint8_t* data = node->data + node->size; + node->size += size; +#endif + memset(data, 0, size); + return data; + } + + void stack_allocator::deallocate() { +#if PRJ_STACK_ALLOCATOR_ORIGINAL_CODE + stack_allocator_node* node = (stack_allocator_node*)begin; +#else + stack_allocator_node* node = next; +#endif + while (node) { + stack_allocator_node* next_node = node->next; + free(node); + node = next_node; + } + next = 0; + } +} \ No newline at end of file diff --git a/src/KKdLib/prj/stack_allocator.hpp b/src/KKdLib/prj/stack_allocator.hpp new file mode 100644 index 0000000..7be09f9 --- /dev/null +++ b/src/KKdLib/prj/stack_allocator.hpp @@ -0,0 +1,85 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "../default.hpp" + +#define PRJ_STACK_ALLOCATOR_ORIGINAL_CODE 0 + +namespace prj { + struct stack_allocator_node { + stack_allocator_node * next; +#if !PRJ_STACK_ALLOCATOR_ORIGINAL_CODE + size_t size; + size_t capacity; +#endif +#pragma warning(suppress: 4200) + uint8_t data[]; + }; + + struct stack_allocator { + size_t size; +#if PRJ_STACK_ALLOCATOR_ORIGINAL_CODE + uint8_t* begin; + uint8_t* end; + uint8_t* capacity_end; +#else + stack_allocator_node* next; +#endif + +#if PRJ_STACK_ALLOCATOR_ORIGINAL_CODE + inline stack_allocator() : begin(), end(), capacity_end() { + size = 4000; + } +#else + inline stack_allocator() : next() { + size = 4000; + } +#endif + + inline ~stack_allocator() { + deallocate(); + } + + void* allocate(size_t size); + void deallocate(); + + template + inline T* allocate() { + return new((T*)allocate(sizeof(T))) T; + } + + template + inline T* allocate(size_t size) { + if (!size) + return 0; + + T* arr = (T*)allocate(sizeof(T) * size); + for (size_t i = 0; i < size; i++) + new(&arr[i]) T(); + return arr; + } + + template + inline T* allocate(const T* src) { + if (!src) + return 0; + + return new((T*)allocate(sizeof(T))) T(*src); + } + + template + inline T* allocate(const T* src, size_t size) { + if (!src || !size) + return 0; + + T* dst = (T*)allocate(sizeof(T) * size); + for (size_t i = 0; i < size; i++) + new(&dst[i]) T(src[i]); + return dst; + } + }; +} diff --git a/src/KKdLib/prj/time.cpp b/src/KKdLib/prj/time.cpp new file mode 100644 index 0000000..3c86edd --- /dev/null +++ b/src/KKdLib/prj/time.cpp @@ -0,0 +1,39 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "time.hpp" +#include + +namespace prj { + prj::time prj::time::get_default() { + static prj::time start_of_2005 = prj::strptime("2005-01-01 00:00:00"); + return start_of_2005; + } + + prj::time strptime(std::string& str) { + int32_t year; + int32_t month; + int32_t day; + int32_t hour; + int32_t min; + int32_t sec; + if (sscanf_s(str.c_str(), "%4d-%2d-%2d %2d:%2d:%2d", &year, &month, &day, &hour, &min, &sec) == 6) { + struct tm time; + time.tm_isdst = -1; + time.tm_year = year - 1900; + time.tm_mon = month - 1; + time.tm_mday = day; + time.tm_hour = hour; + time.tm_min = min; + time.tm_sec = sec; + return prj::time(_mkgmtime(&time)); + } + return {}; + } + + prj::time strptime(std::string&& str) { + return strptime(str); + } +} diff --git a/src/KKdLib/prj/time.hpp b/src/KKdLib/prj/time.hpp new file mode 100644 index 0000000..7d71995 --- /dev/null +++ b/src/KKdLib/prj/time.hpp @@ -0,0 +1,27 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "../default.hpp" + +namespace prj { + struct time { + time_t value; + + inline time() { + value = -1; + } + + inline time(time_t value) { + this->value = value; + } + + static prj::time get_default(); + }; + + prj::time strptime(std::string& str); + prj::time strptime(std::string&& str); +} diff --git a/src/KKdLib/prj/vector_pair.hpp b/src/KKdLib/prj/vector_pair.hpp new file mode 100644 index 0000000..8aebc47 --- /dev/null +++ b/src/KKdLib/prj/vector_pair.hpp @@ -0,0 +1,144 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder + + Taken from MSVC's VC/include/memory +*/ + +#pragma once + +#include "../default.hpp" +#include +#include + +namespace prj { + template + class vector_pair : public std::vector> { + public: + using value_pair = std::pair; + + inline void push_back(const T& first, const U& second) { + push_back({ first, second }); + } + + inline void push_back(const T& first, U&& second) { + push_back({ first, second }); + } + + inline void push_back(T&& first, const U& second) { + push_back({ first, second }); + } + + inline void push_back(T&& first, U&& second) { + push_back({ first, second }); + } + + inline void push_back(const value_pair& value) { + std::vector>::push_back(value); + } + + inline void push_back(value_pair&& value) { + std::vector>::push_back(value); + } + + inline void sort() { + std::sort(this->begin(), this->end(), + [](const std::pair& a, const std::pair& b) { + return a.first < b.first; + }); + } + + inline void unique() { + if (this->size() <= 1) + return; + + auto begin = this->begin(); + auto end = this->end(); + for (auto i = begin, j = begin + 1; i != end && j != end; ) + if (i->first == j->first) { + std::move(j + 1, end, j); + end--; + } + else { + i++; + j++; + } + + if (this->size() != end - begin) + this->resize(end - begin); + } + + inline void sort_unique() { + sort(); + unique(); + } + + inline typename auto find(const T& key) { + auto k = this->begin(); + size_t l = this->size(); + size_t temp; + while (l > 0) { + if (k[temp = l / 2].first >= key) + l /= 2; + else { + k += temp + 1; + l -= temp + 1; + } + } + if (k == this->end() || key < k->first) + return this->end(); + return k; + } + + inline typename auto find(const T& key) const { + auto k = this->begin(); + size_t l = this->size(); + size_t temp; + while (l > 0) { + if (k[temp = l / 2].first >= key) + l /= 2; + else { + k += temp + 1; + l -= temp + 1; + } + } + if (k == this->end() || key < k->first) + return this->end(); + return k; + } + + inline typename auto find(T&& key) { + auto k = this->begin(); + size_t l = this->size(); + size_t temp; + while (l > 0) { + if (k[temp = l / 2].first >= key) + l /= 2; + else { + k += temp + 1; + l -= temp + 1; + } + } + if (k == this->end() || key < k->first) + return this->end(); + return k; + } + + inline typename auto find(T&& key) const { + auto k = this->begin(); + size_t l = this->size(); + size_t temp; + while (l > 0) { + if (k[temp = l / 2].first >= key) + l /= 2; + else { + k += temp + 1; + l -= temp + 1; + } + } + if (k == this->end() || key < k->first) + return this->end(); + return k; + } + }; +} diff --git a/src/KKdLib/prj/vector_pair_combine.hpp b/src/KKdLib/prj/vector_pair_combine.hpp new file mode 100644 index 0000000..cbdcb2a --- /dev/null +++ b/src/KKdLib/prj/vector_pair_combine.hpp @@ -0,0 +1,262 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder + + Taken from MSVC's VC/include/memory +*/ + +#pragma once + +#include "../default.hpp" +#include +#include + +namespace prj { + template + class vector_pair_combine { + public: + using value_pair = std::pair; + using iterator = typename std::vector::iterator; + using const_iterator = typename std::vector::const_iterator; + std::vector data; + std::vector new_data; + + inline auto find(const T& key) { + auto k = data.begin(); + size_t l = data.size(); + size_t temp; + while (l > 0) { + if (k[temp = l / 2].first >= key) + l /= 2; + else { + k += temp + 1; + l -= temp + 1; + } + } + if (k == data.end() || key < k->first) + return data.end(); + return k; + } + + inline auto find(const T& key) const { + auto k = data.begin(); + size_t l = data.size(); + size_t temp; + while (l > 0) { + if (k[temp = l / 2].first >= key) + l /= 2; + else { + k += temp + 1; + l -= temp + 1; + } + } + if (k == data.end() || key < k->first) + return data.end(); + return k; + } + + inline auto find(T&& key) { + auto k = data.begin(); + size_t l = data.size(); + size_t temp; + while (l > 0) { + if (k[temp = l / 2].first >= key) + l /= 2; + else { + k += temp + 1; + l -= temp + 1; + } + } + if (k == data.end() || key < k->first) + return data.end(); + return k; + } + + inline auto find(T&& key) const { + auto k = data.begin(); + size_t l = data.size(); + size_t temp; + while (l > 0) { + if (k[temp = l / 2].first >= key) + l /= 2; + else { + k += temp + 1; + l -= temp + 1; + } + } + if (k == data.end() || key < k->first) + return data.end(); + return k; + } + + inline void combine() { + if (data.size() > 1) + std::sort(data.begin(), data.end(), + [](const value_pair& a, const value_pair& b) { + return a.first < b.first; + }); + + for (auto& i : new_data) { + auto elem = find(i.first); + if (elem != data.end()) + elem->second = i.second; + else + data.push_back(i); + } + + new_data.clear(); + + if (data.size() > 1) { + std::sort(data.begin(), data.end(), + [](const value_pair& a, const value_pair& b) { + return a.first < b.first; + }); + + auto begin = data.begin(); + auto end = data.end(); + for (auto i = begin, j = begin + 1; i != end && j != end; ) + if (i->first == j->first) { + std::move(j + 1, end, j); + end--; + } + else { + i++; + j++; + } + + if (data.size() != end - begin) + data.resize(end - begin); + } + } + + inline auto begin() { + return data.begin(); + } + + inline auto begin() const { + return data.begin(); + } + + inline auto cbegin() const { + return data.cbegin(); + } + + inline auto end() { + return data.end(); + } + + inline auto end() const { + return data.end(); + } + + inline auto cend() const { + return data.cend(); + } + + inline auto rbegin() { + return data.rbegin(); + } + + inline auto rbegin() const { + return data.rbegin(); + } + + inline auto crbegin() const { + return data.crbegin(); + } + + inline auto rend() { + return data.rend(); + } + + inline auto rend() const { + return data.rend(); + } + + inline auto crend() const { + return data.crend(); + } + + inline void push_back(const T& first, const U& second) { + new_data.push_back({ first, second }); + } + + inline void push_back(const T& first, U&& second) { + new_data.push_back({ first, second }); + } + + inline void push_back(T&& first, const U& second) { + new_data.push_back({ first, second }); + } + + inline void push_back(T&& first, U&& second) { + new_data.push_back({ first, second }); + } + + inline void push_back(const value_pair& value) { + new_data.push_back(value); + } + + inline void push_back(value_pair&& value) { + new_data.push_back(value); + } + + inline auto erase(const_iterator where) noexcept { + return data.erase(where); + } + + inline auto erase(const_iterator first, const_iterator last) noexcept { + return data.erase(first, last); + } + + inline void clear() noexcept { + data.clear(); + new_data.clear(); + } + + inline void shrink_to_fit() noexcept { + data.shrink_to_fit(); + new_data.shrink_to_fit(); + } + + inline void reserve(size_t new_capacity) { + new_data.reserve(new_capacity); + } + + inline size_t size() const { + return data.size(); + } + + inline value_pair& operator[](const size_t pos) noexcept { + return data[pos]; + } + + inline const value_pair& operator[](const size_t pos) const noexcept { + return data[pos]; + } + + inline value_pair& at(const size_t pos) { + return data.at(pos); + } + + inline const value_pair& at(const size_t pos) const { + return data.at(pos); + } + + inline value_pair& front() noexcept { + return data.front(); + } + + inline const value_pair& front() const noexcept { + return data.front(); + } + + inline value_pair& back() noexcept { + return data.back(); + } + + inline const value_pair& back() const noexcept { + return data.back(); + } + }; +} diff --git a/src/KKdLib/quat.cpp b/src/KKdLib/quat.cpp new file mode 100644 index 0000000..4487041 --- /dev/null +++ b/src/KKdLib/quat.cpp @@ -0,0 +1,10 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "quat.hpp" +#include +#include + +static const quat quat_identity = { 0.0f, 0.0f, 0.0f, 1.0f }; diff --git a/src/KKdLib/quat.hpp b/src/KKdLib/quat.hpp new file mode 100644 index 0000000..09583de --- /dev/null +++ b/src/KKdLib/quat.hpp @@ -0,0 +1,462 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "default.hpp" +#include "vec.hpp" + +struct quat { + float_t x; + float_t y; + float_t z; + float_t w; + + quat(); + quat(float_t value); + quat(float_t x, float_t y, float_t z, float_t w); + quat(const vec3& axis, const float_t angle); + quat(float_t m00, float_t m01, float_t m02, float_t m10, + float_t m11, float_t m12, float_t m20, float_t m21, float_t m22); + + static __m128 load_xmm(const float_t data); + static __m128 load_xmm(const quat& data); + static __m128 load_xmm(const quat&& data); + static quat store_xmm(const __m128& data); + static quat store_xmm(const __m128&& data); + + static quat mul(const quat& in_q1, const quat& in_q2); + + static float_t dot(const quat& left, const quat& right); + static float_t length(const quat& left); + static float_t length_squared(const quat& left); + static float_t distance(const quat& left, const quat& right); + static float_t distance_squared(const quat& left, const quat& right); + static quat abs(const quat& left); + static quat lerp(const quat& left, const quat& right, const float_t blend); + static quat slerp(const quat& left, const quat& right, const float_t blend); + static quat normalize(const quat& left); + static quat rcp(const quat& left); + static quat min(const quat& min, const quat& max); + static quat max(const quat& min, const quat& max); + static quat clamp(const quat& left, const quat& min, const quat& max); + static quat clamp(const quat& left, const float_t min, const float_t max); + static quat mult_min_max(const quat& left, const quat& min, const quat& max); + static quat mult_min_max(const quat& left, const float_t min, const float_t max); + static quat div_min_max(const quat& left, const quat& min, const quat& max); + static quat div_min_max(const quat& left, const float_t min, const float_t max); +}; + +extern const quat quat_identity; + +inline quat::quat() : x(), y(), z(), w() { + +} + +inline quat::quat(float_t value) : x(value), y(value), z(value), w(value) { + +} + +inline quat::quat(float_t x, float_t y, float_t z, float_t w) : x(x), y(y), z(z), w(w) { + +} + +inline quat::quat(const vec3& axis, const float_t angle) { + vec3 _axis = vec3::normalize(axis) * sinf(angle * 0.5f); + x = _axis.x; + y = _axis.y; + z = _axis.z; + w = cosf(angle * 0.5f); +} + +inline quat::quat(float_t m00, float_t m01, float_t m02, float_t m10, + float_t m11, float_t m12, float_t m20, float_t m21, float_t m22) { + if (m00 + m11 + m22 >= 0.0f) { + float_t sq = sqrtf(m00 + m11 + m22 + 1.0f); + w = sq * 0.5f; + sq = 0.5f / sq; + x = (m21 - m12) * sq; + y = (m02 - m20) * sq; + z = (m10 - m01) * sq; + return; + } + + float_t max = max_def(m22, max_def(m11, m00)); + if (max == m00) { + float_t sq = sqrtf(m00 - (m11 + m22) + 1.0f); + x = sq * 0.5f; + sq = 0.5f / sq; + y = (m01 + m10) * sq; + z = (m02 + m20) * sq; + w = (m21 - m12) * sq; + } + else if (max == m11) { + float_t sq = sqrtf(m11 - (m00 + m22) + 1.0f); + y = sq * 0.5f; + sq = 0.5f / sq; + x = (m01 + m10) * sq; + z = (m12 + m21) * sq; + w = (m02 - m20) * sq; + } + else { + float_t sq = sqrtf(m22 - (m00 + m11) + 1.0f); + z = sq * 0.5f; + sq = 0.5f / sq; + x = (m02 + m20) * sq; + y = (m12 + m21) * sq; + w = (m10 - m01) * sq; + } +} + +inline quat operator +(const quat& left, const quat& right) { + __m128 yt; + quat z; + *(quat*)&yt = right; + _mm_storeu_ps((float*)&z, _mm_add_ps(_mm_loadu_ps((const float*)&left), yt)); + return z; +} + +inline quat operator +(const quat& left, const float_t right) { + __m128 yt; + quat z; + yt = _mm_set_ss(right); + _mm_storeu_ps((float*)&z, _mm_add_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0))); + return z; +} + +inline quat operator -(const quat& left, const quat& right) { + __m128 yt; + quat z; + *(quat*)&yt = right; + _mm_storeu_ps((float*)&z, _mm_sub_ps(_mm_loadu_ps((const float*)&left), yt)); + return z; +} + +inline quat operator -(const quat& left, const float_t right) { + __m128 yt; + quat z; + yt = _mm_set_ss(right); + _mm_storeu_ps((float*)&z, _mm_sub_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0))); + return z; +} + +inline quat operator *(const quat& left, const quat& right) { + __m128 yt; + quat z; + *(quat*)&yt = right; + _mm_storeu_ps((float*)&z, _mm_mul_ps(_mm_loadu_ps((const float*)&left), yt)); + return z; +} + +inline quat operator *(const quat& left, const float_t right) { + __m128 yt; + quat z; + yt = _mm_set_ss(right); + _mm_storeu_ps((float*)&z, _mm_mul_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0))); + return z; +} + +inline quat operator /(const quat& left, const quat& right) { + __m128 yt; + quat z; + *(quat*)&yt = right; + _mm_storeu_ps((float*)&z, _mm_div_ps(_mm_loadu_ps((const float*)&left), yt)); + return z; +} + +inline quat operator /(const quat& left, const float_t right) { + __m128 yt; + quat z; + yt = _mm_set_ss(right); + _mm_storeu_ps((float*)&z, _mm_div_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0))); + return z; +} + +inline quat operator &(const quat& left, const quat& right) { + __m128 yt; + quat z; + *(quat*)&yt = right; + _mm_storeu_ps((float*)&z, _mm_and_ps(_mm_loadu_ps((const float*)&left), yt)); + return z; +} + +inline quat operator &(const quat& left, const float_t right) { + __m128 yt; + quat z; + yt = _mm_set_ss(right); + _mm_storeu_ps((float*)&z, _mm_and_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0))); + return z; +} + +inline quat operator ^(const quat& left, const quat& right) { + __m128 yt; + quat z; + *(quat*)&yt = right; + _mm_storeu_ps((float*)&z, _mm_xor_ps(_mm_loadu_ps((const float*)&left), yt)); + return z; +} + +inline quat operator ^(const quat& left, const float_t right) { + __m128 yt; + quat z; + yt = _mm_set_ss(right); + _mm_storeu_ps((float*)&z, _mm_xor_ps(_mm_loadu_ps((const float*)&left), _mm_shuffle_ps(yt, yt, 0))); + return z; +} + +inline quat operator -(const quat& left) { + quat z; + _mm_storeu_ps((float*)&z, _mm_xor_ps(_mm_loadu_ps((const float*)&left), vec4_neg)); + return z; +} + +inline __m128 quat::load_xmm(const float_t data) { + __m128 _data = _mm_set_ss(data); + return _mm_shuffle_ps(_data, _data, 0); +} + +inline __m128 quat::load_xmm(const quat& data) { + return _mm_loadu_ps((const float*)&data); +} + +inline __m128 quat::load_xmm(const quat&& data) { + return _mm_loadu_ps((const float*)&data); +} + +inline quat quat::store_xmm(const __m128& data) { + quat _data; + _mm_storeu_ps((float*)&_data, data); + return _data; +} + +inline quat quat::store_xmm(const __m128&& data) { + quat _data; + _mm_storeu_ps((float*)&_data, data); + return _data; +} + +inline quat quat::mul(const quat& in_q1, const quat& in_q2) { + __m128 xt; + __m128 yt; + __m128 zt0; + __m128 zt1; + __m128 zt2; + __m128 zt3; + + xt = quat::load_xmm(in_q1); + yt = quat::load_xmm(in_q2); + zt0 = _mm_mul_ps(xt, _mm_shuffle_ps(yt, yt, 0x1B)); + zt1 = _mm_mul_ps(xt, _mm_shuffle_ps(yt, yt, 0x4E)); + zt2 = _mm_mul_ps(xt, _mm_shuffle_ps(yt, yt, 0xB1)); + zt3 = _mm_mul_ps(xt, _mm_shuffle_ps(yt, yt, 0xE4)); + zt0 = _mm_xor_ps(zt0, __m128({ 0.0f, 0.0f, -0.0f, 0.0f })); + zt1 = _mm_xor_ps(zt1, __m128({ -0.0f, 0.0f, 0.0f, 0.0f })); + zt2 = _mm_xor_ps(zt2, __m128({ 0.0f, -0.0f, 0.0f, 0.0f })); + zt3 = _mm_xor_ps(zt3, __m128({ -0.0f, -0.0f, -0.0f, 0.0f })); + zt0 = _mm_hadd_ps(zt0, zt0); + zt1 = _mm_hadd_ps(zt1, zt1); + zt2 = _mm_hadd_ps(zt2, zt2); + zt3 = _mm_hadd_ps(zt3, zt3); + + quat out_q; + out_q.x = _mm_cvtss_f32(_mm_hadd_ps(zt0, zt0)); + out_q.y = _mm_cvtss_f32(_mm_hadd_ps(zt1, zt1)); + out_q.z = _mm_cvtss_f32(_mm_hadd_ps(zt2, zt2)); + out_q.w = _mm_cvtss_f32(_mm_hadd_ps(zt3, zt3)); + return out_q; +} + +inline float_t quat::dot(const quat& left, const quat& right) { + __m128 zt; + zt = _mm_mul_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right))); + zt = _mm_hadd_ps(zt, zt); + return _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline float_t quat::length(const quat& left) { + __m128 xt; + __m128 zt; + xt = _mm_loadu_ps((const float*)&left); + zt = _mm_mul_ps(xt, xt); + zt = _mm_hadd_ps(zt, zt); + return _mm_cvtss_f32(_mm_sqrt_ss(_mm_hadd_ps(zt, zt))); +} + +inline float_t quat::length_squared(const quat& left) { + __m128 xt; + __m128 zt; + xt = _mm_loadu_ps((const float*)&left); + zt = _mm_mul_ps(xt, xt); + zt = _mm_hadd_ps(zt, zt); + return _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline float_t quat::distance(const quat& left, const quat& right) { + __m128 zt; + zt = _mm_sub_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right))); + zt = _mm_mul_ps(zt, zt); + zt = _mm_hadd_ps(zt, zt); + return _mm_cvtss_f32(_mm_sqrt_ss(_mm_hadd_ps(zt, zt))); +} + +inline float_t quat::distance_squared(const quat& left, const quat& right) { + __m128 zt; + zt = _mm_sub_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right))); + zt = _mm_mul_ps(zt, zt); + zt = _mm_hadd_ps(zt, zt); + return _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline quat quat::abs(const quat& left) { + return quat::store_xmm(_mm_castsi128_ps(_mm_and_si128(_mm_castps_si128(quat::load_xmm(left)), vec4i_abs))); +} + +inline quat quat::lerp(const quat& left, const quat& right, const float_t blend) { + quat x_t; + quat y_t; + x_t = left; + y_t = right; + + if (quat::dot(x_t, y_t) < 0.0f) + x_t = -x_t; + + return quat::normalize(x_t * (1.0f - blend) + y_t * blend); +} + +inline quat quat::slerp(const quat& left, const quat& right, const float_t blend) { + quat x_t; + quat y_t; + x_t = left; + y_t = right; + + float_t dot = quat::dot(x_t, y_t); + if (dot < 0.0f) { + dot = -dot; + x_t = -x_t; + } + + dot = min_def(dot, 1.0f); + + float_t theta = acosf(dot); + if (theta == 0.0f) + return x_t; + + float_t st = 1.0f / sinf(theta); + float_t s0 = sinf((1.0f - blend) * theta) * st; + float_t s1 = sinf(theta * blend) * st; + return quat::normalize(x_t * s0 + y_t * s1); +} + +inline quat quat::normalize(const quat& left) { + __m128 xt; + __m128 zt; + quat z; + xt = _mm_loadu_ps((const float*)&left); + zt = _mm_mul_ps(xt, xt); + zt = _mm_hadd_ps(zt, zt); + zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt)); + if (zt.m128_f32[0] != 0.0f) + zt.m128_f32[0] = 1.0f / zt.m128_f32[0]; + _mm_storeu_ps((float*)&z, _mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0))); + return z; +} + +inline quat quat::rcp(const quat& left) { + quat z; + _mm_storeu_ps((float*)&z, _mm_div_ps(_mm_loadu_ps((const float*)&(quat_identity)), _mm_loadu_ps((const float*)&left))); + return z; +} + +inline quat quat::min(const quat& left, const quat& right) { + quat z; + _mm_storeu_ps((float*)&z, _mm_min_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right)))); + return z; +} + +inline quat quat::max(const quat& left, const quat& right) { + quat z; + _mm_storeu_ps((float*)&z, _mm_max_ps(_mm_loadu_ps((const float*)&(left)), _mm_loadu_ps((const float*)&(right)))); + return z; +} + +inline quat quat::clamp(const quat& left, const quat& min, const quat& max) { + quat w; + _mm_storeu_ps((float*)&w, _mm_min_ps(_mm_max_ps(_mm_loadu_ps((const float*)&left), + _mm_loadu_ps((const float*)&(min))), _mm_loadu_ps((const float*)&(max)))); + return w; +} + +inline quat quat::clamp(const quat& left, const float_t min, const float_t max) { + __m128 yt; + __m128 zt; + quat w; + yt = _mm_set_ss(min); + zt = _mm_set_ss(max); + _mm_storeu_ps((float*)&w, _mm_min_ps(_mm_max_ps(_mm_loadu_ps((const float*)&left), + _mm_shuffle_ps(yt, yt, 0)), _mm_shuffle_ps(zt, zt, 0))); + return w; +} + +inline quat quat::mult_min_max(const quat& left, const quat& min, const quat& max) { + __m128 xt; + __m128 yt; + __m128 wt; + quat w; + xt = _mm_loadu_ps((const float*)&left); + yt = _mm_xor_ps(_mm_loadu_ps((const float*)&(min)), vec4_neg); + wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))), + _mm_and_ps(_mm_loadu_ps((const float*)&(max)), _mm_cmpge_ps(xt, vec4::load_xmm(0.0f)))); + _mm_storeu_ps((float*)&w, _mm_mul_ps(xt, wt)); + return w; +} + +inline quat quat::mult_min_max(const quat& left, const float_t min, const float_t max) { + __m128 xt; + __m128 yt; + __m128 zt; + __m128 wt; + quat w; + xt = _mm_loadu_ps((const float*)&left); + yt = _mm_set_ss(min); + yt = _mm_shuffle_ps(yt, yt, 0); + zt = _mm_set_ss(max); + zt = _mm_shuffle_ps(zt, zt, 0); + yt = _mm_xor_ps(yt, vec4_neg); + wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))), + _mm_and_ps(zt, _mm_cmpge_ps(xt, vec4::load_xmm(0.0f)))); + _mm_storeu_ps((float*)&w, _mm_mul_ps(xt, wt)); + return w; +} + +inline quat quat::div_min_max(const quat& left, const quat& min, const quat& max) { + __m128 xt; + __m128 yt; + __m128 wt; + quat w; + xt = _mm_loadu_ps((const float*)&left); + yt = _mm_xor_ps(_mm_loadu_ps((const float*)&(min)), vec4_neg); + wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))), + _mm_and_ps(_mm_loadu_ps((const float*)&(max)), _mm_cmpge_ps(xt, vec4::load_xmm(0.0f)))); + _mm_storeu_ps((float*)&w, _mm_div_ps(xt, wt)); + return w; +} + +inline quat quat::div_min_max(const quat& left, const float_t min, const float_t max) { + __m128 xt; + __m128 yt; + __m128 zt; + __m128 wt; + quat w; + xt = _mm_loadu_ps((const float*)&left); + yt = _mm_set_ss(min); + yt = _mm_shuffle_ps(yt, yt, 0); + zt = _mm_set_ss(max); + zt = _mm_shuffle_ps(zt, zt, 0); + yt = _mm_xor_ps(yt, vec4_neg); + wt = _mm_or_ps(_mm_and_ps(yt, _mm_cmplt_ps(xt, vec4::load_xmm(0.0f))), + _mm_and_ps(zt, _mm_cmpge_ps(xt, vec4::load_xmm(0.0f)))); + _mm_storeu_ps((float*)&w, _mm_div_ps(xt, wt)); + return w; +} diff --git a/src/KKdLib/str_utils.cpp b/src/KKdLib/str_utils.cpp new file mode 100644 index 0000000..6b841c7 --- /dev/null +++ b/src/KKdLib/str_utils.cpp @@ -0,0 +1,690 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "str_utils.hpp" + +bool str_utils_check_ends_with(const char* str, const char* mask) { + if (!str || !mask) + return false; + + size_t mask_len = utf8_length(mask); + size_t len = utf8_length(str); + const char* t = str; + while (t) { + t = strstr(t, mask); + if (t) { + t += mask_len; + if (t == str + len) + return true; + } + } + return false; +} + +bool str_utils_check_ends_with(const wchar_t* str, const wchar_t* mask) { + if (!str || !mask) + return false; + + size_t mask_len = utf16_length(mask); + size_t len = utf16_length(str); + const wchar_t* t = str; + while (t) { + t = wcsstr(t, mask); + if (t) { + t += mask_len; + if (t == str + len) + return true; + } + } + return false; +} + +const char* str_utils_get_next_int32_t(const char* str, int32_t& value, const char split) { + std::string s; + str = str_utils_get_next_string(str, s, split); + sscanf_s(s.c_str(), "%d", &value); + return str; +} + +const wchar_t* str_utils_get_next_int32_t(const wchar_t* str, int32_t& value, const wchar_t split) { + std::wstring s; + str = str_utils_get_next_string(str, s, split); + swscanf_s(s.c_str(), L"%d", &value); + return str; +} + +const char* str_utils_get_next_float_t(const char* str, float_t& value, const char split) { + std::string s; + str = str_utils_get_next_string(str, s, split); + sscanf_s(s.c_str(), "%f", &value); + return str; +} + +const wchar_t* str_utils_get_next_float_t(const wchar_t* str, float_t& value, const wchar_t split) { + std::wstring s; + str = str_utils_get_next_string(str, s, split); + swscanf_s(s.c_str(), L"%f", &value); + return str; +} + +const char* str_utils_get_next_string(const char* str, std::string& value, const char split) { + value.clear(); + + if (!str) + return 0; + + const char* t = strchr(str, split); + if (!t) { + value.assign(str); + return 0; + } + + value.assign(str, t - str); + t++; + return *t ? t : 0; +} + +const wchar_t* str_utils_get_next_string(const wchar_t* str, std::wstring& value, const wchar_t split) { + value.clear(); + + if (!str) + return 0; + + const wchar_t* t = wcschr(str, split); + if (!t) { + value.assign(str); + return 0; + } + + value.assign(str, t - str); + t++; + return *t ? t : 0; +} + +char* str_utils_split_get_right(const char* str, const char split) { + if (!str) + return 0; + + const char* t = strchr(str, split); + if (!t) + return str_utils_copy(str); + t++; + + size_t len = utf8_length(t); + char* p = force_malloc(len + 1); + memcpy(p, t, len); + p[len] = 0; + return p; +} + +wchar_t* str_utils_split_get_right(const wchar_t* str, const wchar_t split) { + if (!str) + return 0; + + const wchar_t* t = wcschr(str, split); + if (!t) + return str_utils_copy(str); + t++; + + size_t len = utf16_length(t); + wchar_t* p = force_malloc(len + 1); + memcpy(p, t, sizeof(wchar_t) * len); + p[len] = 0; + return p; +} + +char* str_utils_split_get_left(const char* str, const char split) { + if (!str) + return 0; + + const char* t = strchr(str, split); + + size_t len = t ? t - str : utf8_length(str); + char* p = force_malloc(len + 1); + memcpy(p, str, len); + p[len] = 0; + return p; +} + +wchar_t* str_utils_split_get_left(const wchar_t* str, const wchar_t split) { + if (!str) + return 0; + + const wchar_t* t = wcschr(str, split); + + size_t len = t ? t - str : utf16_length(str); + wchar_t* p = force_malloc(len + 1); + memcpy(p, str, sizeof(wchar_t) * len); + p[len] = 0; + return p; +} + +char* str_utils_split_get_right_include(const char* str, const char split) { + if (!str) + return 0; + + const char* t = strchr(str, split); + if (!t) + return str_utils_copy(str); + + size_t len = utf8_length(t); + char* p = force_malloc(len + 1); + memcpy(p, t, len); + p[len] = 0; + return p; +} + +wchar_t* str_utils_split_get_right_include(const wchar_t* str, const wchar_t split) { + if (!str) + return 0; + + const wchar_t* t = wcschr(str, split); + if (!t) + return str_utils_copy(str); + + size_t len = utf16_length(t); + wchar_t* p = force_malloc(len + 1); + memcpy(p, t, sizeof(wchar_t) * len); + p[len] = 0; + return p; +} + +char* str_utils_split_get_left_include(const char* str, const char split) { + if (!str) + return 0; + + const char* t = strchr(str, split); + t++; + + size_t len = t ? t - str : utf8_length(str); + char* p = force_malloc(len + 1); + memcpy(p, str, len); + p[len] = 0; + return p; +} + +wchar_t* str_utils_split_get_left_include(const wchar_t* str, const wchar_t split) { + if (!str) + return 0; + + const wchar_t* t = wcschr(str, split); + t++; + + size_t len = t ? t - str : utf16_length(str); + wchar_t* p = force_malloc(len + 1); + memcpy(p, str, sizeof(wchar_t) * len); + p[len] = 0; + return p; +} + +char* str_utils_split_right_get_right(const char* str, const char split) { + if (!str) + return 0; + + const char* t = strrchr(str, split); + if (!t) + return str_utils_copy(str); + t++; + + size_t len = t - str; + char* p = force_malloc(len + 1); + memcpy(p, t, len); + p[len] = 0; + return p; +} + +wchar_t* str_utils_split_right_get_right(const wchar_t* str, const wchar_t split) { + if (!str) + return 0; + + const wchar_t* t = wcsrchr(str, split); + if (!t) + return str_utils_copy(str); + t++; + + size_t len = t - str; + wchar_t* p = force_malloc(len + 1); + memcpy(p, t, sizeof(wchar_t) * len); + p[len] = 0; + return p; +} + +char* str_utils_split_right_get_left(const char* str, const char split) { + if (!str) + return 0; + + const char* t = strrchr(str, split); + + size_t len = t ? t - str : utf8_length(str); + char* p = force_malloc(len + 1); + memcpy(p, str, len); + p[len] = 0; + return p; +} + +wchar_t* str_utils_split_right_get_left(const wchar_t* str, const wchar_t split) { + if (!str) + return 0; + + const wchar_t* t = wcsrchr(str, split); + + size_t len = t ? t - str : utf16_length(str); + wchar_t* p = force_malloc(len + 1); + memcpy(p, str, sizeof(wchar_t) * len); + p[len] = 0; + return p; +} + +char* str_utils_split_right_get_right_include(const char* str, const char split) { + if (!str) + return 0; + + const char* t = strrchr(str, split); + if (!t) + return str_utils_copy(str); + + size_t len = utf8_length(t); + char* p = force_malloc(len + 1); + memcpy(p, t, len); + p[len] = 0; + return p; +} + +wchar_t* str_utils_split_right_get_right_include(const wchar_t* str, const wchar_t split) { + if (!str) + return 0; + + const wchar_t* t = wcsrchr(str, split); + if (!t) + return str_utils_copy(str); + + size_t len = utf16_length(t); + wchar_t* p = force_malloc(len + 1); + memcpy(p, t, sizeof(wchar_t) * len); + p[len] = 0; + return p; +} + +char* str_utils_split_right_get_left_include(const char* str, const char split) { + if (!str) + return 0; + + const char* t = strrchr(str, split); + if (t) + t++; + + size_t len = t ? t - str : utf8_length(str); + char* p = force_malloc(len + 1); + memcpy(p, str, len); + p[len] = 0; + return p; +} + +wchar_t* str_utils_split_right_get_left_include(const wchar_t* str, const wchar_t split) { + if (!str) + return 0; + + const wchar_t* t = wcsrchr(str, split); + if (t) + t++; + + size_t len = t ? t - str : utf16_length(str); + wchar_t* p = force_malloc(len + 1); + memcpy(p, str, sizeof(wchar_t) * len); + p[len] = 0; + return p; +} + +char* str_utils_get_extension(const char* str) { + if (!str) + return 0; + + const char* t = strrchr(str, '\\'); + return str_utils_split_right_get_right_include(t ? t + 1 : str, '.'); +} + +wchar_t* str_utils_get_extension(const wchar_t* str) { + if (!str) + return 0; + + const wchar_t* t = wcsrchr(str, L'\\'); + return str_utils_split_right_get_right_include(t ? t + 1 : str, L'.'); +} + +char* str_utils_get_without_extension(const char* str) { + if (!str) + return 0; + + const char* t = strrchr(str, '\\'); + return str_utils_split_right_get_left(t ? t + 1 : str, '.'); +} + +wchar_t* str_utils_get_without_extension(const wchar_t* str) { + if (!str) + return 0; + + const wchar_t* t = wcsrchr(str, L'\\'); + return str_utils_split_right_get_left(t ? t + 1 : str, L'.'); +} + +char* str_utils_add(const char* str0, const char* str1) { + if (str0 && str1) { + size_t str0_len = utf8_length(str0); + size_t str1_len = utf8_length(str1); + char* p = force_malloc(str0_len + str1_len + 1); + memcpy(p, str0, str0_len + 1); + memcpy(p + str0_len, str1, str1_len + 1); + return p; + } + else if (str0) + return str_utils_copy(str0); + else if (str1) + return str_utils_copy(str1); + else + return 0; +} + +wchar_t* str_utils_add(const wchar_t* str0, const wchar_t* str1) { + if (str0 && str1) { + size_t str0_len = utf16_length(str0); + size_t str1_len = utf16_length(str1); + wchar_t* p = force_malloc(str0_len + str1_len + 1); + memcpy(p, str0, sizeof(wchar_t) * (str0_len + 1)); + memcpy(p + str0_len, str1, sizeof(wchar_t) * (str1_len + 1)); + return p; + } + else if (str0) + return str_utils_copy(str0); + else if (str1) + return str_utils_copy(str1); + else + return 0; +} + +char* str_utils_copy(const char* str) { + if (!str) + return 0; + + size_t len = utf8_length(str) + 1; + char* p = force_malloc(len); + memcpy(p, str, len); + return p; +} + +wchar_t* str_utils_copy(const wchar_t* str) { + if (!str) + return 0; + + size_t len = utf16_length(str) + 1; + wchar_t* p = force_malloc(len); + memcpy(p, str, sizeof(wchar_t) * len); + return p; +} + +inline int32_t str_utils_compare_length(const char* str0, size_t str0_len, const char* str1, size_t str1_len) { + if (!str0_len) + return -*str1; + else if (!str1_len) + return *str0; + + size_t str0_len_act = str0_len; + const char* i0 = str0; + for (size_t i = str0_len; i; i--) + if (!*i0++) { + str0_len_act = i + 1; + break; + } + + size_t str1_len_act = str1_len; + const char* i1 = str1; + for (size_t i = str1_len; i; i--) + if (!*i1++) { + str1_len_act = i + 1; + break; + } + + str0_len = str0_len_act; + str1_len = str1_len_act; + + int32_t diff = 0; + char c0; + char c1; + do { + c0 = *str0++; + c1 = *str1++; + if (!c0 || !c1) + return c0 - c1; + } while (c0 == c1 && --str0_len && --str1_len); + return c0 - c1; +} + +inline int32_t str_utils_compare_length(const wchar_t* str0, size_t str0_len, const wchar_t* str1, size_t str1_len) { + if (!str0_len) + return -*str1; + else if (!str1_len) + return *str0; + + size_t str0_len_act = str0_len; + const wchar_t* i0 = str0; + for (size_t i = str0_len; i; i--) + if (!*i0++) { + str0_len_act = i + 1; + break; + } + + size_t str1_len_act = str1_len; + const wchar_t* i1 = str1; + for (size_t i = str1_len; i; i--) + if (!*i1++) { + str1_len_act = i + 1; + break; + } + + str0_len = str0_len_act; + str1_len = str1_len_act; + + int32_t diff = 0; + wchar_t c0; + wchar_t c1; + do { + c0 = *str0++; + c1 = *str1++; + if (!c0 || !c1) + return c0 - c1; + } while (c0 == c1 && --str0_len && --str1_len); + return c0 - c1; +} + +size_t str_utils_get_substring_offset(const char* str0, size_t str0_len, + size_t str0_off, const char* str1, size_t str1_len) { + if (!str1_len && str0_off <= str0_len) + return str0_off; + + if (str0_off < str0_len && str1_len <= str0_len - str0_off) { + size_t len = str0_len - str1_len - str0_off + 1; + const char* str = &str0[str0_off]; + for (; len; ) { + const char* s = (const char*)memchr(str, *str1, len); + if (!s) + break; + + if (!str1_len || !memcmp(s, str1, str1_len)) + return s - str0; + + len += str - (s + 1); + str = s + 1; + } + } + return -1; +} + +size_t str_utils_get_substring_offset(const wchar_t* str0, size_t str0_len, + size_t str0_off, const wchar_t* str1, size_t str1_len) { + if (!str1_len && str0_off <= str0_len) + return str0_off; + + if (str0_off < str0_len && str1_len <= str0_len - str0_off) { + size_t len = str0_len - str1_len - str0_off + 1; + const wchar_t* str = &str0[str0_off]; + for (; len; ) { + const wchar_t* s = wmemchr(str, *str1, len); + if (!s) + break; + + if (!str1_len || !memcmp(s, str1, sizeof(wchar_t) * str1_len)) + return s - str0; + + len += str - (s + 1); + str = s + 1; + } + } + return -1; +} + +bool str_utils_text_file_parse(const void* data, size_t size, + char*& buf, char**& lines, size_t& count) { + if (!data || !size) + return false; + + const char* d = (const char*)data; + bool del = false; + size_t c; + buf = 0; + lines = 0; + count = 0; + if ((uint8_t)d[0] == 0x00) + return false; + else if (d[0] == 0xFF) { + if (size == 1 || (uint8_t)d[1] != 0xFE || size == 2) + return false; + + wchar_t* w_d = (wchar_t*)data + 1; + d = utf16_to_utf8(w_d); + size = utf8_length(d); + del = true; + goto decode_utf8_ansi; + } + else if ((uint8_t)d[0] == 0xFE) { + if (size == 1 || (uint8_t)d[1] != 0xFF || size == 2) + return false; + + size /= 2; + wchar_t* w_d = (wchar_t*)data + 1; + w_d = str_utils_copy(w_d); + for (size_t i = 0; i < size; i++) + w_d[i] = (wchar_t)reverse_endianness_uint16_t((uint16_t)w_d[i]); + d = utf16_to_utf8(w_d); + size = utf8_length(d); + del = true; + free_def(w_d); + goto decode_utf8_ansi; + } + else if ((uint8_t)d[0] == 0xEF) { + if (size == 1 || (uint8_t)d[1] != 0xBB || size == 2 || (uint8_t)d[2] != 0xBF || size == 3) + return false; + + d += 3; + size -= 3; + goto decode_utf8_ansi; + } + else { + decode_utf8_ansi: + c = 1; + bool lf; + char ch; + const char* t; + lf = false; + t = d; + ch = 0; + + size_t buf_len = size; + for (size_t i = 0, l = 0, m = 0; i < size; i++) { + ch = *t++; + if (ch == '\r') { + if (i + 1 < size && *t == '\n') { + i++; + t++; + l++; + } + lf = true; + } + else if (ch == '\n') + lf = true; + + if (lf) { + if (!l && c > 1) + buf_len--; + c++; + l = 0; + m = 0; + lf = false; + } + else { + l++; + m++; + } + } + + if (ch != '\r' && ch != '\n') + buf_len++; + else + c--; + + lf = false; + t = d; + char* temp_buf = force_malloc(buf_len); + char** temp_lines = force_malloc(c); + + char* b = temp_buf; + for (size_t i = 0, j = 0, l = 0, m = 0; j < c; i++) { + ch = *t++; + if (ch == '\r') { + if (i + 1 < size && *t == '\n') { + i++; + t++; + l++; + } + lf = true; + } + else if (ch == '\n') + lf = true; + + if (i >= size || lf) { + temp_lines[j] = b; + if (l) { + memcpy(b, d + i - l, m); + b[l] = 0; + b += l + 1; + } + else if (j) + temp_lines[j]--; + else + *b++ = 0; + j++; + + if (!lf) + break; + + l = 0; + m = 0; + lf = false; + } + else { + l++; + m++; + } + } + + buf = temp_buf; + lines = temp_lines; + count = c; + } + + if (del) { + void* data = (void*)d; + free_def(data); + } + return true; +} diff --git a/src/KKdLib/str_utils.hpp b/src/KKdLib/str_utils.hpp new file mode 100644 index 0000000..16ce8b0 --- /dev/null +++ b/src/KKdLib/str_utils.hpp @@ -0,0 +1,59 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include +#include +#include "default.hpp" + +inline int32_t str_utils_compare(const char* str0, const char* str1) { + return strcmp(str0, str1); +} + +inline int32_t str_utils_compare(const wchar_t* str0, const wchar_t* str1) { + return wcscmp(str0, str1); +} + +extern bool str_utils_check_ends_with(const char* str, const char* mask); +extern bool str_utils_check_ends_with(const wchar_t* str, const wchar_t* mask); +extern const char* str_utils_get_next_int32_t(const char* str, int32_t& value, const char split); +extern const wchar_t* str_utils_get_next_int32_t(const wchar_t* str, int32_t& value, const wchar_t split); +extern const char* str_utils_get_next_float_t(const char* str, float_t& value, const char split); +extern const wchar_t* str_utils_get_next_float_t(const wchar_t* str, float_t& value, const wchar_t split); +extern const char* str_utils_get_next_string(const char* str, std::string& value, const char split); +extern const wchar_t* str_utils_get_next_string(const wchar_t* str, std::wstring& value, const wchar_t split); +extern char* str_utils_split_get_right(const char* str, const char split); +extern wchar_t* str_utils_split_get_right(const wchar_t* str, const wchar_t split); +extern char* str_utils_split_get_left(const char* str, const char split); +extern wchar_t* str_utils_split_get_left(const wchar_t* str, const wchar_t split); +extern char* str_utils_split_get_right_include(const char* str, const char split); +extern wchar_t* str_utils_split_get_right_include(const wchar_t* str, const wchar_t split); +extern char* str_utils_split_get_left_include(const char* str, const char split); +extern wchar_t* str_utils_split_get_left_include(const wchar_t* str, const wchar_t split); +extern char* str_utils_split_right_get_right(const char* str, const char split); +extern wchar_t* str_utils_split_right_get_right(const wchar_t* str, const wchar_t split); +extern char* str_utils_split_right_get_left(const char* str, const char split); +extern wchar_t* str_utils_split_right_get_left(const wchar_t* str, const wchar_t split); +extern char* str_utils_split_right_get_right_include(const char* str, const char split); +extern wchar_t* str_utils_split_right_get_right_include(const wchar_t* str, const wchar_t split); +extern char* str_utils_split_right_get_left_include(const char* str, const char split); +extern wchar_t* str_utils_split_right_get_left_include(const wchar_t* str, const wchar_t split); +extern char* str_utils_get_extension(const char* str); +extern wchar_t* str_utils_get_extension(const wchar_t* str); +extern char* str_utils_get_without_extension(const char* str); +extern wchar_t* str_utils_get_without_extension(const wchar_t* str); +extern char* str_utils_add(const char* str0, const char* str1); +extern wchar_t* str_utils_add(const wchar_t* str0, const wchar_t* str1); +extern char* str_utils_copy(const char* str); +extern wchar_t* str_utils_copy(const wchar_t* str); +extern int32_t str_utils_compare_length(const char* str0, size_t str0_len, const char* str1, size_t str1_len); +extern int32_t str_utils_compare_length(const wchar_t* str0, size_t str0_len, const wchar_t* str1, size_t str1_len); +extern size_t str_utils_get_substring_offset(const char* str0, size_t str0_len, + size_t str0_off, const char* str1, size_t str1_len); +extern size_t str_utils_get_substring_offset(const wchar_t* str0, size_t str0_len, + size_t str0_off, const wchar_t* str1, size_t str1_len); +extern bool str_utils_text_file_parse(const void* data, size_t size, + char*& buf, char**& lines, size_t& count); diff --git a/src/KKdLib/time.cpp b/src/KKdLib/time.cpp new file mode 100644 index 0000000..f395acb --- /dev/null +++ b/src/KKdLib/time.cpp @@ -0,0 +1,39 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "time.hpp" + +static double_t time_struct_get_freq(); + +time_struct::time_struct() : timestamp() { + get_timestamp(); + inv_freq = time_struct_get_freq(); +} + +double_t time_struct::calc_time() { + LARGE_INTEGER timestamp; + if (QueryPerformanceCounter(×tamp)) + return (double_t)(timestamp.QuadPart - this->timestamp.QuadPart) * inv_freq; + return 0.0; +} + +int64_t time_struct::calc_time_int() { + LARGE_INTEGER timestamp; + if (QueryPerformanceCounter(×tamp)) + return (int64_t)((double_t)(timestamp.QuadPart - this->timestamp.QuadPart) * inv_freq * 1000.0); + return 0; +} + +void time_struct::get_timestamp() { + if (!QueryPerformanceCounter(×tamp)) + timestamp.QuadPart = 0; +} + +static double_t time_struct_get_freq() { + LARGE_INTEGER freq; + if (QueryPerformanceFrequency(&freq)) + return 1000.0 / (double_t)freq.LowPart; + return 0.0; +} diff --git a/src/KKdLib/time.hpp b/src/KKdLib/time.hpp new file mode 100644 index 0000000..2f8d966 --- /dev/null +++ b/src/KKdLib/time.hpp @@ -0,0 +1,19 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "default.hpp" + +struct time_struct { + LARGE_INTEGER timestamp; + double_t inv_freq; + + time_struct(); + + double_t calc_time(); + int64_t calc_time_int(); + void get_timestamp(); +}; diff --git a/src/KKdLib/timer.cpp b/src/KKdLib/timer.cpp new file mode 100644 index 0000000..bd8bb9c --- /dev/null +++ b/src/KKdLib/timer.cpp @@ -0,0 +1,110 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "timer.hpp" + +timer::timer(double_t freq) : history(), curr_time(), prev_time(), inv_freq() { + for (double_t& i : history) + i = freq; + history_counter = 0; + this->freq = freq; + this->freq_hist = freq; + + LARGE_INTEGER frequency; + if (QueryPerformanceFrequency(&frequency)) + inv_freq = 1000.0 / (double_t)frequency.LowPart; + else + inv_freq = 0.0; + + if (!QueryPerformanceCounter(&curr_time)) + curr_time.QuadPart = 0; + if (!QueryPerformanceCounter(&prev_time)) + prev_time.QuadPart = 0; +} + +timer::~timer() { + +} + +void timer::start_of_cycle() { + LARGE_INTEGER timestamp; + if (!QueryPerformanceCounter(×tamp)) + timestamp.QuadPart = 0; + + double_t time = (timestamp.QuadPart - curr_time.QuadPart) * inv_freq; + + if (!QueryPerformanceCounter(&curr_time)) + curr_time.QuadPart = 0; + + history[history_counter] = time; + + double_t freq = 0.0; + for (uint8_t i = 0; i < history_counter; i++) + freq += history[i]; + freq += time; + for (uint8_t i = history_counter + 1; i < HISTORY_COUNT; i++) + freq += history[i]; + + history_counter++; + if (history_counter >= HISTORY_COUNT) + history_counter = 0; + + std::unique_lock u_lock(freq_mtx); + freq_hist = 1000.0 / (freq / HISTORY_COUNT); +} + +void timer::end_of_cycle() { + LONGLONG& curr_timestamp = curr_time.QuadPart; + LONGLONG& prev_timestamp = prev_time.QuadPart; + + double_t msec = 1000.0 / get_freq() - (curr_timestamp - prev_timestamp) * inv_freq; + if (floor(msec) > 0) + wait_timer.sleep_float(msec); + + if (msec < -get_freq() / 4.0) + prev_timestamp = curr_timestamp - (LONGLONG)(get_freq() / 4.0 / inv_freq); + else + prev_timestamp = curr_timestamp + (LONGLONG)(msec / inv_freq); +} + +double_t timer::get_freq() { + std::unique_lock u_lock(freq_mtx); + return freq; +} + +void timer::set_freq(double_t freq) { + std::unique_lock u_lock(freq_mtx); + this->freq = freq; +} + +double_t timer::get_freq_hist() { + std::unique_lock u_lock(freq_hist_mtx); + return freq_hist; +} + +double_t timer::get_freq_ratio() { + double_t freq_ratio = 0.0; + { + std::unique_lock u_lock(freq_mtx); + freq_ratio = freq; + } + + { + std::unique_lock u_lock(freq_hist_mtx); + freq_ratio /= freq_hist; + } + return freq_ratio; +} + +void timer::reset() { + if (!QueryPerformanceCounter(&curr_time)) + curr_time.QuadPart = 0; + if (!QueryPerformanceCounter(&prev_time)) + prev_time.QuadPart = 0; +} + +void timer::sleep(double_t msec) { + wait_timer.sleep_float(msec); +} diff --git a/src/KKdLib/timer.hpp b/src/KKdLib/timer.hpp new file mode 100644 index 0000000..b571111 --- /dev/null +++ b/src/KKdLib/timer.hpp @@ -0,0 +1,37 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "default.hpp" +#include "waitable_timer.hpp" +#include + +#define HISTORY_COUNT 0x08 + +struct timer { + double_t history[HISTORY_COUNT]; + uint8_t history_counter; + LARGE_INTEGER curr_time; + LARGE_INTEGER prev_time; + double_t inv_freq; + double_t freq; + double_t freq_hist; + std::mutex freq_mtx; + std::mutex freq_hist_mtx; + waitable_timer wait_timer; + + timer(double_t freq); + ~timer(); + + void start_of_cycle(); + void end_of_cycle(); + double_t get_freq(); + void set_freq(double_t freq); + double_t get_freq_hist(); + double_t get_freq_ratio(); + void reset(); + void sleep(double_t msec); +}; diff --git a/src/KKdLib/txp.cpp b/src/KKdLib/txp.cpp new file mode 100644 index 0000000..36b41f8 --- /dev/null +++ b/src/KKdLib/txp.cpp @@ -0,0 +1,429 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "txp.hpp" +#include "f2/struct.hpp" +#include "io/memory_stream.hpp" + +txp_mipmap::txp_mipmap() : width(), height(), format(), size() { + +} + +txp_mipmap::~txp_mipmap() { + +} + +uint32_t txp_mipmap::get_size() { + return txp::get_size(format, width, height); +} + +txp::txp() : has_cube_map(), array_size(), mipmaps_count() { + +} + +txp::~txp() { + +} + +uint32_t txp::get_size(txp_format format, uint32_t width, uint32_t height) { + uint32_t size = width * height; + switch (format) { + case TXP_A8: + return size; + case TXP_RGB8: + return size * 3; + case TXP_RGBA8: + return size * 4; + case TXP_RGB5: + return size * 2; + case TXP_RGB5A1: + return size * 2; + case TXP_RGBA4: + return size * 2; + case TXP_L8: + return size; + case TXP_L8A8: + return size * 2; + case TXP_BC1: + case TXP_BC1a: + case TXP_BC2: + case TXP_BC3: + case TXP_BC4: + case TXP_BC5: + width = align_val(width, 4); + height = align_val(height, 4); + size = width * height; + switch (format) { + case TXP_BC1: + return size / 2; + case TXP_BC1a: + return size / 2; + case TXP_BC2: + return size; + case TXP_BC3: + return size; + case TXP_BC4: + return size / 2; + case TXP_BC5: + return size; + } + break; + } + return 0; +} + +txp_set::txp_set() { + +} + +txp_set::~txp_set() { + +} + +bool txp_set::pack_file(void** data, size_t* size, bool big_endian) { + size_t l; + txp* tex; + txp_mipmap* tex_mipmap; + + if (!data || !size) + return false; + + *data = 0; + *size = 0; + + size_t count = textures.size(); + if (count < 1) + return false; + + size_t* txp4_offset = force_malloc(count); + size_t** txp2_offset = force_malloc(count); + tex = textures.data(); + for (size_t i = 0; i < count; i++, tex++) + txp2_offset[i] = force_malloc((size_t)tex->mipmaps_count * tex->array_size); + + l = 12 + count * 4; + + tex = textures.data(); + for (size_t i = 0; i < count; i++, tex++) { + txp4_offset[i] = l; + l += 12 + (size_t)tex->array_size * tex->mipmaps_count * 4; + + tex_mipmap = tex->mipmaps.data(); + for (size_t j = 0; j < tex->array_size; j++) { + for (size_t k = 0; k < tex->mipmaps_count; k++, tex_mipmap++) { + txp2_offset[i][j * tex->mipmaps_count + k] = l; + l += 24; + l += tex_mipmap->size; + } + } + } + + memory_stream s; + s.open(0, l); + s.big_endian = big_endian; + s.write_uint32_t_reverse_endianness(0x03505854); + s.write_uint32_t_reverse_endianness((uint32_t)count); + s.write_uint32_t_reverse_endianness((uint8_t)count | 0x01010100); + for (size_t i = 0; i < count; i++) + s.write_uint32_t_reverse_endianness((uint32_t)txp4_offset[i]); + + tex = textures.data(); + for (size_t i = 0; i < count; i++, tex++) { + s.set_position(txp4_offset[i], SEEK_SET); + s.write_uint32_t_reverse_endianness(tex->array_size > 1 ? 0x05505854 : 0x04505854); + s.write_uint32_t_reverse_endianness(tex->mipmaps_count * tex->array_size); + s.write_uint32_t_reverse_endianness((uint8_t)tex->mipmaps_count + | ((uint8_t)tex->array_size << 8) | 0x01010000); + for (size_t j = 0; j < tex->array_size; j++) + for (size_t k = 0; k < tex->mipmaps_count; k++) + s.write_uint32_t_reverse_endianness( + (uint32_t)(txp2_offset[i][j * tex->mipmaps_count + k] - txp4_offset[i])); + + tex_mipmap = tex->mipmaps.data(); + for (size_t j = 0; j < tex->array_size; j++) + for (size_t k = 0; k < tex->mipmaps_count; k++, tex_mipmap++) { + s.set_position(txp2_offset[i][j * tex->mipmaps_count + k], SEEK_SET); + s.write_uint32_t_reverse_endianness(0x02505854); + s.write_uint32_t_reverse_endianness(tex_mipmap->width); + s.write_uint32_t_reverse_endianness(tex_mipmap->height); + s.write_uint32_t_reverse_endianness(tex_mipmap->format); + s.write_uint32_t_reverse_endianness((uint32_t)(j * tex->mipmaps_count + k)); + s.write_uint32_t_reverse_endianness(tex_mipmap->size); + s.write(tex_mipmap->data.data(), tex_mipmap->size); + s.align_write(0x04); + } + } + s.set_position(0, SEEK_END); + + s.align_write(0x10); + s.copy(data, size); + + for (size_t i = 0; i < count; i++) + free_def(txp2_offset[i]); + free_def(txp2_offset); + free_def(txp4_offset); + return true; +} + +bool txp_set::pack_file(std::vector& data, bool big_endian) { + size_t l; + txp* tex; + txp_mipmap* tex_mipmap; + + data.clear(); + data.shrink_to_fit(); + + size_t count = textures.size(); + if (count < 1) + return false; + + size_t* txp4_offset = force_malloc(count); + size_t** txp2_offset = force_malloc(count); + tex = textures.data(); + for (size_t i = 0; i < count; i++, tex++) + txp2_offset[i] = force_malloc((size_t)tex->mipmaps_count * tex->array_size); + + l = 12 + count * 4; + + tex = textures.data(); + for (size_t i = 0; i < count; i++, tex++) { + txp4_offset[i] = l; + l += 12 + (size_t)tex->array_size * tex->mipmaps_count * 4; + + tex_mipmap = tex->mipmaps.data(); + for (size_t j = 0; j < tex->array_size; j++) { + for (size_t k = 0; k < tex->mipmaps_count; k++, tex_mipmap++) { + txp2_offset[i][j * tex->mipmaps_count + k] = l; + l += 24; + l += tex_mipmap->size; + } + } + } + + memory_stream s; + s.open(0, l); + s.big_endian = big_endian; + s.write_uint32_t_reverse_endianness(0x03505854); + s.write_uint32_t_reverse_endianness((uint32_t)count); + s.write_uint32_t_reverse_endianness((uint8_t)count | 0x01010100); + for (size_t i = 0; i < count; i++) + s.write_uint32_t_reverse_endianness((uint32_t)txp4_offset[i]); + + tex = textures.data(); + for (size_t i = 0; i < count; i++, tex++) { + s.set_position(txp4_offset[i], SEEK_SET); + s.write_uint32_t_reverse_endianness(tex->array_size > 1 ? 0x05505854 : 0x04505854); + s.write_uint32_t_reverse_endianness(tex->mipmaps_count * tex->array_size); + s.write_uint32_t_reverse_endianness((uint8_t)tex->mipmaps_count + | ((uint8_t)tex->array_size << 8) | 0x01010000); + for (size_t j = 0; j < tex->array_size; j++) + for (size_t k = 0; k < tex->mipmaps_count; k++) + s.write_uint32_t_reverse_endianness( + (uint32_t)(txp2_offset[i][j * tex->mipmaps_count + k] - txp4_offset[i])); + + tex_mipmap = tex->mipmaps.data(); + for (size_t j = 0; j < tex->array_size; j++) + for (size_t k = 0; k < tex->mipmaps_count; k++, tex_mipmap++) { + s.set_position(txp2_offset[i][j * tex->mipmaps_count + k], SEEK_SET); + s.write_uint32_t_reverse_endianness(0x02505854); + s.write_uint32_t_reverse_endianness(tex_mipmap->width); + s.write_uint32_t_reverse_endianness(tex_mipmap->height); + s.write_uint32_t_reverse_endianness(tex_mipmap->format); + s.write_uint32_t_reverse_endianness((uint32_t)(j * tex->mipmaps_count + k)); + s.write_uint32_t_reverse_endianness(tex_mipmap->size); + s.write(tex_mipmap->data.data(), tex_mipmap->size); + s.align_write(0x04); + } + } + s.set_position(0, SEEK_END); + + s.align_write(0x10); + s.copy(data); + + for (size_t i = 0; i < count; i++) + free_def(txp2_offset[i]); + free_def(txp2_offset); + free_def(txp4_offset); + return true; +} + +bool txp_set::pack_file_modern(void** data, size_t* size, bool big_endian, uint32_t signature) { + f2_struct st; + if (!pack_file(st.data, big_endian)) { + *data = 0; + *size = 0; + return false; + } + + produce_enrs(&st.enrs); + + st.header.signature = reverse_endianness_uint32_t(signature); + st.header.length = 0x20; + st.header.use_big_endian = big_endian; + st.header.use_section_size = true; + st.write(data, size, true, false); + return true; +} + +bool txp_set::produce_enrs(enrs* enrs) { + size_t l; + txp* tex; + txp_mipmap* tex_mipmap; + + if (!enrs) + return false; + + enrs->vec.clear(); + l = 0; + + size_t count = textures.size(); + if (count < 1) + return false; + + uint32_t o; + enrs_entry ee; + + ee = { 0, 1, 12, 1 }; + ee.append(0, 3, ENRS_DWORD); + enrs->vec.push_back(ee); + l += o = 12; + + ee = { o, 1, (uint32_t)(count * 4), 1 }; + ee.append(0, (uint32_t)count, ENRS_DWORD); + enrs->vec.push_back(ee); + l += (size_t)(o = (uint32_t)(count * 4ULL)); + + tex = textures.data(); + for (size_t i = 0; i < count; i++, tex++) { + ee = { o, 1, 12, 1 }; + ee.append(0, 3, ENRS_DWORD); + enrs->vec.push_back(ee); + l += o = 12; + + ee = { o, 1, tex->array_size * 4, tex->mipmaps_count }; + ee.append(0, tex->array_size, ENRS_DWORD); + enrs->vec.push_back(ee); + l += (size_t)(o = (uint32_t)((size_t)tex->array_size * tex->mipmaps_count * 4)); + + tex_mipmap = tex->mipmaps.data(); + for (size_t j = 0; j < tex->array_size; j++) { + for (size_t k = 0; k < tex->mipmaps_count; k++, tex_mipmap++) { + ee = { o, 1, 24, 1 }; + ee.append(0, 6, ENRS_DWORD); + enrs->vec.push_back(ee); + l += (size_t)(o = (uint32_t)(24 + tex_mipmap->size)); + } + } + } + return true; +} + +bool txp_set::unpack_file(const void* data, bool big_endian) { + uint32_t signature; + uint32_t tex_count; + txp* tex; + txp_mipmap* tex_mipmap; + size_t set_d; + size_t d; + size_t mipmap_d; + uint32_t sub_tex_count; + uint32_t info; + + if (!data) + return false; + + if (big_endian) + signature = load_reverse_endianness_uint32_t((void*)data); + else + signature = *(uint32_t*)data; + + if (signature != 0x03505854) + return false; + + set_d = (size_t)data; + if (big_endian) + tex_count = load_reverse_endianness_uint32_t((void*)(set_d + 4)); + else + tex_count = *(uint32_t*)(set_d + 4); + + textures.resize(tex_count); + for (size_t i = 0; i < tex_count; i++) { + if (big_endian) { + d = set_d + (size_t)load_reverse_endianness_uint32_t((uint32_t*)(set_d + 12) + i); + signature = load_reverse_endianness_uint32_t((void*)d); + } + else { + d = set_d + (size_t)((uint32_t*)(set_d + 12))[i]; + signature = *(uint32_t*)d; + } + + if (signature != 0x04505854 && signature != 0x05505854) { + textures.pop_back(); + continue; + } + + if (big_endian) { + sub_tex_count = load_reverse_endianness_uint32_t((void*)(d + 4)); + info = load_reverse_endianness_uint32_t((void*)(d + 8)); + } + else { + sub_tex_count = *(uint32_t*)(d + 4); + info = *(uint32_t*)(d + 8); + } + + tex = &textures[i - (tex_count - textures.size())]; + tex->has_cube_map = signature == 0x05505854; + tex->mipmaps_count = info & 0xFF; + tex->array_size = (info >> 8) & 0xFF; + + if (tex->array_size == 1 && tex->mipmaps_count != sub_tex_count) + tex->mipmaps_count = sub_tex_count & 0xFF; + + uint32_t mipmaps_count = tex->mipmaps_count; + tex->mipmaps.resize((size_t)tex->array_size * tex->mipmaps_count); + tex_mipmap = tex->mipmaps.data(); + for (size_t j = 0; j < tex->array_size; j++) + for (size_t k = 0; k < tex->mipmaps_count; k++, tex_mipmap++) { + if (big_endian) { + mipmap_d = d + (size_t)load_reverse_endianness_uint32_t((uint32_t*)(d + 12) + j * mipmaps_count + k); + signature = load_reverse_endianness_uint32_t((void*)mipmap_d); + } + else { + mipmap_d = d + (size_t)((uint32_t*)(d + 12))[j * mipmaps_count + k]; + signature = *(uint32_t*)mipmap_d; + } + + if (big_endian) { + tex_mipmap->width = load_reverse_endianness_uint32_t((void*)(mipmap_d + 4)); + tex_mipmap->height = load_reverse_endianness_uint32_t((void*)(mipmap_d + 8)); + tex_mipmap->format = (txp_format)load_reverse_endianness_uint32_t((void*)(mipmap_d + 12)); + tex_mipmap->size = load_reverse_endianness_uint32_t((void*)(mipmap_d + 20)); + } + else { + tex_mipmap->width = *(uint32_t*)(mipmap_d + 4); + tex_mipmap->height = *(uint32_t*)(mipmap_d + 8); + tex_mipmap->format = (txp_format)*(uint32_t*)(mipmap_d + 12); + tex_mipmap->size = *(uint32_t*)(mipmap_d + 20); + } + + ssize_t size = tex_mipmap->get_size(); + tex_mipmap->data.resize(max_def(size, tex_mipmap->size)); + memcpy(tex_mipmap->data.data(), (void*)(mipmap_d + 24), tex_mipmap->size); + size -= tex_mipmap->size; + if (size > 0) + memset((void*)((size_t)tex_mipmap->data.data() + tex_mipmap->size), 0, size); + } + } + return true; +} + +bool txp_set::unpack_file_modern(const void* data, size_t size, uint32_t signature) { + bool ret = false; + f2_struct st; + st.read(data, size); + if (st.header.signature == reverse_endianness_uint32_t(signature)) + ret = unpack_file(st.data.data(), st.header.use_big_endian); + return ret; +} diff --git a/src/KKdLib/txp.hpp b/src/KKdLib/txp.hpp new file mode 100644 index 0000000..7be18ca --- /dev/null +++ b/src/KKdLib/txp.hpp @@ -0,0 +1,66 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include +#include "default.hpp" +#include "f2/enrs.hpp" + +enum txp_format { + TXP_A8 = 0, + TXP_RGB8 = 1, + TXP_RGBA8 = 2, + TXP_RGB5 = 3, + TXP_RGB5A1 = 4, + TXP_RGBA4 = 5, + TXP_BC1 = 6, + TXP_BC1a = 7, + TXP_BC2 = 8, + TXP_BC3 = 9, + TXP_BC4 = 10, + TXP_BC5 = 11, + TXP_L8 = 12, + TXP_L8A8 = 13, +}; + +struct txp_mipmap { + uint32_t width; + uint32_t height; + txp_format format; + uint32_t size; + std::vector data; + + txp_mipmap(); + ~txp_mipmap(); + + uint32_t get_size(); +}; + +struct txp { + bool has_cube_map; + uint32_t array_size; + uint32_t mipmaps_count; + std::vector mipmaps; + + txp(); + ~txp(); + + static uint32_t get_size(txp_format format, uint32_t width, uint32_t height); +}; + +struct txp_set { + std::vector textures; + + txp_set(); + ~txp_set(); + + bool pack_file(void** data, size_t* size, bool big_endian); + bool pack_file(std::vector& data, bool big_endian); + bool pack_file_modern(void** data, size_t* size, bool big_endian, uint32_t signature); + bool produce_enrs(enrs* enrs); + bool unpack_file(const void* data, bool big_endian); + bool unpack_file_modern(const void* data, size_t size, uint32_t signature); +}; diff --git a/src/KKdLib/types.hpp b/src/KKdLib/types.hpp new file mode 100644 index 0000000..7e089ea --- /dev/null +++ b/src/KKdLib/types.hpp @@ -0,0 +1,32 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include +#include +#include + +#define FASTCALL __fastcall +#define ALIGN(n) __declspec(align(n)) + +#ifdef ssize_t +#undef ssize_t +#endif +#ifdef _WIN64 +typedef __int64 ssize_t; +#else +typedef int size_t; +#endif + +#ifdef float_t +#undef float_t +#endif +typedef float float_t; + +#ifdef double_t +#undef double_t +#endif +typedef double double_t; diff --git a/src/KKdLib/vec.cpp b/src/KKdLib/vec.cpp new file mode 100644 index 0000000..f499c76 --- /dev/null +++ b/src/KKdLib/vec.cpp @@ -0,0 +1,31 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#include "vec.hpp" + +const __m128 vec2_neg = { -0.0f, -0.0f, 0.0f, 0.0f }; +const __m128 vec3_neg = { -0.0f, -0.0f, -0.0f, 0.0f }; +const __m128 vec4_neg = { -0.0f, -0.0f, -0.0f, -0.0f }; + +const __m128i vec2i_abs = { + (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F, + (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F, + (char)0x00, (char)0x00, (char)0x00, (char)0x00, + (char)0x00, (char)0x00, (char)0x00, (char)0x00, +}; + +const __m128i vec3i_abs = { + (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F, + (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F, + (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F, + (char)0x00, (char)0x00, (char)0x00, (char)0x00, +}; + +const __m128i vec4i_abs = { + (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F, + (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F, + (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F, + (char)0xFF, (char)0xFF, (char)0xFF, (char)0x7F, +}; diff --git a/src/KKdLib/vec.hpp b/src/KKdLib/vec.hpp new file mode 100644 index 0000000..0563943 --- /dev/null +++ b/src/KKdLib/vec.hpp @@ -0,0 +1,2021 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "default.hpp" +#include "half_t.hpp" +#include +#include +#include +#include + +struct vec2i8 { + int8_t x; + int8_t y; + + inline vec2i8() : x(), y() { + + } + + inline vec2i8(int8_t value) : x(value), y(value) { + + } + + inline vec2i8(int8_t x, int8_t y) : x(x), y(y) { + + } +}; + +struct vec3i8 { + int8_t x; + int8_t y; + int8_t z; + + inline vec3i8() : x(), y(), z() { + + } + + inline vec3i8(int8_t value) : x(value), y(value), z(value) { + + } + + inline vec3i8(int8_t x, int8_t y, int8_t z) : x(x), y(y), z(z) { + + } +}; + +struct vec4i8 { + int8_t x; + int8_t y; + int8_t z; + int8_t w; + + inline vec4i8() : x(), y(), z(), w() { + + } + + inline vec4i8(int8_t value) : x(value), y(value), z(value), w(value) { + + } + + inline vec4i8(int8_t x, int8_t y, int8_t z, int8_t w) : x(x), y(y), z(z), w(w) { + + } +}; + +struct vec2u8 { + uint8_t x; + uint8_t y; + + inline vec2u8() : x(), y() { + + } + + inline vec2u8(uint8_t value) : x(value), y(value) { + + } + + inline vec2u8(uint8_t x, uint8_t y) : x(x), y(y) { + + } +}; + +struct vec3u8 { + uint8_t x; + uint8_t y; + uint8_t z; + + inline vec3u8() : x(), y(), z() { + + } + + inline vec3u8(uint8_t value) : x(value), y(value), z(value) { + + } + + inline vec3u8(uint8_t x, uint8_t y, uint8_t z) : x(x), y(y), z(z) { + + } +}; + +struct vec4u8 { + uint8_t x; + uint8_t y; + uint8_t z; + uint8_t w; + + inline vec4u8() : x(), y(), z(), w() { + + } + + inline vec4u8(uint8_t value) : x(value), y(value), z(value), w(value) { + + } + + inline vec4u8(uint8_t x, uint8_t y, uint8_t z, uint8_t w) : x(x), y(y), z(z), w(w) { + + } +}; + +struct vec2i16 { + int16_t x; + int16_t y; + + inline vec2i16() : x(), y() { + + } + + inline vec2i16(int16_t value) : x(value), y(value) { + + } + + inline vec2i16(int16_t x, int16_t y) : x(x), y(y) { + + } +}; + +struct vec3i16 { + int16_t x; + int16_t y; + int16_t z; + + inline vec3i16() : x(), y(), z() { + + } + + inline vec3i16(int16_t value) : x(value), y(value), z(value) { + + } + + inline vec3i16(int16_t x, int16_t y, int16_t z) : x(x), y(y), z(z) { + + } +}; + +struct vec4i16 { + int16_t x; + int16_t y; + int16_t z; + int16_t w; + + inline vec4i16() : x(), y(), z(), w() { + + } + + inline vec4i16(int16_t value) : x(value), y(value), z(value), w(value) { + + } + + inline vec4i16(int16_t x, int16_t y, int16_t z, int16_t w) : x(x), y(y), z(z), w(w) { + + } +}; + +struct vec2u16 { + uint16_t x; + uint16_t y; + + inline vec2u16() : x(), y() { + + } + + inline vec2u16(uint16_t value) : x(value), y(value) { + + } + + inline vec2u16(uint16_t x, uint16_t y) : x(x), y(y) { + + } +}; + +struct vec3u16 { + uint16_t x; + uint16_t y; + uint16_t z; + + inline vec3u16() : x(), y(), z() { + + } + + inline vec3u16(uint16_t value) : x(value), y(value), z(value) { + + } + + inline vec3u16(uint16_t x, uint16_t y, uint16_t z) : x(x), y(y), z(z) { + + } +}; + +struct vec4u16 { + uint16_t x; + uint16_t y; + uint16_t z; + uint16_t w; + + inline vec4u16() : x(), y(), z(), w() { + + } + + inline vec4u16(uint16_t value) : x(value), y(value), z(value), w(value) { + + } + + inline vec4u16(uint16_t x, uint16_t y, uint16_t z, uint16_t w) : x(x), y(y), z(z), w(w) { + + } +}; + +struct vec2h { + half_t x; + half_t y; + + inline vec2h() : x(), y() { + + } + + inline vec2h(half_t value) : x(value), y(value) { + + } + + inline vec2h(half_t x, half_t y) : x(x), y(y) { + + } +}; + +struct vec3h { + half_t x; + half_t y; + half_t z; + + inline vec3h() : x(), y(), z() { + + } + + inline vec3h(half_t value) : x(value), y(value), z(value) { + + } + + inline vec3h(half_t x, half_t y, half_t z) : x(x), y(y), z(z) { + + } +}; + +struct vec4h { + half_t x; + half_t y; + half_t z; + half_t w; + + inline vec4h() : x(), y(), z(), w() { + + } + + inline vec4h(half_t value) : x(value), y(value), z(value), w(value) { + + } + + inline vec4h(half_t x, half_t y, half_t z, half_t w) : x(x), y(y), z(z), w(w) { + + } +}; + +struct vec2 { + float_t x; + float_t y; + + vec2(); + vec2(float_t value); + vec2(float_t x, float_t y); + + static __m128 load_xmm(const float_t data); + static __m128 load_xmm(const vec2& data); + static __m128 load_xmm(const vec2&& data); + static vec2 store_xmm(const __m128& data); + static vec2 store_xmm(const __m128&& data); + + static float_t angle(const vec2& left, const vec2& right); + static float_t dot(const vec2& left, const vec2& right); + static float_t length(const vec2& left); + static float_t length_squared(const vec2& left); + static float_t distance(const vec2& left, const vec2& right); + static float_t distance_squared(const vec2& left, const vec2& right); + static vec2 abs(const vec2& left); + static vec2 lerp(const vec2& left, const vec2& right, const vec2& blend); + static vec2 lerp(const vec2& left, const vec2& right, const float_t blend); + static vec2 normalize(const vec2& left); + static vec2 rcp(const vec2& left); + static vec2 min(const vec2& left, const vec2& right); + static vec2 max(const vec2& left, const vec2& right); + static vec2 clamp(const vec2& left, const vec2& min, const vec2& max); + static vec2 clamp(const vec2& left, const float_t min, const float_t max); + static vec2 mult_min_max(const vec2& left, const vec2& min, const vec2& max); + static vec2 mult_min_max(const vec2& left, const float_t min, const float_t max); + static vec2 div_min_max(const vec2& left, const vec2& min, const vec2& max); + static vec2 div_min_max(const vec2& left, const float_t min, const float_t max); +}; + +struct vec3 { + float_t x; + float_t y; + float_t z; + + vec3(); + vec3(float_t value); + vec3(float_t x, float_t y, float_t z); + + static __m128 load_xmm(const float_t data); + static __m128 load_xmm(const vec3& data); + static __m128 load_xmm(const vec3&& data); + static vec3 store_xmm(const __m128& data); + static vec3 store_xmm(const __m128&& data); + + static float_t angle(const vec3& left, const vec3& right); + static float_t dot(const vec3& left, const vec3& right); + static float_t length(const vec3& left); + static float_t length_squared(const vec3& left); + static float_t distance(const vec3& left, const vec3& right); + static float_t distance_squared(const vec3& left, const vec3& right); + static vec3 abs(const vec3& left); + static vec3 lerp(const vec3& left, const vec3& right, const vec3& blend); + static vec3 lerp(const vec3& left, const vec3& right, const float_t blend); + static vec3 normalize(const vec3& left); + static vec3 rcp(const vec3& left); + static vec3 min(const vec3& left, const vec3& right); + static vec3 max(const vec3& left, const vec3& right); + static vec3 clamp(const vec3& left, const vec3& min, const vec3& max); + static vec3 clamp(const vec3& left, const float_t min, const float_t max); + static vec3 mult_min_max(const vec3& left, const vec3& min, const vec3& max); + static vec3 mult_min_max(const vec3& left, const float_t min, const float_t max); + static vec3 div_min_max(const vec3& left, const vec3& min, const vec3& max); + static vec3 div_min_max(const vec3& left, const float_t min, const float_t max); + static vec3 cross(const vec3& left, const vec3& right); +}; + +struct vec4 { + float_t x; + float_t y; + float_t z; + float_t w; + + vec4(); + vec4(float_t value); + vec4(float_t x, float_t y, float_t z, float_t w); + + static __m128 load_xmm(const float_t data); + static __m128 load_xmm(const vec4& data); + static __m128 load_xmm(const vec4&& data); + static vec4 store_xmm(const __m128& data); + static vec4 store_xmm(const __m128&& data); + + static float_t angle(const vec4& left, const vec4& right); + static float_t dot(const vec4& left, const vec4& right); + static float_t length(const vec4& left); + static float_t length_squared(const vec4& left); + static float_t distance(const vec4& left, const vec4& right); + static float_t distance_squared(const vec4& left, const vec4& right); + static vec4 abs(const vec4& left); + static vec4 lerp(const vec4& left, const vec4& right, const vec4& blend); + static vec4 lerp(const vec4& left, const vec4& right, const float_t blend); + static vec4 normalize(const vec4& left); + static vec4 rcp(const vec4& left); + static vec4 min(const vec4& min, const vec4& max); + static vec4 max(const vec4& min, const vec4& max); + static vec4 clamp(const vec4& left, const vec4& min, const vec4& max); + static vec4 clamp(const vec4& left, const float_t min, const float_t max); + static vec4 mult_min_max(const vec4& left, const vec4& min, const vec4& max); + static vec4 mult_min_max(const vec4& left, const float_t min, const float_t max); + static vec4 div_min_max(const vec4& left, const vec4& min, const vec4& max); + static vec4 div_min_max(const vec4& left, const float_t min, const float_t max); +}; + +struct vec2i { + int32_t x; + int32_t y; + + vec2i(); + vec2i(int32_t value); + vec2i(int32_t x, int32_t y); + + static __m128i load_xmm(const int32_t data); + static __m128i load_xmm(const vec2i& data); + static __m128i load_xmm(const vec2i&& data); + static vec2i store_xmm(const __m128i& data); + static vec2i store_xmm(const __m128i&& data); + + static vec2i min(const vec2i& left, const vec2i& right); + static vec2i max(const vec2i& left, const vec2i& right); + static vec2i clamp(const vec2i& left, const vec2i& min, const vec2i& max); + static vec2i clamp(const vec2i& left, const int32_t min, const int32_t max); +}; + +struct vec3i { + int32_t x; + int32_t y; + int32_t z; + + vec3i(); + vec3i(int32_t value); + vec3i(int32_t x, int32_t y, int32_t z); + + static __m128i load_xmm(const int32_t data); + static __m128i load_xmm(const vec3i& data); + static __m128i load_xmm(const vec3i&& data); + static vec3i store_xmm(const __m128i& data); + static vec3i store_xmm(const __m128i&& data); + + static vec3i min(const vec3i& left, const vec3i& right); + static vec3i max(const vec3i& left, const vec3i& right); + static vec3i clamp(const vec3i& left, const vec3i& min, const vec3i& max); + static vec3i clamp(const vec3i& left, const int32_t min, const int32_t max); +}; + +struct vec4i { + int32_t x; + int32_t y; + int32_t z; + int32_t w; + + vec4i(); + vec4i(int32_t value); + vec4i(int32_t x, int32_t y, int32_t z, int32_t w); + + static __m128i load_xmm(const int32_t data); + static __m128i load_xmm(const vec4i& data); + static __m128i load_xmm(const vec4i&& data); + static vec4i store_xmm(const __m128i& data); + static vec4i store_xmm(const __m128i&& data); + + static vec4i min(const vec4i& left, const vec4i& right); + static vec4i max(const vec4i& left, const vec4i& right); + static vec4i clamp(const vec4i& left, const vec4i& min, const vec4i& max); + static vec4i clamp(const vec4i& left, const int32_t min, const int32_t max); +}; + +extern const __m128 vec2_neg; +extern const __m128 vec3_neg; +extern const __m128 vec4_neg; + +extern const __m128i vec2i_abs; +extern const __m128i vec3i_abs; +extern const __m128i vec4i_abs; + +inline vec2::vec2() : x(), y() { + +} + +inline vec2::vec2(float_t value) : x(value), y(value) { + +} + +inline vec2::vec2(float_t x, float_t y) : x(x), y(y) { + +} + +inline __m128 vec2::load_xmm(const float_t data) { + __m128 _data = _mm_set_ss(data); + return _mm_shuffle_ps(_data, _data, 0x50); +} + +inline __m128 vec2::load_xmm(const vec2& data) { + return _mm_castsi128_ps(_mm_loadl_epi64((const __m128i*) & data)); +} + +inline __m128 vec2::load_xmm(const vec2&& data) { + return _mm_castsi128_ps(_mm_loadl_epi64((const __m128i*) & data)); +} + +inline vec2 vec2::store_xmm(const __m128& data) { + vec2 _data; + _mm_storel_epi64((__m128i*) & _data, _mm_castps_si128(data)); + return _data; +} + +inline vec2 vec2::store_xmm(const __m128&& data) { + vec2 _data; + _mm_storel_epi64((__m128i*) & _data, _mm_castps_si128(data)); + return _data; +} + +inline vec2 operator +(const vec2& left, const vec2& right) { + return vec2::store_xmm(_mm_add_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator +(const vec2& left, const float_t right) { + return vec2::store_xmm(_mm_add_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator +(const float_t left, const vec2& right) { + return vec2::store_xmm(_mm_add_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline void operator +=(vec2& left, const vec2& right) { + left = vec2::store_xmm(_mm_add_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline void operator +=(vec2& left, const float_t right) { + left = vec2::store_xmm(_mm_add_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator -(const vec2& left, const vec2& right) { + return vec2::store_xmm(_mm_sub_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator -(const vec2& left, const float_t right) { + return vec2::store_xmm(_mm_sub_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator -(const float_t left, const vec2& right) { + return vec2::store_xmm(_mm_sub_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline void operator -=(vec2& left, const vec2& right) { + left = vec2::store_xmm(_mm_sub_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline void operator -=(vec2& left, const float_t right) { + left = vec2::store_xmm(_mm_sub_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator *(const vec2& left, const vec2& right) { + return vec2::store_xmm(_mm_mul_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator *(const vec2& left, const float_t right) { + return vec2::store_xmm(_mm_mul_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator *(const float_t left, const vec2& right) { + return vec2::store_xmm(_mm_mul_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline void operator *=(vec2& left, const vec2& right) { + left = vec2::store_xmm(_mm_mul_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline void operator *=(vec2& left, const float_t right) { + left = vec2::store_xmm(_mm_mul_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator /(const vec2& left, const vec2& right) { + return vec2::store_xmm(_mm_div_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator /(const vec2& left, const float_t right) { + return vec2::store_xmm(_mm_div_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator /(const float_t left, const vec2& right) { + return vec2::store_xmm(_mm_div_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline void operator /=(vec2& left, const vec2& right) { + left = vec2::store_xmm(_mm_div_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline void operator /=(vec2& left, const float_t right) { + left = vec2::store_xmm(_mm_div_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator &(const vec2& left, const vec2& right) { + return vec2::store_xmm(_mm_and_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator &(const vec2& left, const float_t right) { + return vec2::store_xmm(_mm_and_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator &(const float_t left, const vec2& right) { + return vec2::store_xmm(_mm_and_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline void operator &=(vec2& left, const vec2& right) { + left = vec2::store_xmm(_mm_and_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline void operator &=(vec2& left, const float_t right) { + left = vec2::store_xmm(_mm_and_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator ^(const vec2& left, const vec2& right) { + return vec2::store_xmm(_mm_xor_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator ^(const vec2& left, const float_t right) { + return vec2::store_xmm(_mm_xor_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator ^(const float_t left, const vec2& right) { + return vec2::store_xmm(_mm_xor_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline void operator ^=(vec2& left, const vec2& right) { + left = vec2::store_xmm(_mm_xor_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline void operator ^=(vec2& left, const float_t right) { + left = vec2::store_xmm(_mm_xor_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 operator -(const vec2& left) { + return vec2::store_xmm(_mm_xor_ps(vec2::load_xmm(left), vec2_neg)); +} + +inline bool operator ==(const vec2& left, const vec2& right) { + return !memcmp(&left, &right, sizeof(vec2)); +} + +inline bool operator !=(const vec2& left, const vec2& right) { + return !!memcmp(&left, &right, sizeof(vec2)); +} + +inline float_t vec2::angle(const vec2& left, const vec2& right) { + return acosf(vec2::dot(left, right) / (vec2::length(left) * vec2::length(right))); +} + +inline float_t vec2::dot(const vec2& left, const vec2& right) { + __m128 zt; + zt = _mm_mul_ps(vec2::load_xmm(left), vec2::load_xmm(right)); + return _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline float_t vec2::length(const vec2& left) { + __m128 xt; + __m128 zt; + xt = vec2::load_xmm(left); + zt = _mm_mul_ps(xt, xt); + return _mm_cvtss_f32(_mm_sqrt_ss(_mm_hadd_ps(zt, zt))); +} + +inline float_t vec2::length_squared(const vec2& left) { + __m128 xt; + __m128 zt; + xt = vec2::load_xmm(left); + zt = _mm_mul_ps(xt, xt); + return _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline float_t vec2::distance(const vec2& left, const vec2& right) { + __m128 zt; + zt = _mm_sub_ps(vec2::load_xmm(left), vec2::load_xmm(right)); + zt = _mm_mul_ps(zt, zt); + return _mm_cvtss_f32(_mm_sqrt_ss(_mm_hadd_ps(zt, zt))); +} + +inline float_t vec2::distance_squared(const vec2& left, const vec2& right) { + __m128 zt; + zt = _mm_sub_ps(vec2::load_xmm(left), vec2::load_xmm(right)); + zt = _mm_mul_ps(zt, zt); + return _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline vec2 vec2::abs(const vec2& left) { + return vec2::store_xmm(_mm_castsi128_ps(_mm_and_si128(_mm_castps_si128(vec2::load_xmm(left)), vec2i_abs))); +} + +inline vec2 vec2::lerp(const vec2& left, const vec2& right, const vec2& blend) { + __m128 b1; + __m128 b2; + b1 = vec2::load_xmm(blend); + b2 = _mm_sub_ps(vec2::load_xmm(1.0f), b1); + return vec2::store_xmm(_mm_add_ps(_mm_mul_ps(vec2::load_xmm(left), b2), + _mm_mul_ps(vec2::load_xmm(right), b1))); +} + +inline vec2 vec2::lerp(const vec2& left, const vec2& right, const float_t blend) { + __m128 b1; + __m128 b2; + b1 = vec2::load_xmm(blend); + b2 = _mm_sub_ps(vec2::load_xmm(1.0f), b1); + return vec2::store_xmm(_mm_add_ps(_mm_mul_ps(vec2::load_xmm(left), b2), + _mm_mul_ps(vec2::load_xmm(right), b1))); +} + +inline vec2 vec2::normalize(const vec2& left) { + __m128 xt; + __m128 zt; + xt = vec2::load_xmm(left); + zt = _mm_mul_ps(xt, xt); + zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt)); + if (_mm_cvtss_f32(zt) != 0.0f) + zt = _mm_div_ss(_mm_set_ss(1.0f), zt); + return vec2::store_xmm(_mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0))); +} + +inline vec2 vec2::rcp(const vec2& left) { + return vec2::store_xmm(_mm_div_ps(vec2::load_xmm(1.0f), vec2::load_xmm(left))); +} + +inline vec2 vec2::min(const vec2& left, const vec2& right) { + return vec2::store_xmm(_mm_min_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 vec2::max(const vec2& left, const vec2& right) { + return vec2::store_xmm(_mm_max_ps(vec2::load_xmm(left), vec2::load_xmm(right))); +} + +inline vec2 vec2::clamp(const vec2& left, const vec2& min, const vec2& max) { + return vec2::store_xmm(_mm_min_ps(_mm_max_ps(vec2::load_xmm(left), + vec2::load_xmm(min)), vec2::load_xmm(max))); +} + +inline vec2 vec2::clamp(const vec2& left, const float_t min, const float_t max) { + return vec2::store_xmm(_mm_min_ps(_mm_max_ps(vec2::load_xmm(left), + vec2::load_xmm(min)), vec2::load_xmm(max))); +} + +inline vec2 vec2::mult_min_max(const vec2& left, const vec2& min, const vec2& max) { + __m128 xt; + __m128 yt; + __m128 zt; + xt = vec2::load_xmm(left); + yt = _mm_and_ps(_mm_cmplt_ps(xt, vec4::load_xmm(0.0f)), vec2::load_xmm(-min)); + zt = _mm_and_ps(_mm_cmpge_ps(xt, vec4::load_xmm(0.0f)), vec2::load_xmm(max)); + return vec2::store_xmm(_mm_mul_ps(xt, _mm_or_ps(yt, zt))); +} + +inline vec2 vec2::mult_min_max(const vec2& left, const float_t min, const float_t max) { + __m128 xt; + __m128 yt; + __m128 zt; + xt = vec2::load_xmm(left); + yt = _mm_and_ps(_mm_cmplt_ps(xt, vec4::load_xmm(0.0f)), vec2::load_xmm(-min)); + zt = _mm_and_ps(_mm_cmpge_ps(xt, vec4::load_xmm(0.0f)), vec2::load_xmm(max)); + return vec2::store_xmm(_mm_mul_ps(xt, _mm_or_ps(yt, zt))); +} + +inline vec2 vec2::div_min_max(const vec2& left, const vec2& min, const vec2& max) { + __m128 xt; + __m128 yt; + __m128 zt; + xt = vec2::load_xmm(left); + yt = _mm_and_ps(_mm_cmplt_ps(xt, vec4::load_xmm(0.0f)), vec2::load_xmm(-min)); + zt = _mm_and_ps(_mm_cmpge_ps(xt, vec4::load_xmm(0.0f)), vec2::load_xmm(max)); + return vec2::store_xmm(_mm_div_ps(xt, _mm_or_ps(yt, zt))); +} + +inline vec2 vec2::div_min_max(const vec2& left, const float_t min, const float_t max) { + __m128 xt; + __m128 yt; + __m128 zt; + xt = vec2::load_xmm(left); + yt = _mm_and_ps(_mm_cmplt_ps(xt, vec4::load_xmm(0.0f)), vec2::load_xmm(-min)); + zt = _mm_and_ps(_mm_cmpge_ps(xt, vec4::load_xmm(0.0f)), vec2::load_xmm(max)); + return vec2::store_xmm(_mm_div_ps(xt, _mm_or_ps(yt, zt))); +} + +inline vec3::vec3() : x(), y(), z() { + +} + +inline vec3::vec3(float_t value) : x(value), y(value), z(value) { + +} + +inline vec3::vec3(float_t x, float_t y, float_t z) : x(x), y(y), z(z) { + +} + +inline vec3 operator +(const vec3& left, const vec3& right) { + return vec3::store_xmm(_mm_add_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator +(const vec3& left, const float_t right) { + return vec3::store_xmm(_mm_add_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator +(const float_t left, const vec3& right) { + return vec3::store_xmm(_mm_add_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline void operator +=(vec3& left, const vec3& right) { + left = vec3::store_xmm(_mm_add_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline void operator +=(vec3& left, const float_t right) { + left = vec3::store_xmm(_mm_add_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator -(const vec3& left, const vec3& right) { + return vec3::store_xmm(_mm_sub_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator -(const vec3& left, const float_t right) { + return vec3::store_xmm(_mm_sub_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator -(const float_t left, const vec3& right) { + return vec3::store_xmm(_mm_sub_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline void operator -=(vec3& left, const vec3& right) { + left = vec3::store_xmm(_mm_sub_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline void operator -=(vec3& left, const float_t right) { + left = vec3::store_xmm(_mm_sub_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator *(const vec3& left, const vec3& right) { + return vec3::store_xmm(_mm_mul_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator *(const vec3& left, const float_t right) { + return vec3::store_xmm(_mm_mul_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator *(const float_t left, const vec3& right) { + return vec3::store_xmm(_mm_mul_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline void operator *=(vec3& left, const vec3& right) { + left = vec3::store_xmm(_mm_mul_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline void operator *=(vec3& left, const float_t right) { + left = vec3::store_xmm(_mm_mul_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator /(const vec3& left, const vec3& right) { + return vec3::store_xmm(_mm_div_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator /(const vec3& left, const float_t right) { + return vec3::store_xmm(_mm_div_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator /(const float_t left, const vec3& right) { + return vec3::store_xmm(_mm_div_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline void operator /=(vec3& left, const vec3& right) { + left = vec3::store_xmm(_mm_div_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline void operator /=(vec3& left, const float_t right) { + left = vec3::store_xmm(_mm_div_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator &(const vec3& left, const vec3& right) { + return vec3::store_xmm(_mm_and_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator &(const vec3& left, const float_t right) { + return vec3::store_xmm(_mm_and_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator &(const float_t left, const vec3& right) { + return vec3::store_xmm(_mm_and_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline void operator &=(vec3& left, const vec3& right) { + left = vec3::store_xmm(_mm_and_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline void operator &=(vec3& left, const float_t right) { + left = vec3::store_xmm(_mm_and_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator ^(const vec3& left, const vec3& right) { + return vec3::store_xmm(_mm_xor_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator ^(const vec3& left, const float_t right) { + return vec3::store_xmm(_mm_xor_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator ^(const float_t left, const vec3& right) { + return vec3::store_xmm(_mm_xor_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline void operator ^=(vec3& left, const vec3& right) { + left = vec3::store_xmm(_mm_xor_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline void operator ^=(vec3& left, const float_t right) { + left = vec3::store_xmm(_mm_xor_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 operator -(const vec3& left) { + return vec3::store_xmm(_mm_xor_ps(vec3::load_xmm(left), vec3_neg)); +} + +inline bool operator ==(const vec3& left, const vec3& right) { + return !memcmp(&left, &right, sizeof(vec3)); +} + +inline bool operator !=(const vec3& left, const vec3& right) { + return !!memcmp(&left, &right, sizeof(vec3)); +} + +inline __m128 vec3::load_xmm(const float_t data) { + __m128 _data = _mm_set_ss(data); + return _mm_shuffle_ps(_data, _data, 0x40); +} + +inline __m128 vec3::load_xmm(const vec3& data) { + return _mm_movelh_ps(_mm_castsi128_ps(_mm_loadl_epi64((const __m128i*) & data)), _mm_load_ss(&data.z)); +} + +inline __m128 vec3::load_xmm(const vec3&& data) { + return _mm_movelh_ps(_mm_castsi128_ps(_mm_loadl_epi64((const __m128i*) & data)), _mm_load_ss(&data.z)); +} + +inline vec3 vec3::store_xmm(const __m128& data) { + vec3 _data; + _mm_storel_epi64((__m128i*) & _data, _mm_castps_si128(data)); + _mm_store_ss(&_data.z, _mm_castsi128_ps(_mm_srli_si128(_mm_castps_si128(data), 8))); + return _data; +} + +inline vec3 vec3::store_xmm(const __m128&& data) { + vec3 _data; + _mm_storel_epi64((__m128i*) & _data, _mm_castps_si128(data)); + _mm_store_ss(&_data.z, _mm_castsi128_ps(_mm_srli_si128(_mm_castps_si128(data), 8))); + return _data; +} + +inline float_t vec3::angle(const vec3& left, const vec3& right) { + return acosf(vec3::dot(left, right) / (vec3::length(left) * vec3::length(right))); +} + +inline float_t vec3::dot(const vec3& left, const vec3& right) { + __m128 zt; + zt = _mm_mul_ps(vec3::load_xmm(left), vec3::load_xmm(right)); + zt = _mm_hadd_ps(zt, zt); + return _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline float_t vec3::length(const vec3& left) { + __m128 xt; + __m128 zt; + xt = vec3::load_xmm(left); + zt = _mm_mul_ps(xt, xt); + zt = _mm_hadd_ps(zt, zt); + return _mm_cvtss_f32(_mm_sqrt_ss(_mm_hadd_ps(zt, zt))); +} + +inline float_t vec3::length_squared(const vec3& left) { + __m128 xt; + __m128 zt; + xt = vec3::load_xmm(left); + zt = _mm_mul_ps(xt, xt); + zt = _mm_hadd_ps(zt, zt); + return _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline float_t vec3::distance(const vec3& left, const vec3& right) { + __m128 xt; + __m128 yt; + __m128 zt; + xt = vec3::load_xmm(left); + yt = vec3::load_xmm(right); + zt = _mm_sub_ps(xt, yt); + zt = _mm_mul_ps(zt, zt); + zt = _mm_hadd_ps(zt, zt); + return _mm_cvtss_f32(_mm_sqrt_ss(_mm_hadd_ps(zt, zt))); +} + +inline float_t vec3::distance_squared(const vec3& left, const vec3& right) { + __m128 xt; + __m128 yt; + __m128 zt; + xt = vec3::load_xmm(left); + yt = vec3::load_xmm(right); + zt = _mm_sub_ps(xt, yt); + zt = _mm_mul_ps(zt, zt); + zt = _mm_hadd_ps(zt, zt); + return _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline vec3 vec3::abs(const vec3& left) { + return vec3::store_xmm(_mm_castsi128_ps(_mm_and_si128(_mm_castps_si128(vec3::load_xmm(left)), vec3i_abs))); +} + +inline vec3 vec3::lerp(const vec3& left, const vec3& right, const vec3& blend) { + __m128 b1; + __m128 b2; + b1 = vec3::load_xmm(blend); + b2 = _mm_sub_ps(vec3::load_xmm(1.0f), b1); + return vec3::store_xmm(_mm_add_ps(_mm_mul_ps(vec3::load_xmm(left), b2), + _mm_mul_ps(vec3::load_xmm(right), b1))); +} + +inline vec3 vec3::lerp(const vec3& left, const vec3& right, const float_t blend) { + __m128 b1; + __m128 b2; + b1 = vec3::load_xmm(blend); + b2 = _mm_sub_ps(vec3::load_xmm(1.0f), b1); + return vec3::store_xmm(_mm_add_ps(_mm_mul_ps(vec3::load_xmm(left), b2), + _mm_mul_ps(vec3::load_xmm(right), b1))); +} + +inline vec3 vec3::normalize(const vec3& left) { + __m128 xt; + __m128 zt; + xt = vec3::load_xmm(left); + zt = _mm_mul_ps(xt, xt); + zt = _mm_hadd_ps(zt, zt); + zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt)); + if (_mm_cvtss_f32(zt) != 0.0f) + zt = _mm_div_ss(_mm_set_ss(1.0f), zt); + return vec3::store_xmm(_mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0))); +} + +inline vec3 vec3::rcp(const vec3& left) { + return vec3::store_xmm(_mm_div_ps(vec3::load_xmm(1.0f), vec3::load_xmm(left))); +} + +inline vec3 vec3::min(const vec3& left, const vec3& right) { + return vec3::store_xmm(_mm_min_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 vec3::max(const vec3& left, const vec3& right) { + return vec3::store_xmm(_mm_max_ps(vec3::load_xmm(left), vec3::load_xmm(right))); +} + +inline vec3 vec3::clamp(const vec3& left, const vec3& min, const vec3& max) { + return vec3::store_xmm(_mm_min_ps(_mm_max_ps(vec3::load_xmm(left), + vec3::load_xmm(min)), vec3::load_xmm(max))); +} + +inline vec3 vec3::clamp(const vec3& left, const float_t min, const float_t max) { + return vec3::store_xmm(_mm_min_ps(_mm_max_ps(vec3::load_xmm(left), + vec3::load_xmm(min)), vec3::load_xmm(max))); +} + +inline vec3 vec3::mult_min_max(const vec3& left, const vec3& min, const vec3& max) { + __m128 xt; + __m128 yt; + __m128 zt; + xt = vec3::load_xmm(left); + yt = _mm_and_ps(_mm_cmplt_ps(xt, vec4::load_xmm(0.0f)), vec3::load_xmm(-min)); + zt = _mm_and_ps(_mm_cmpge_ps(xt, vec4::load_xmm(0.0f)), vec3::load_xmm(max)); + return vec3::store_xmm(_mm_mul_ps(xt, _mm_or_ps(yt, zt))); +} + +inline vec3 vec3::mult_min_max(const vec3& left, const float_t min, const float_t max) { + __m128 xt; + __m128 yt; + __m128 zt; + xt = vec3::load_xmm(left); + yt = _mm_and_ps(_mm_cmplt_ps(xt, vec4::load_xmm(0.0f)), vec3::load_xmm(-min)); + zt = _mm_and_ps(_mm_cmpge_ps(xt, vec4::load_xmm(0.0f)), vec3::load_xmm(max)); + return vec3::store_xmm(_mm_mul_ps(xt, _mm_or_ps(yt, zt))); +} + +inline vec3 vec3::div_min_max(const vec3& left, const vec3& min, const vec3& max) { + __m128 xt; + __m128 yt; + __m128 zt; + xt = vec3::load_xmm(left); + yt = _mm_and_ps(_mm_cmplt_ps(xt, vec4::load_xmm(0.0f)), vec3::load_xmm(-min)); + zt = _mm_and_ps(_mm_cmpge_ps(xt, vec4::load_xmm(0.0f)), vec3::load_xmm(max)); + return vec3::store_xmm(_mm_div_ps(xt, _mm_or_ps(yt, zt))); +} + +inline vec3 vec3::div_min_max(const vec3& left, const float_t min, const float_t max) { + __m128 xt; + __m128 yt; + __m128 zt; + xt = vec3::load_xmm(left); + yt = _mm_and_ps(_mm_cmplt_ps(xt, vec4::load_xmm(0.0f)), vec3::load_xmm(-min)); + zt = _mm_and_ps(_mm_cmpge_ps(xt, vec4::load_xmm(0.0f)), vec3::load_xmm(max)); + return vec3::store_xmm(_mm_div_ps(xt, _mm_or_ps(yt, zt))); +} + +inline vec3 vec3::cross(const vec3& left, const vec3& right) { + __m128 xt; + __m128 yt; + __m128 zt; + xt = vec3::load_xmm(left); + yt = vec3::load_xmm(right); + zt = _mm_sub_ps( + _mm_mul_ps(xt, _mm_shuffle_ps(yt, yt, 0x09)), + _mm_mul_ps(yt, _mm_shuffle_ps(xt, xt, 0x09)) + ); + return vec3::store_xmm(_mm_shuffle_ps(zt, zt, 0x09)); +} + +inline vec4::vec4() : x(), y(), z(), w() { + +} + +inline vec4::vec4(float_t value) : x(value), y(value), z(value), w(value) { + +} + +inline vec4::vec4(float_t x, float_t y, float_t z, float_t w) : x(x), y(y), z(z), w(w) { + +} + +inline vec4 operator +(const vec4& left, const vec4& right) { + return vec4::store_xmm(_mm_add_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator +(const vec4& left, const float_t right) { + return vec4::store_xmm(_mm_add_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator +(const float_t left, const vec4& right) { + return vec4::store_xmm(_mm_add_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline void operator +=(vec4& left, const vec4& right) { + left = vec4::store_xmm(_mm_add_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline void operator +=(vec4& left, const float_t right) { + left = vec4::store_xmm(_mm_add_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator -(const vec4& left, const vec4& right) { + return vec4::store_xmm(_mm_sub_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator -(const vec4& left, const float_t right) { + return vec4::store_xmm(_mm_sub_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator -(const float_t left, const vec4& right) { + return vec4::store_xmm(_mm_sub_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline void operator -=(vec4& left, const vec4& right) { + left = vec4::store_xmm(_mm_sub_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline void operator -=(vec4& left, const float_t right) { + left = vec4::store_xmm(_mm_sub_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator *(const vec4& left, const vec4& right) { + return vec4::store_xmm(_mm_mul_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator *(const vec4& left, const float_t right) { + return vec4::store_xmm(_mm_mul_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator *(const float_t left, const vec4& right) { + return vec4::store_xmm(_mm_mul_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline void operator *=(vec4& left, const vec4& right) { + left = vec4::store_xmm(_mm_mul_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline void operator *=(vec4& left, const float_t right) { + left = vec4::store_xmm(_mm_mul_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator /(const vec4& left, const vec4& right) { + return vec4::store_xmm(_mm_div_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator /(const vec4& left, const float_t right) { + return vec4::store_xmm(_mm_div_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator /(const float_t left, const vec4& right) { + return vec4::store_xmm(_mm_div_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline void operator /=(vec4& left, const vec4& right) { + left = vec4::store_xmm(_mm_div_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline void operator /=(vec4& left, const float_t right) { + left = vec4::store_xmm(_mm_div_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator &(const vec4& left, const vec4& right) { + return vec4::store_xmm(_mm_and_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator &(const vec4& left, const float_t right) { + return vec4::store_xmm(_mm_and_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator &(const float_t left, const vec4& right) { + return vec4::store_xmm(_mm_and_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline void operator &=(vec4& left, const vec4& right) { + left = vec4::store_xmm(_mm_and_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline void operator &=(vec4& left, const float_t right) { + left = vec4::store_xmm(_mm_and_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator ^(const vec4& left, const vec4& right) { + return vec4::store_xmm(_mm_xor_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator ^(const vec4& left, const float_t right) { + return vec4::store_xmm(_mm_xor_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator ^(const float_t left, const vec4& right) { + return vec4::store_xmm(_mm_xor_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline void operator ^=(vec4& left, const vec4& right) { + left = vec4::store_xmm(_mm_xor_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline void operator ^=(vec4& left, const float_t right) { + left = vec4::store_xmm(_mm_xor_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 operator -(const vec4& left) { + return vec4::store_xmm(_mm_xor_ps(vec4::load_xmm(left), vec4_neg)); +} + +inline bool operator ==(const vec4& left, const vec4& right) { + return !memcmp(&left, &right, sizeof(vec4)); +} + +inline bool operator !=(const vec4& left, const vec4& right) { + return !!memcmp(&left, &right, sizeof(vec4)); +} + +inline __m128 vec4::load_xmm(const float_t data) { + __m128 _data = _mm_set_ss(data); + return _mm_shuffle_ps(_data, _data, 0); +} + +inline __m128 vec4::load_xmm(const vec4& data) { + return _mm_loadu_ps((const float*)&data); +} + +inline __m128 vec4::load_xmm(const vec4&& data) { + return _mm_loadu_ps((const float*)&data); +} + +inline vec4 vec4::store_xmm(const __m128& data) { + vec4 _data; + _mm_storeu_ps((float*)&_data, data); + return _data; +} + +inline vec4 vec4::store_xmm(const __m128&& data) { + vec4 _data; + _mm_storeu_ps((float*)&_data, data); + return _data; +} + +inline float_t vec4::angle(const vec4& left, const vec4& right) { + return acosf(vec4::dot(left, right) / (vec4::length(left) * vec4::length(right))); +} + +inline float_t vec4::dot(const vec4& left, const vec4& right) { + __m128 zt; + zt = _mm_mul_ps(vec4::load_xmm(left), vec4::load_xmm(right)); + zt = _mm_hadd_ps(zt, zt); + return _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline float_t vec4::length(const vec4& left) { + __m128 xt; + __m128 zt; + xt = vec4::load_xmm(left); + zt = _mm_mul_ps(xt, xt); + zt = _mm_hadd_ps(zt, zt); + return _mm_cvtss_f32(_mm_sqrt_ss(_mm_hadd_ps(zt, zt))); +} + +inline float_t vec4::length_squared(const vec4& left) { + __m128 xt; + __m128 zt; + xt = vec4::load_xmm(left); + zt = _mm_mul_ps(xt, xt); + zt = _mm_hadd_ps(zt, zt); + return _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline float_t vec4::distance(const vec4& left, const vec4& right) { + __m128 zt; + zt = _mm_sub_ps(vec4::load_xmm(left), vec4::load_xmm(right)); + zt = _mm_mul_ps(zt, zt); + zt = _mm_hadd_ps(zt, zt); + return _mm_cvtss_f32(_mm_sqrt_ss(_mm_hadd_ps(zt, zt))); +} + +inline float_t vec4::distance_squared(const vec4& left, const vec4& right) { + __m128 zt; + zt = _mm_sub_ps(vec4::load_xmm(left), vec4::load_xmm(right)); + zt = _mm_mul_ps(zt, zt); + zt = _mm_hadd_ps(zt, zt); + return _mm_cvtss_f32(_mm_hadd_ps(zt, zt)); +} + +inline vec4 vec4::abs(const vec4& left) { + return vec4::store_xmm(_mm_castsi128_ps(_mm_and_si128(_mm_castps_si128(vec4::load_xmm(left)), vec4i_abs))); +} + +inline vec4 vec4::lerp(const vec4& left, const vec4& right, const vec4& blend) { + __m128 b1; + __m128 b2; + b1 = vec4::load_xmm(blend); + b2 = _mm_sub_ps(vec4::load_xmm(1.0f), b1); + return vec4::store_xmm(_mm_add_ps(_mm_mul_ps(vec4::load_xmm(left), b2), + _mm_mul_ps(vec4::load_xmm(right), b1))); +} + +inline vec4 vec4::lerp(const vec4& left, const vec4& right, const float_t blend) { + __m128 b1; + __m128 b2; + b1 = vec4::load_xmm(blend); + b2 = _mm_sub_ps(vec4::load_xmm(1.0f), b1); + return vec4::store_xmm(_mm_add_ps(_mm_mul_ps(vec4::load_xmm(left), b2), + _mm_mul_ps(vec4::load_xmm(right), b1))); +} + +inline vec4 vec4::normalize(const vec4& left) { + __m128 xt; + __m128 zt; + xt = vec4::load_xmm(left); + zt = _mm_mul_ps(xt, xt); + zt = _mm_hadd_ps(zt, zt); + zt = _mm_sqrt_ss(_mm_hadd_ps(zt, zt)); + if (_mm_cvtss_f32(zt) != 0.0f) + zt = _mm_div_ss(_mm_set_ss(1.0f), zt); + return vec4::store_xmm(_mm_mul_ps(xt, _mm_shuffle_ps(zt, zt, 0))); +} + +inline vec4 vec4::rcp(const vec4& left) { + return vec4::store_xmm(_mm_div_ps(vec4::load_xmm(1.0f), vec4::load_xmm(left))); +} + +inline vec4 vec4::min(const vec4& left, const vec4& right) { + return vec4::store_xmm(_mm_min_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 vec4::max(const vec4& left, const vec4& right) { + return vec4::store_xmm(_mm_max_ps(vec4::load_xmm(left), vec4::load_xmm(right))); +} + +inline vec4 vec4::clamp(const vec4& left, const vec4& min, const vec4& max) { + return vec4::store_xmm(_mm_min_ps(_mm_max_ps(vec4::load_xmm(left), + vec4::load_xmm(min)), vec4::load_xmm(max))); +} + +inline vec4 vec4::clamp(const vec4& left, const float_t min, const float_t max) { + return vec4::store_xmm(_mm_min_ps(_mm_max_ps(vec4::load_xmm(left), + vec4::load_xmm(min)), vec4::load_xmm(max))); +} + +inline vec4 vec4::mult_min_max(const vec4& left, const vec4& min, const vec4& max) { + __m128 xt; + __m128 yt; + __m128 zt; + xt = vec4::load_xmm(left); + yt = _mm_and_ps(_mm_cmplt_ps(xt, vec4::load_xmm(0.0f)), vec4::load_xmm(-min)); + zt = _mm_and_ps(_mm_cmpge_ps(xt, vec4::load_xmm(0.0f)), vec4::load_xmm(max)); + return vec4::store_xmm(_mm_mul_ps(xt, _mm_or_ps(yt, zt))); +} + +inline vec4 vec4::mult_min_max(const vec4& left, const float_t min, const float_t max) { + __m128 xt; + __m128 yt; + __m128 zt; + xt = vec4::load_xmm(left); + yt = _mm_and_ps(_mm_cmplt_ps(xt, vec4::load_xmm(0.0f)), vec4::load_xmm(-min)); + zt = _mm_and_ps(_mm_cmpge_ps(xt, vec4::load_xmm(0.0f)), vec4::load_xmm(max)); + return vec4::store_xmm(_mm_mul_ps(xt, _mm_or_ps(yt, zt))); +} + +inline vec4 vec4::div_min_max(const vec4& left, const vec4& min, const vec4& max) { + __m128 xt; + __m128 yt; + __m128 zt; + xt = vec4::load_xmm(left); + yt = _mm_and_ps(_mm_cmplt_ps(xt, vec4::load_xmm(0.0f)), vec4::load_xmm(-min)); + zt = _mm_and_ps(_mm_cmpge_ps(xt, vec4::load_xmm(0.0f)), vec4::load_xmm(max)); + return vec4::store_xmm(_mm_div_ps(xt, _mm_or_ps(yt, zt))); +} + +inline vec4 vec4::div_min_max(const vec4& left, const float_t min, const float_t max) { + __m128 xt; + __m128 yt; + __m128 zt; + xt = vec4::load_xmm(left); + yt = _mm_and_ps(_mm_cmplt_ps(xt, vec4::load_xmm(0.0f)), vec4::load_xmm(-min)); + zt = _mm_and_ps(_mm_cmpge_ps(xt, vec4::load_xmm(0.0f)), vec4::load_xmm(max)); + return vec4::store_xmm(_mm_div_ps(xt, _mm_or_ps(yt, zt))); +} + +inline vec2i::vec2i() : x(), y() { + +} + +inline vec2i::vec2i(int32_t value) : x(value), y(value) { + +} + +inline vec2i::vec2i(int32_t x, int32_t y) : x(x), y(y) { + +} + +inline vec2i operator +(const vec2i& left, const vec2i& right) { + return vec2i::store_xmm(_mm_add_epi32(vec2i::load_xmm(left), vec2i::load_xmm(right))); +} + +inline vec2i operator +(const vec2i& left, const int32_t right) { + return vec2i::store_xmm(_mm_add_epi32(vec2i::load_xmm(left), vec2i::load_xmm(right))); +} + +inline vec2i operator +(const int32_t left, const vec2i& right) { + return vec2i::store_xmm(_mm_add_epi32(vec2i::load_xmm(left), vec2i::load_xmm(right))); +} + +inline vec2i operator -(const vec2i& left, const vec2i& right) { + return vec2i::store_xmm(_mm_sub_epi32(vec2i::load_xmm(left), vec2i::load_xmm(right))); +} + +inline vec2i operator -(const vec2i& left, const int32_t right) { + return vec2i::store_xmm(_mm_sub_epi32(vec2i::load_xmm(left), vec2i::load_xmm(right))); +} + +inline vec2i operator -(const int32_t left, const vec2i& right) { + return vec2i::store_xmm(_mm_sub_epi32(vec2i::load_xmm(left), vec2i::load_xmm(right))); +} + +inline __m128i vec2i::load_xmm(const int32_t data) { + __m128i _data = _mm_cvtsi32_si128(data); + return _mm_shuffle_epi32(_data, 0); +} + +inline __m128i vec2i::load_xmm(const vec2i& data) { + return _mm_loadl_epi64((const __m128i*) & data); +} + +inline __m128i vec2i::load_xmm(const vec2i&& data) { + return _mm_loadl_epi64((const __m128i*) & data); +} + +inline vec2i vec2i::store_xmm(const __m128i& data) { + vec2i _data; + _mm_storel_epi64((__m128i*) & _data, data); + return _data; +} + +inline vec2i vec2i::store_xmm(const __m128i&& data) { + vec2i _data; + _mm_storel_epi64((__m128i*) & _data, data); + return _data; +} + +inline vec2i vec2i::min(const vec2i& left, const vec2i& right) { + return vec2i::store_xmm(_mm_min_epi32(vec2i::load_xmm(left), vec2i::load_xmm(right))); +} + +inline vec2i vec2i::max(const vec2i& left, const vec2i& right) { + return vec2i::store_xmm(_mm_max_epi32(vec2i::load_xmm(left), vec2i::load_xmm(right))); +} + +inline vec2i vec2i::clamp(const vec2i& left, const vec2i& min, const vec2i& max) { + return vec2i::store_xmm(_mm_min_epi32(_mm_max_epi32(vec2i::load_xmm(left), + vec2i::load_xmm(min)), vec2i::load_xmm(max))); +} + +inline vec2i vec2i::clamp(const vec2i& left, const int32_t min, const int32_t max) { + return vec2i::store_xmm(_mm_min_epi32(_mm_max_epi32(vec2i::load_xmm(left), + vec2i::load_xmm(min)), vec2i::load_xmm(max))); +} + +inline vec3i::vec3i() : x(), y(), z() { + +} + +inline vec3i::vec3i(int32_t value) : x(value), y(value), z(value) { + +} + +inline vec3i::vec3i(int32_t x, int32_t y, int32_t z) : x(x), y(y), z(z) { + +} + +inline vec3i operator +(const vec3i& left, const vec3i& right) { + return vec3i::store_xmm(_mm_add_epi32(vec3i::load_xmm(left), vec3i::load_xmm(right))); +} + +inline vec3i operator +(const vec3i& left, const int32_t right) { + return vec3i::store_xmm(_mm_add_epi32(vec3i::load_xmm(left), vec3i::load_xmm(right))); +} + +inline vec3i operator +(const int32_t left, const vec3i& right) { + return vec3i::store_xmm(_mm_add_epi32(vec3i::load_xmm(left), vec3i::load_xmm(right))); +} + +inline vec3i operator -(const vec3i& left, const vec3i& right) { + return vec3i::store_xmm(_mm_sub_epi32(vec3i::load_xmm(left), vec3i::load_xmm(right))); +} + +inline vec3i operator -(const vec3i& left, const int32_t right) { + return vec3i::store_xmm(_mm_sub_epi32(vec3i::load_xmm(left), vec3i::load_xmm(right))); +} + +inline vec3i operator -(const int32_t left, const vec3i& right) { + return vec3i::store_xmm(_mm_sub_epi32(vec3i::load_xmm(left), vec3i::load_xmm(right))); +} + +inline __m128i vec3i::load_xmm(const int32_t data) { + __m128i _data = _mm_cvtsi32_si128(data); + return _mm_shuffle_epi32(_data, 0); +} + +inline __m128i vec3i::load_xmm(const vec3i& data) { + return _mm_unpacklo_epi64(_mm_loadl_epi64((const __m128i*) & data), _mm_cvtsi32_si128(data.z)); +} + +inline __m128i vec3i::load_xmm(const vec3i&& data) { + return _mm_unpacklo_epi64(_mm_loadl_epi64((const __m128i*) & data), _mm_cvtsi32_si128(data.z)); +} + +inline vec3i vec3i::store_xmm(const __m128i& data) { + vec3i _data; + _mm_storel_epi64((__m128i*) & _data, data); + _data.z = _mm_cvtsi128_si32(_mm_srli_si128(data, 8)); + return _data; +} + +inline vec3i vec3i::store_xmm(const __m128i&& data) { + vec3i _data; + _mm_storel_epi64((__m128i*) & _data, data); + _data.z = _mm_cvtsi128_si32(_mm_srli_si128(data, 8)); + return _data; +} + +inline vec3i vec3i::min(const vec3i& left, const vec3i& right) { + return vec3i::store_xmm(_mm_min_epi32(vec3i::load_xmm(left), vec3i::load_xmm(right))); +} + +inline vec3i vec3i::max(const vec3i& left, const vec3i& right) { + return vec3i::store_xmm(_mm_max_epi32(vec3i::load_xmm(left), vec3i::load_xmm(right))); +} + +inline vec3i vec3i::clamp(const vec3i& left, const vec3i& min, const vec3i& max) { + return vec3i::store_xmm(_mm_min_epi32(_mm_max_epi32(vec3i::load_xmm(left), + vec3i::load_xmm(min)), vec3i::load_xmm(max))); +} + +inline vec3i vec3i::clamp(const vec3i& left, const int32_t min, const int32_t max) { + return vec3i::store_xmm(_mm_min_epi32(_mm_max_epi32(vec3i::load_xmm(left), + vec3i::load_xmm(min)), vec3i::load_xmm(max))); +} + +inline vec4i::vec4i() : x(), y(), z(), w() { + +} + +inline vec4i::vec4i(int32_t value) : x(value), y(value), z(value), w(value) { + +} + +inline vec4i::vec4i(int32_t x, int32_t y, int32_t z, int32_t w) : x(x), y(y), z(z), w(w) { + +} + +inline vec4i operator +(const vec4i& left, const vec4i& right) { + return vec4i::store_xmm(_mm_add_epi32(vec4i::load_xmm(left), vec4i::load_xmm(right))); +} + +inline vec4i operator +(const vec4i& left, const int32_t right) { + return vec4i::store_xmm(_mm_add_epi32(vec4i::load_xmm(left), vec4i::load_xmm(right))); +} + +inline vec4i operator +(const int32_t left, const vec4i& right) { + return vec4i::store_xmm(_mm_add_epi32(vec4i::load_xmm(left), vec4i::load_xmm(right))); +} + +inline vec4i operator -(const vec4i& left, const vec4i& right) { + return vec4i::store_xmm(_mm_sub_epi32(vec4i::load_xmm(left), vec4i::load_xmm(right))); +} + +inline vec4i operator -(const vec4i& left, const int32_t right) { + return vec4i::store_xmm(_mm_sub_epi32(vec4i::load_xmm(left), vec4i::load_xmm(right))); +} + +inline vec4i operator -(const int32_t left, const vec4i& right) { + return vec4i::store_xmm(_mm_sub_epi32(vec4i::load_xmm(left), vec4i::load_xmm(right))); +} + +inline __m128i vec4i::load_xmm(const int32_t data) { + __m128i _data = _mm_cvtsi32_si128(data); + return _mm_shuffle_epi32(_data, 0); +} + +inline __m128i vec4i::load_xmm(const vec4i& data) { + return _mm_loadu_si128((const __m128i*) & data); +} + +inline __m128i vec4i::load_xmm(const vec4i&& data) { + return _mm_loadu_si128((const __m128i*) & data); +} + +inline vec4i vec4i::store_xmm(const __m128i& data) { + vec4i _data; + _mm_storeu_si128((__m128i*) & _data, data); + return _data; +} + +inline vec4i vec4i::store_xmm(const __m128i&& data) { + vec4i _data; + _mm_storeu_si128((__m128i*) & _data, data); + return _data; +} + +inline vec4i vec4i::min(const vec4i& left, const vec4i& right) { + return vec4i::store_xmm(_mm_min_epi32(vec4i::load_xmm(left), vec4i::load_xmm(right))); +} + +inline vec4i vec4i::max(const vec4i& left, const vec4i& right) { + return vec4i::store_xmm(_mm_max_epi32(vec4i::load_xmm(left), vec4i::load_xmm(right))); +} + +inline vec4i vec4i::clamp(const vec4i& left, const vec4i& min, const vec4i& max) { + return vec4i::store_xmm(_mm_min_epi32(_mm_max_epi32(vec4i::load_xmm(left), + vec4i::load_xmm(min)), vec4i::load_xmm(max))); +} + +inline vec4i vec4i::clamp(const vec4i& left, const int32_t min, const int32_t max) { + return vec4i::store_xmm(_mm_min_epi32(_mm_max_epi32(vec4i::load_xmm(left), + vec4i::load_xmm(min)), vec4i::load_xmm(max))); +} + +inline void vec2i8_to_vec2(const vec2i8& src, vec2& dst) { + dst.x = (float_t)src.x; + dst.y = (float_t)src.y; +} + +inline void vec3i8_to_vec3(const vec3i8& src, vec3& dst) { + dst.x = (float_t)src.x; + dst.y = (float_t)src.y; + dst.z = (float_t)src.z; +} + +inline void vec4i8_to_vec4(const vec4i8& src, vec4& dst) { + dst.x = (float_t)src.x; + dst.y = (float_t)src.y; + dst.z = (float_t)src.z; + dst.w = (float_t)src.w; +} + +inline void vec2_to_vec2i8(const vec2& src, vec2i8& dst) { + dst.x = (int8_t)src.x; + dst.y = (int8_t)src.y; +} + +inline void vec3_to_vec3i8(const vec3& src, vec3i8& dst) { + dst.x = (int8_t)src.x; + dst.y = (int8_t)src.y; + dst.z = (int8_t)src.z; +} + +inline void vec4_to_vec4i8(const vec4& src, vec4i8& dst) { + dst.x = (int8_t)src.x; + dst.y = (int8_t)src.y; + dst.z = (int8_t)src.z; + dst.w = (int8_t)src.w; +} + +inline void vec2u8_to_vec2(const vec2u8& src, vec2& dst) { + dst.x = (float_t)src.x; + dst.y = (float_t)src.y; +} + +inline void vec3u8_to_vec3(const vec3u8& src, vec3& dst) { + dst.x = (float_t)src.x; + dst.y = (float_t)src.y; + dst.z = (float_t)src.z; +} + +inline void vec4u8_to_vec4(const vec4u8& src, vec4& dst) { + dst.x = (float_t)src.x; + dst.y = (float_t)src.y; + dst.z = (float_t)src.z; + dst.w = (float_t)src.w; +} + +inline void vec2_to_vec2u8(const vec2& src, vec2u8& dst) { + dst.x = (uint8_t)src.x; + dst.y = (uint8_t)src.y; +} + +inline void vec3_to_vec3u8(const vec3& src, vec3u8& dst) { + dst.x = (uint8_t)src.x; + dst.y = (uint8_t)src.y; + dst.z = (uint8_t)src.z; +} + +inline void vec4_to_vec4u8(const vec4& src, vec4u8& dst) { + dst.x = (uint8_t)src.x; + dst.y = (uint8_t)src.y; + dst.z = (uint8_t)src.z; + dst.w = (uint8_t)src.w; +} + +inline void vec2i16_to_vec2(const vec2i16& src, vec2& dst) { + dst.x = (float_t)src.x; + dst.y = (float_t)src.y; +} + +inline void vec3i16_to_vec3(const vec3i16& src, vec3& dst) { + dst.x = (float_t)src.x; + dst.y = (float_t)src.y; + dst.z = (float_t)src.z; +} + +inline void vec4i16_to_vec4(const vec4i16& src, vec4& dst) { + dst.x = (float_t)src.x; + dst.y = (float_t)src.y; + dst.z = (float_t)src.z; + dst.w = (float_t)src.w; +} + +inline void vec2_to_vec2i16(const vec2& src, vec2i16& dst) { + dst.x = (int16_t)src.x; + dst.y = (int16_t)src.y; +} + +inline void vec3_to_vec3i16(const vec3& src, vec3i16& dst) { + dst.x = (int16_t)src.x; + dst.y = (int16_t)src.y; + dst.z = (int16_t)src.z; +} + +inline void vec4_to_vec4i16(const vec4& src, vec4i16& dst) { + dst.x = (int16_t)src.x; + dst.y = (int16_t)src.y; + dst.z = (int16_t)src.z; + dst.w = (int16_t)src.w; +} + +inline void vec2u16_to_vec2(const vec2u16& src, vec2& dst) { + dst.x = (float_t)src.x; + dst.y = (float_t)src.y; +} + +inline void vec3u16_to_vec3(const vec3u16& src, vec3& dst) { + dst.x = (float_t)src.x; + dst.y = (float_t)src.y; + dst.z = (float_t)src.z; +} + +inline void vec4u16_to_vec4(const vec4u16& src, vec4& dst) { + dst.x = (float_t)src.x; + dst.y = (float_t)src.y; + dst.z = (float_t)src.z; + dst.w = (float_t)src.w; +} + +inline void vec2_to_vec2u16(const vec2& src, vec2u16& dst) { + dst.x = (uint16_t)src.x; + dst.y = (uint16_t)src.y; +} + +inline void vec3_to_vec3u16(const vec3& src, vec3u16& dst) { + dst.x = (uint16_t)src.x; + dst.y = (uint16_t)src.y; + dst.z = (uint16_t)src.z; +} + +inline void vec4_to_vec4u16(const vec4& src, vec4u16& dst) { + dst.x = (uint16_t)src.x; + dst.y = (uint16_t)src.y; + dst.z = (uint16_t)src.z; + dst.w = (uint16_t)src.w; +} + +inline void vec2h_to_vec2(const vec2h& src, vec2& dst) { + extern bool f16c; + if (f16c) { + dst = vec2::store_xmm(_mm_cvtph_ps(_mm_cvtsi32_si128(*(int32_t*)&src))); + return; + } + + dst.x = half_to_float_convert(src.x); + dst.y = half_to_float_convert(src.y); +} + +inline void vec3h_to_vec3(const vec3h& src, vec3& dst) { + dst.x = half_to_float_convert(src.x); + dst.y = half_to_float_convert(src.y); + dst.z = half_to_float_convert(src.z); +} + +inline void vec4h_to_vec4(const vec4h& src, vec4& dst) { + extern bool f16c; + if (f16c) { + dst = vec4::store_xmm(_mm_cvtph_ps(_mm_cvtsi64_si128(*(int64_t*)&src))); + return; + } + + dst.x = half_to_float_convert(src.x); + dst.y = half_to_float_convert(src.y); + dst.z = half_to_float_convert(src.z); + dst.w = half_to_float_convert(src.w); +} + +inline void vec2_to_vec2h(const vec2& src, vec2h& dst) { + extern bool f16c; + if (f16c) { + *(int32_t*)&dst = _mm_cvtsi128_si32(_mm_cvtps_ph(vec2::load_xmm(src), _MM_FROUND_CUR_DIRECTION)); + return; + } + + dst.x = float_to_half_convert(src.x); + dst.y = float_to_half_convert(src.y); +} + +inline void vec3_to_vec3h(const vec3& src, vec3h& dst) { + dst.x = float_to_half_convert(src.x); + dst.y = float_to_half_convert(src.y); + dst.z = float_to_half_convert(src.z); +} + +inline void vec4_to_vec4h(const vec4& src, vec4h& dst) { + extern bool f16c; + if (f16c) { + *(int64_t*)&dst = _mm_cvtsi128_si64(_mm_cvtps_ph(vec4::load_xmm(src), _MM_FROUND_CUR_DIRECTION)); + return; + } + + dst.x = float_to_half_convert(src.x); + dst.y = float_to_half_convert(src.y); + dst.z = float_to_half_convert(src.z); + dst.w = float_to_half_convert(src.w); +} + +inline void vec2i8_to_vec2i(const vec2i8& src, vec2i& dst) { + dst.x = src.x; + dst.y = src.y; +} + +inline void vec3i8_to_vec3i(const vec3i8& src, vec3i& dst) { + dst.x = src.x; + dst.y = src.y; + dst.z = src.z; +} + +inline void vec4i8_to_vec4i(const vec4i8& src, vec4i& dst) { + dst.x = src.x; + dst.y = src.y; + dst.z = src.z; + dst.w = src.w; +} + +inline void vec2i_to_vec2i8(const vec2i& src, vec2i8& dst) { + dst.x = (int8_t)src.x; + dst.y = (int8_t)src.y; +} + +inline void vec3i_to_vec3i8(const vec3i& src, vec3i8& dst) { + dst.x = (int8_t)src.x; + dst.y = (int8_t)src.y; + dst.z = (int8_t)src.z; +} + +inline void vec4i_to_vec4i8(const vec4i& src, vec4i8& dst) { + dst.x = (int8_t)src.x; + dst.y = (int8_t)src.y; + dst.z = (int8_t)src.z; + dst.w = (int8_t)src.w; +} + +inline void vec2u8_to_vec2i(const vec2u8& src, vec2i& dst) { + dst.x = src.x; + dst.y = src.y; +} + +inline void vec3u8_to_vec4i(const vec3u8& src, vec3i& dst) { + dst.x = src.x; + dst.y = src.y; + dst.z = src.z; +} + +inline void vec4u8_to_vec4i(const vec4u8& src, vec4i& dst) { + dst.x = src.x; + dst.y = src.y; + dst.z = src.z; + dst.w = src.w; +} + +inline void vec2i_to_vec2u8(const vec2i& src, vec2u8& dst) { + dst.x = (uint8_t)src.x; + dst.y = (uint8_t)src.y; +} + +inline void vec3i_to_vec3u8(const vec3i& src, vec3u8& dst) { + dst.x = (uint8_t)src.x; + dst.y = (uint8_t)src.y; + dst.z = (uint8_t)src.z; +} + +inline void vec4i_to_vec4u8(const vec4i& src, vec4u8& dst) { + dst.x = (uint8_t)src.x; + dst.y = (uint8_t)src.y; + dst.z = (uint8_t)src.z; + dst.w = (uint8_t)src.w; +} + +inline void vec2i16_to_vec2i(const vec2i16& src, vec2i& dst) { + dst.x = src.x; + dst.y = src.y; +} + +inline void vec3i16_to_vec4i(const vec3i16& src, vec3i& dst) { + dst.x = src.x; + dst.y = src.y; + dst.z = src.z; +} + +inline void vec4i16_to_vec4i(const vec4i16& src, vec4i& dst) { + dst.x = src.x; + dst.y = src.y; + dst.z = src.z; + dst.w = src.w; +} + +inline void vec2i_to_vec2i16(const vec2i& src, vec2i16& dst) { + dst.x = (int16_t)src.x; + dst.y = (int16_t)src.y; +} + +inline void vec3i_to_vec3i16(const vec3i& src, vec3i16& dst) { + dst.x = (int16_t)src.x; + dst.y = (int16_t)src.y; + dst.z = (int16_t)src.z; +} + +inline void vec4i_to_vec4i16(const vec4i& src, vec4i16& dst) { + dst.x = (int16_t)src.x; + dst.y = (int16_t)src.y; + dst.z = (int16_t)src.z; + dst.w = (int16_t)src.w; +} + +inline void vec2u16_to_vec2i(const vec2u16& src, vec2i& dst) { + dst.x = src.x; + dst.y = src.y; +} + +inline void vec3u16_to_vec4i(const vec3u16& src, vec3i& dst) { + dst.x = src.x; + dst.y = src.y; + dst.z = src.z; +} + +inline void vec4u16_to_vec4i(const vec4u16& src, vec4i& dst) { + dst.x = src.x; + dst.y = src.y; + dst.z = src.z; + dst.w = src.w; +} + +inline void vec2i_to_vec2u16(const vec2i& src, vec2u16& dst) { + dst.x = (uint16_t)src.x; + dst.y = (uint16_t)src.y; +} + +inline void vec3i_to_vec3u16(const vec3i& src, vec3u16& dst) { + dst.x = (uint16_t)src.x; + dst.y = (uint16_t)src.y; + dst.z = (uint16_t)src.z; +} + +inline void vec4i_to_vec4u16(const vec4i& src, vec4u16& dst) { + dst.x = (uint16_t)src.x; + dst.y = (uint16_t)src.y; + dst.z = (uint16_t)src.z; + dst.w = (uint16_t)src.w; +} + +inline void vec2_to_vec2i(const vec2& x, vec2i& z) { + z = vec2i::store_xmm(_mm_cvtps_epi32(vec2::load_xmm(x))); +} + +inline void vec2i_to_vec2(const vec2i& x, vec2& z) { + z = vec2::store_xmm(_mm_cvtepi32_ps(vec2i::load_xmm(x))); +} + +inline void vec3_to_vec3i(const vec3& x, vec3i& z) { + z = vec3i::store_xmm(_mm_cvtps_epi32(vec3::load_xmm(x))); +} + +inline void vec3i_to_vec3(const vec3i& x, vec3& z) { + z = vec3::store_xmm(_mm_cvtepi32_ps(vec3i::load_xmm(x))); +} + +inline void vec4_to_vec4i(const vec4& x, vec4i& z) { + z = vec4i::store_xmm(_mm_cvtps_epi32(vec4::load_xmm(x))); +} + +inline void vec4i_to_vec4(const vec4i& x, vec4& z) { + z = vec4::store_xmm(_mm_cvtepi32_ps(vec4i::load_xmm(x))); +} diff --git a/src/KKdLib/waitable_timer.hpp b/src/KKdLib/waitable_timer.hpp new file mode 100644 index 0000000..354f448 --- /dev/null +++ b/src/KKdLib/waitable_timer.hpp @@ -0,0 +1,57 @@ +/* + by korenkonder + GitHub/GitLab: korenkonder +*/ + +#pragma once + +#include "default.hpp" + +struct waitable_timer { + HANDLE handle; + + inline waitable_timer() { + handle = CreateWaitableTimerW(0, 0, 0); + } + + inline ~waitable_timer() { + if (handle) { + CloseHandle(handle); + handle = 0; + } + } + + inline void sleep(int64_t msec) { + if (msec <= 0.0) + return; + + if (handle) { + LARGE_INTEGER t; + t.QuadPart = (LONGLONG)(msec * -10000); + SetWaitableTimer(handle, &t, 0, 0, 0, 0); + WaitForSingleObject(handle, INFINITE); + } + else { + DWORD msec_dw = (DWORD)msec; + if (msec_dw) + Sleep(msec_dw); + } + } + + inline void sleep_float(double_t msec) { + if (msec <= 0.0) + return; + + if (handle) { + LARGE_INTEGER t; + t.QuadPart = (LONGLONG)round(msec * -10000.0); + SetWaitableTimer(handle, &t, 0, 0, 0, 0); + WaitForSingleObject(handle, INFINITE); + } + else { + DWORD msec_dw = (DWORD)round(msec); + if (msec_dw) + Sleep(msec_dw); + } + } +};