diff --git a/breakingChanges.md b/breakingChanges.md index 7b89ea73..ac6e5112 100644 --- a/breakingChanges.md +++ b/breakingChanges.md @@ -1,5 +1,11 @@ # Breaking changes +## Animation performance improvements (2026-10-06) + +The `matrix` member of `T3DBone` is now a new `T3DMat4x3`. +Any kind of matrix math needs adjustments. +Note that a pointer to that struct cannot simply be cast to a `T3DMat4`. + ## Precision fix (2026-09-28, [e14ff480e6ab15b723373754d1edd95af22cc9f5](https://github.com/HailToDodongo/tiny3d/commit/e14ff480e6ab15b723373754d1edd95af22cc9f5)) This change increased depth precision, and by extension UV and position too by a bit. diff --git a/examples/96_animtest/Makefile b/examples/96_animtest/Makefile new file mode 100644 index 00000000..de5bd322 --- /dev/null +++ b/examples/96_animtest/Makefile @@ -0,0 +1,59 @@ +BUILD_DIR=build +T3D_INST=$(shell realpath ../..) + +include $(N64_INST)/include/n64.mk +include $(T3D_INST)/t3d.mk + +N64_CFLAGS += -std=gnu2x -O2 + +PROJECT_NAME=t3d_96_animtest + +src = main.c +assets_png = $(wildcard assets/*.png) +assets_gltf = $(wildcard assets/*.glb) +assets_ttf = $(wildcard assets/*.ttf) + +assets_conv = $(addprefix filesystem/,$(notdir $(assets_png:%.png=%.sprite))) \ + $(addprefix filesystem/,$(notdir $(assets_ttf:%.ttf=%.font64))) \ + $(addprefix filesystem/,$(notdir $(assets_gltf:%.glb=%.t3dm))) + +all: $(PROJECT_NAME).z64 + +filesystem/%.sprite: assets/%.png + @mkdir -p $(dir $@) + @echo " [SPRITE] $@" + $(N64_MKSPRITE) $(MKSPRITE_FLAGS) -o filesystem "$<" + +filesystem/%.font64: assets/%.ttf + @mkdir -p $(dir $@) + @echo " [FONT] $@" + $(N64_MKFONT) $(MKFONT_FLAGS) -s 9 -o filesystem "$<" + +filesystem/%.t3dm: assets/%.glb + @mkdir -p $(dir $@) + @echo " [T3D-MODEL] $@" + $(T3D_GLTF_TO_3D) "$<" $@ + $(N64_BINDIR)/mkasset -c 2 -w 256 -o filesystem $@ + +$(BUILD_DIR)/$(PROJECT_NAME).dfs: $(assets_conv) +$(BUILD_DIR)/$(PROJECT_NAME).elf: $(src:%.c=$(BUILD_DIR)/%.o) + +$(PROJECT_NAME).z64: N64_ROM_TITLE="Tiny3D - AnimTest" +$(PROJECT_NAME).z64: $(BUILD_DIR)/$(PROJECT_NAME).dfs + +clean: + rm -rf $(BUILD_DIR) *.z64 + rm -rf filesystem + +sc64: + make build_lib + sc64deployer upload *.z64 && curl 192.168.0.6:9065/off && sleep 1 && curl 192.168.0.6:9065/on + +build_lib: + rm -rf $(BUILD_DIR) *.z64 + make -C $(T3D_INST) + make all + +-include $(wildcard $(BUILD_DIR)/*.d) + +.PHONY: all clean diff --git a/examples/96_animtest/assets/BootTex.ci4.png b/examples/96_animtest/assets/BootTex.ci4.png new file mode 100644 index 00000000..48ff66a1 Binary files /dev/null and b/examples/96_animtest/assets/BootTex.ci4.png differ diff --git a/examples/96_animtest/assets/ChestTex_ci4.png b/examples/96_animtest/assets/ChestTex_ci4.png new file mode 100644 index 00000000..d43fd8d1 Binary files /dev/null and b/examples/96_animtest/assets/ChestTex_ci4.png differ diff --git a/examples/96_animtest/assets/FaceTex.ci4.png b/examples/96_animtest/assets/FaceTex.ci4.png new file mode 100644 index 00000000..1b1c3387 Binary files /dev/null and b/examples/96_animtest/assets/FaceTex.ci4.png differ diff --git a/examples/96_animtest/assets/cath.glb b/examples/96_animtest/assets/cath.glb new file mode 100644 index 00000000..c94bae87 Binary files /dev/null and b/examples/96_animtest/assets/cath.glb differ diff --git a/examples/96_animtest/assets/catherine_CREDITS.txt b/examples/96_animtest/assets/catherine_CREDITS.txt new file mode 100644 index 00000000..f2cd2512 --- /dev/null +++ b/examples/96_animtest/assets/catherine_CREDITS.txt @@ -0,0 +1,2 @@ +"catherine.blend" Model from: https://github.com/buu342/N64-Sausage64 +License: WTFPL license \ No newline at end of file diff --git a/examples/96_animtest/main.c b/examples/96_animtest/main.c new file mode 100644 index 00000000..0317f3fd --- /dev/null +++ b/examples/96_animtest/main.c @@ -0,0 +1,169 @@ +#include +#include +#include +#include +#include + +/** + * Animation benchmark (96_animtest). + * + * Draws 4 instances of the same model, each with its own skeleton playing a different animation. + * The CPU time of 't3d_anim_update' and 't3d_skeleton_update' (summed over all instances) + * is measured every frame and shown as average and peak over the last 'AVG_FRAMES' frames. + * A fixed delta-time is used so that runs are reproducible. + */ + +#define FB_COUNT 3 +#define INST_COUNT 4 +#define AVG_FRAMES 64 +#define DELTA_TIME (1.0f / 60.0f) + +typedef struct { + T3DSkeleton skel; + T3DAnim anim; + float posX; +} Instance; + +typedef struct { + uint32_t sum; + uint32_t peak; + uint32_t curSum; + uint32_t curPeak; +} TimeStat; + +static void stat_add(TimeStat *s, uint32_t ticks) { + s->curSum += ticks; + if(ticks > s->curPeak)s->curPeak = ticks; +} + +static void stat_flush(TimeStat *s) { + s->sum = s->curSum; + s->peak = s->curPeak; + s->curSum = 0; + s->curPeak = 0; +} + +int main() +{ + debug_init_isviewer(); + debug_init_usblog(); + asset_init_compression(2); + + dfs_init(DFS_DEFAULT_LOCATION); + + display_init(RESOLUTION_320x240, DEPTH_16_BPP, FB_COUNT, GAMMA_NONE, FILTERS_RESAMPLE_ANTIALIAS); + rdpq_init(); + t3d_init((T3DInitParams){}); + rdpq_text_register_font(FONT_BUILTIN_DEBUG_MONO, rdpq_font_load_builtin(FONT_BUILTIN_DEBUG_MONO)); + + T3DViewport viewport = t3d_viewport_create_buffered(FB_COUNT); + T3DMat4FP* modelMatFP = malloc_uncached(sizeof(T3DMat4FP) * FB_COUNT * INST_COUNT); + + const T3DVec3 camPos = {{0, 22.0f, 66.0f}}; + const T3DVec3 camTarget = {{0, 16.0f, 0}}; + + uint8_t colorAmbient[4] = {0xBB, 0xBB, 0xBB, 0xFF}; + uint8_t colorDir[4] = {0xEE, 0xAA, 0xAA, 0xFF}; + T3DVec3 lightDirVec = {{1.0f, 1.0f, 1.0f}}; + t3d_vec3_norm(&lightDirVec); + + // "catherine.blend" Model from: https://github.com/buu342/N64-Sausage64 + T3DModel *model = t3d_model_load("rom:/cath.t3dm"); + const float modelScale = 0.0035f; + const char* animNames[INST_COUNT] = {"Run", "Walk", "Attack1", "Roll"}; + + Instance inst[INST_COUNT]; + for(int i=0; i + #define SQRT_2_INV 0.70710678118f #define KF_TIME_TICK (1.0f / 60.0f) +_Static_assert(_Alignof(T3DAnimTargetQuat) >= 8, "T3DAnimTargetQuat must be 8-byte aligned"); +_Static_assert(offsetof(T3DAnimTargetQuat, kfCurr) % 8 == 0, "kfCurr must be 8-byte aligned"); +_Static_assert(offsetof(T3DAnimTargetQuat, kfNext) % 8 == 0, "kfNext must be 8-byte aligned"); + +typedef uint64_t __attribute__((may_alias)) u64_alias_t; +typedef uint64_t __attribute__((aligned(2), may_alias)) u64_unaligned_t; + // Maps the input data streamed from the animation data file typedef struct { uint16_t nextTime; @@ -16,21 +24,119 @@ typedef struct { uint16_t data[2]; // can be either 1 or 2 16-bit values (scalar / quat) } T3DAnimKF; -T3DAnim t3d_anim_create(const T3DModel *model, const char *name) { +// Starts loading the next part of the file into the given half (async) +static void stream_load(T3DAnim *anim, uint32_t half) { + if(anim->loadOffset >= anim->streamSize)anim->loadOffset = 0; // prefetch the start again for looping + uint32_t size = anim->streamSize - anim->loadOffset; + if(size > anim->bufferHalfSize)size = anim->bufferHalfSize; + + uint8_t *dst = anim->buffer + half * anim->bufferHalfSize; + data_cache_hit_invalidate(dst, anim->bufferHalfSize); + anim->dmaTicket = dma_read_raw_async(dst, anim->romAddr + anim->loadOffset, size); + anim->loadOffset += size; +} + +static inline void stream_wait(T3DAnim *anim) { + if(anim->dmaTicket) { + dma_wait_finished(anim->dmaTicket); + anim->dmaTicket = 0; + } +} + +// (Re-)loads the stream from the start, blocks until the first half is loaded +static void stream_reset(T3DAnim *anim) { + stream_wait(anim); + anim->loadOffset = 0; + stream_load(anim, 0); + stream_wait(anim); + stream_load(anim, 1); + anim->readPos = 0; + anim->streamPos = 0; +} + +static void stream_rewind(T3DAnim *anim) { + const uint32_t halfSize = anim->bufferHalfSize; + uint32_t halfStart = anim->readPos >= halfSize ? halfSize : 0; + uint32_t halfOffset = anim->streamPos - (anim->readPos - halfStart); // file offset of the current half + anim->streamPos = 0; + + // start of the file is in the current half (file smaller than a half, or it was just switched to) + if(halfOffset == 0 || halfOffset >= anim->streamSize) { + anim->readPos = halfStart; + return; + } + // current half contains the end of the file, so the other one was loaded from the start (looping) + if(halfOffset + halfSize >= anim->streamSize) { + stream_wait(anim); + anim->readPos = halfSize - halfStart; + stream_load(anim, halfStart ? 1 : 0); + return; + } + stream_reset(anim); +} + +// Copies the next keyframe out of the stream, returns false at the end of the stream +static inline bool stream_read_kf(T3DAnim *anim, T3DAnimKF *kf) { + uint32_t size = anim->nextKfSize; + if(anim->streamPos + size > anim->streamSize)return false; + + const uint32_t halfSize = anim->bufferHalfSize; + uint32_t pos = anim->readPos; + uint32_t halfEnd = pos >= halfSize ? (halfSize * 2) : halfSize; + const uint16_t *src = (const uint16_t*)(anim->buffer + pos); + uint16_t *dst = (uint16_t*)kf; + + if(pos + 8 <= halfEnd) { // always copy 8 bytes, the check makes sure it never touches the other half + *(u64_alias_t*)kf = *(const u64_unaligned_t*)src; + } else { // crosses into the other half, which may wrap around to the start of the buffer + uint32_t sizeA = (halfEnd - pos) / 2; + for(uint32_t i=0; ibuffer + (halfEnd == halfSize ? halfSize : 0)); + for(uint32_t i=sizeA; istreamPos += size; + if(pos >= halfEnd) { // current half fully read, switch and refill it + stream_wait(anim); + if(pos >= halfSize * 2)pos -= halfSize * 2; + stream_load(anim, halfEnd == halfSize ? 0 : 1); + } + anim->readPos = pos; + return true; +} + +T3DAnim t3d_anim_create_buffered(const T3DModel *model, const char *name, uint32_t bufferSize) { T3DChunkAnim* animDef = t3d_model_get_animation(model, name); assertf(animDef, "Animation '%s' not found in model", name); + assertf(bufferSize >= 32 && (bufferSize % 32) == 0 && bufferSize <= 0x8000, "Invalid animation buffer size: %lu", bufferSize); - return (T3DAnim){ + const char *path = animDef->filePath; + path += 5; + pi_addr_t romAddr = dfs_rom_addr(path); + int streamSize = dfs_rom_size(path); + assertf(romAddr != 0 && streamSize >= 0, "Animation data not found: %s", animDef->filePath); + assertf((romAddr & 1) == 0 && (streamSize & 1) == 0, "Animation data not 2-byte aligned: %s", animDef->filePath); + + T3DAnim anim = { .animRef = animDef, - .targetsScalar = NULL, .targetsQuat = NULL, - .time = 0.0f, + .targetsScalar = NULL, .speed = 1.0f, + .time = 0.0f, + .buffer = memalign(16, bufferSize), // own cache-lines, needed for the invalidate before each DMA + .dmaTicket = 0, + .romAddr = romAddr, + .streamSize = streamSize, + .bufferHalfSize = bufferSize / 2, .nextKfSize = sizeof(T3DAnimKF), - .file = asset_fopen(animDef->filePath, NULL), .isPlaying = 1, .isLooping = 1 }; + // DMAs target the heap buffer, so returning the struct by value is fine + stream_reset(&anim); + return anim; } static void rewind_anim(T3DAnim *anim) @@ -42,7 +148,7 @@ static void rewind_anim(T3DAnim *anim) anim->targetsQuat[c].base.timeEnd = 0; } anim->nextKfSize = sizeof(T3DAnimKF); - rewind(anim->file); + stream_rewind(anim); } void t3d_anim_attach(T3DAnim *anim, const T3DSkeleton *skeleton) { @@ -137,9 +243,8 @@ static inline T3DAnimTargetBase* get_base_target(T3DAnim *anim, uint64_t channel } static inline bool load_keyframe(T3DAnim *anim) { - T3DAnimKF kf; - size_t readBytes = fread(&kf, anim->nextKfSize, 1, anim->file); - if(readBytes == 0)return false; + T3DAnimKF kf __attribute__((aligned(8), uninitialized)); // 8-byte aligned for the 64-bit copy + if(!stream_read_kf(anim, &kf))return false; bool isLarge = kf.nextTime & 0x8000; anim->nextKfSize = isLarge ? sizeof(T3DAnimKF) : (sizeof(T3DAnimKF)-2); @@ -156,7 +261,9 @@ static inline bool load_keyframe(T3DAnim *anim) { if(channelMap->targetType == T3D_ANIM_TARGET_ROTATION) { T3DAnimTargetQuat *target = (T3DAnimTargetQuat*)targetBase; - target->kfCurr = target->kfNext; + // 64-bit copy, otherwise we get a memcpy + ((u64_alias_t*)&target->kfCurr)[0] = ((u64_alias_t*)&target->kfNext)[0]; + ((u64_alias_t*)&target->kfCurr)[1] = ((u64_alias_t*)&target->kfNext)[1]; unpack_quat(kf.data[0], kf.data[1], &target->kfNext); } else { T3DAnimTargetScalar *target = (T3DAnimTargetScalar*)targetBase; @@ -167,6 +274,19 @@ static inline bool load_keyframe(T3DAnim *anim) { return true; } +// Local copy for better cache usage +static inline void local_quat_nlerp(T3DQuat *res, const T3DQuat *a, const T3DQuat *b, float t) { + float blend = 1.0f - t; + if(t3d_quat_dot(a, b) < 0.0f) { + blend = -blend; + } + res->v[0] = blend * a->v[0] + t * b->v[0]; + res->v[1] = blend * a->v[1] + t * b->v[1]; + res->v[2] = blend * a->v[2] + t * b->v[2]; + res->v[3] = blend * a->v[3] + t * b->v[3]; + t3d_quat_normalize(res); +} + void t3d_anim_update(T3DAnim *anim, float deltaTime) { if(!anim->isPlaying)return; int32_t updateFlag = 1; @@ -183,24 +303,31 @@ void t3d_anim_update(T3DAnim *anim, float deltaTime) { } } - uint32_t channelCount = anim->animRef->channelsScalar + anim->animRef->channelsQuat; + // local copies, stores through the target pointers below could alias 'anim' and force reloads otherwise + const float time = anim->time; + const uint32_t channelsQuat = anim->animRef->channelsQuat; + const uint32_t channelCount = anim->animRef->channelsScalar + channelsQuat; + T3DAnimTargetQuat *targetsQuat = anim->targetsQuat; + T3DAnimTargetScalar *targetsScalar = anim->targetsScalar; + for(uint32_t c=0; canimRef->channelsQuat; - T3DAnimTargetBase *target = get_base_target(anim, c, isRot); + bool isRot = c < channelsQuat; + T3DAnimTargetBase *target = isRot ? + (T3DAnimTargetBase*)&targetsQuat[c] : + (T3DAnimTargetBase*)&targetsScalar[c - channelsQuat]; - while(anim->time >= target->timeEnd) { + while(time >= target->timeEnd) { if(!load_keyframe(anim))break; } float timeDiff = target->timeEnd - target->timeStart; - float interp = (anim->time - target->timeStart) / timeDiff; + float interp = (time - target->timeStart) / timeDiff; *target->changedFlag = updateFlag; if(isRot) { T3DAnimTargetQuat *t = (T3DAnimTargetQuat*)target; - t3d_quat_nlerp(t->targetQuat, &t->kfCurr, &t->kfNext, interp); - //t3d_quat_slerp(t->targetQuat, &t->kfCurr, &t->kfNext, interp); + local_quat_nlerp(t->targetQuat, &t->kfCurr, &t->kfNext, interp); } else { T3DAnimTargetScalar *t = (T3DAnimTargetScalar*)target; *t->targetScalar = t3d_lerp(t->kfCurr, t->kfNext, interp); @@ -210,10 +337,13 @@ void t3d_anim_update(T3DAnim *anim, float deltaTime) { void t3d_anim_destroy(T3DAnim *anim) { if(anim->targetsQuat)free(anim->targetsQuat); // 'targetsScalar' is part of this memory-block - if(anim->file)fclose(anim->file); + if(anim->buffer) { + stream_wait(anim); // DMAs could still be writing into the buffer + free(anim->buffer); + } anim->targetsQuat = NULL; anim->targetsScalar = NULL; - anim->file = NULL; + anim->buffer = NULL; } void t3d_anim_set_time(T3DAnim *anim, float time) { diff --git a/src/t3d/t3danim.h b/src/t3d/t3danim.h index ffa42cac..0b471fab 100644 --- a/src/t3d/t3danim.h +++ b/src/t3d/t3danim.h @@ -28,8 +28,8 @@ typedef struct { typedef struct { T3DAnimTargetBase base; T3DQuat* targetQuat; // target to modify - T3DQuat kfCurr; // current keyframe value - T3DQuat kfNext; // next keyframe value + T3DQuat kfCurr __attribute__((aligned(8))); // current keyframe value (aligned for 64-bit copies) + T3DQuat kfNext __attribute__((aligned(8))); // next keyframe value } T3DAnimTargetQuat; typedef struct { @@ -39,6 +39,8 @@ typedef struct { float kfNext; } T3DAnimTargetScalar; +#define T3D_ANIM_DEFAULT_BUFFER_SIZE 512 // default keyframe stream buffer size, split into two halves + typedef struct { T3DChunkAnim *animRef; T3DAnimTargetQuat *targetsQuat; @@ -47,19 +49,46 @@ typedef struct { float speed; float time; - FILE *file; - int nextKfSize; + // keyframe stream, DMA'd from ROM + uint8_t *buffer; + uint64_t dmaTicket; // last queued DMA, always targets the half not being read, 0 if none + pi_addr_t romAddr; + uint32_t streamSize; + uint32_t loadOffset; // file offset of the next DMA + uint32_t streamPos; // bytes read since the last rewind + uint16_t readPos; // read position in 'buffer' (both halves) + uint16_t bufferHalfSize; + + uint8_t nextKfSize; uint8_t isPlaying; uint8_t isLooping; } T3DAnim; /** - * Creates an animation instance from a model's animation definition + * Creates an animation instance from a model's animation definition. + * Keyframes are streamed from ROM via DMA into a buffer of the given size. + * The buffer is split in two halves, while one is read the other one gets loaded in the background. + * Free it with 't3d_anim_destroy'. + * + * @param model The model to create the animation from + * @param name The name of the animation to create + * @param bufferSize size of the stream buffer in bytes, must be a multiple of 32 + * @return The created animation + */ +T3DAnim t3d_anim_create_buffered(const T3DModel *model, const char* name, uint32_t bufferSize); + +/** + * Creates an animation instance from a model's animation definition, + * using the default stream buffer size ('T3D_ANIM_DEFAULT_BUFFER_SIZE'). + * Free it with 't3d_anim_destroy'. + * * @param model The model to create the animation from * @param name The name of the animation to create * @return The created animation */ -T3DAnim t3d_anim_create(const T3DModel *model, const char* name); +static inline T3DAnim t3d_anim_create(const T3DModel *model, const char* name) { + return t3d_anim_create_buffered(model, name, T3D_ANIM_DEFAULT_BUFFER_SIZE); +} /** * Attaches an animation to a skeleton. @@ -110,7 +139,7 @@ void t3d_anim_update(T3DAnim* anim, float deltaTime); /** * Sets the animation to a specific time. - * Note: this may cause some work internally due to potential DMAs. + * Note: going back in time needs a rewind, which may cause a blocking DMA. * @param anim animation to set time for * @param time time in seconds */ diff --git a/src/t3d/t3dmath.c b/src/t3d/t3dmath.c index 9817fe1c..88eee964 100644 --- a/src/t3d/t3dmath.c +++ b/src/t3d/t3dmath.c @@ -219,6 +219,15 @@ void t3d_mat4_ortho(T3DMat4 *mat, float left, float right, float bottom, float t void t3d_mat4_from_srt(T3DMat4 *mat, const float scale[3], const float quat[4], const float translate[3]) { + // read all inputs first, since 'mat' could alias them the compiler would otherwise + // write the matrix unscaled first and then again after loading the scale + float scaleX = scale[0]; + float scaleY = scale[1]; + float scaleZ = scale[2]; + float posX = translate[0]; + float posY = translate[1]; + float posZ = translate[2]; + float qxx = quat[0] * quat[0]; float qyy = quat[1] * quat[1]; float qzz = quat[2] * quat[2]; @@ -230,12 +239,11 @@ void t3d_mat4_from_srt(T3DMat4 *mat, const float scale[3], const float quat[4], float qwz = quat[3] * quat[2]; *mat = (T3DMat4){{ - {1.0f - 2.0f * (qyy + qzz), 2.0f * (qxy + qwz), 2.0f * (qxz - qwy), 0.0f}, - { 2.0f * (qxy - qwz), 1.0f - 2.0f * (qxx + qzz), 2.0f * (qyz + qwx), 0.0f}, - { 2.0f * (qxz + qwy), 2.0f * (qyz - qwx), 1.0f - 2.0f * (qxx + qyy), 0.0f}, - { translate[0], translate[1], translate[2], 1.0f} + {(1.0f - 2.0f * (qyy + qzz)) * scaleX, (2.0f * (qxy + qwz)) * scaleX, (2.0f * (qxz - qwy)) * scaleX, 0.0f}, + {(2.0f * (qxy - qwz)) * scaleY, (1.0f - 2.0f * (qxx + qzz)) * scaleY, (2.0f * (qyz + qwx)) * scaleY, 0.0f}, + {(2.0f * (qxz + qwy)) * scaleZ, (2.0f * (qyz - qwx)) * scaleZ, (1.0f - 2.0f * (qxx + qyy)) * scaleZ, 0.0f}, + {posX, posY, posZ, 1.0f} }}; - t3d_mat4_scale(mat, scale[0], scale[1], scale[2]); } void t3d_mat4_from_srt_euler(T3DMat4 *mat, const float scale[3], const float rot[3], const float translate[3]) diff --git a/src/t3d/t3dmath.h b/src/t3d/t3dmath.h index 32a8a3f5..4b948786 100644 --- a/src/t3d/t3dmath.h +++ b/src/t3d/t3dmath.h @@ -23,6 +23,11 @@ typedef fm_vec4_t T3DVec4; typedef fm_quat_t T3DQuat; typedef fm_mat4_t T3DMat4; +// 4x3 float matrix, missing values are implicitly (0,0,0,1) +typedef struct { + float m[4][3]; +} T3DMat4x3; + // 3D s16.16 fixed-point vector, used as-is by the RSP. typedef struct { int16_t i[4]; diff --git a/src/t3d/t3dskeleton.c b/src/t3d/t3dskeleton.c index 476b9ada..cddebcee 100644 --- a/src/t3d/t3dskeleton.c +++ b/src/t3d/t3dskeleton.c @@ -3,13 +3,88 @@ * @license MIT */ #include "t3dskeleton.h" +#include + +static_assert(sizeof(T3DBone) == 96, "T3DBone should be exactly 6 cache-lines"); + +// Same as 't3d_mat4_from_srt', but for a 4x3 matrix +static void mat4x3_from_srt(T3DMat4x3 *mat, const float scale[3], const float quat[4], const float translate[3]) +{ + // read all inputs first, since 'mat' could alias them + float scaleX = scale[0]; + float scaleY = scale[1]; + float scaleZ = scale[2]; + float posX = translate[0]; + float posY = translate[1]; + float posZ = translate[2]; + + float qxx = quat[0] * quat[0]; + float qyy = quat[1] * quat[1]; + float qzz = quat[2] * quat[2]; + float qxz = quat[0] * quat[2]; + float qxy = quat[0] * quat[1]; + float qyz = quat[1] * quat[2]; + float qwx = quat[3] * quat[0]; + float qwy = quat[3] * quat[1]; + float qwz = quat[3] * quat[2]; + + *mat = (T3DMat4x3){{ + {(1.0f - 2.0f * (qyy + qzz)) * scaleX, (2.0f * (qxy + qwz)) * scaleX, (2.0f * (qxz - qwy)) * scaleX}, + {(2.0f * (qxy - qwz)) * scaleY, (1.0f - 2.0f * (qxx + qzz)) * scaleY, (2.0f * (qyz + qwx)) * scaleY}, + {(2.0f * (qxz + qwy)) * scaleZ, (2.0f * (qyz - qwx)) * scaleZ, (1.0f - 2.0f * (qxx + qyy)) * scaleZ}, + {posX, posY, posZ} + }}; +} + +// Same as 't3d_mat4_mul', but for affine 4x3 matrices +static void mat4x3_mul(T3DMat4x3 *matRes, const T3DMat4x3 *matA, const T3DMat4x3 *matB) +{ + for(uint32_t i=0; i<3; i++) { + for(uint32_t j=0; j<3; j++) { + matRes->m[j][i] = matA->m[0][i] * matB->m[j][0] + + matA->m[1][i] * matB->m[j][1] + + matA->m[2][i] * matB->m[j][2]; + } + matRes->m[3][i] = matA->m[0][i] * matB->m[3][0] + + matA->m[1][i] * matB->m[3][1] + + matA->m[2][i] * matB->m[3][2] + + matA->m[3][i]; + } +} + +// Same as 't3d_mat4_to_fixed_3x4', but from a 4x3 matrix +static void mat4x3_to_fixed(T3DMat4FP *matOut, const T3DMat4x3 *matIn) { + for(uint32_t y=0; y<4; ++y) { + uint32_t fixed0 = T3D_F32_TO_FIXED(matIn->m[y][0]); + uint32_t fixed1 = T3D_F32_TO_FIXED(matIn->m[y][1]); + uint32_t fixed2 = T3D_F32_TO_FIXED(matIn->m[y][2]); + + // prepare 64bit values, this creates less writes to memory later + uint64_t I = (fixed0 & 0xFFFF0000) | (fixed1 >> 16); + I <<= 32; // needs to be separate, otherwise -Os generates wrong code + I |= (fixed2 & 0xFFFF0000); + + // puts a '1' into the last value of the matrix + I |= (y+1) >> 2; + + uint64_t F = (fixed0 << 16) | (fixed1 & 0xFFFF); + F <<= 32; // needs to be separate, otherwise -Os generates wrong code + F |= (fixed2 << 16); + + #pragma GCC diagnostic push + #pragma GCC diagnostic ignored "-Wstrict-aliasing" + *(uint64_t*)matOut->m[y].i = I; // guaranteed to be 64-bit aligned + *(uint64_t*)matOut->m[y].f = F; + #pragma GCC diagnostic pop + } +} T3DSkeleton t3d_skeleton_create_buffered(const T3DModel *model, int bufferCount) { const T3DChunkSkeleton *skelRef = t3d_model_get_skeleton(model); assert(skelRef != NULL); T3DSkeleton skel = (T3DSkeleton){ - .bones = malloc(sizeof(T3DBone) * skelRef->boneCount), + .bones = memalign(16, sizeof(T3DBone) * skelRef->boneCount), .boneMatricesFP = malloc_uncached(sizeof(T3DMat4FP) * skelRef->boneCount * bufferCount), .skeletonRef = skelRef, .bufferCount = bufferCount, @@ -22,7 +97,7 @@ T3DSkeleton t3d_skeleton_create_buffered(const T3DModel *model, int bufferCount) T3DSkeleton t3d_skeleton_clone(const T3DSkeleton *skel, bool useMatrices) { T3DSkeleton result = { - .bones = malloc(sizeof(T3DBone) * skel->skeletonRef->boneCount), + .bones = memalign(16, sizeof(T3DBone) * skel->skeletonRef->boneCount), .boneMatricesFP = NULL, .skeletonRef = skel->skeletonRef, }; @@ -43,6 +118,8 @@ void t3d_skeleton_reset(T3DSkeleton *skeleton) { sizeof(T3DVec3) + sizeof(T3DQuat) + sizeof(T3DVec3) // copy all 3 vectors (SRT) at once ); skeleton->bones[i].hasChanged = true; + skeleton->bones[i].parentIdx = boneDef->parentIdx; + skeleton->bones[i].depth = boneDef->depth; } } @@ -62,22 +139,28 @@ void t3d_skeleton_blend(const T3DSkeleton *skelRes, const T3DSkeleton *skelA, co void t3d_skeleton_update(T3DSkeleton *skeleton) { int updateLevel = -1; - bool forceUpdate = false; + uint32_t forceUpdate = 0; T3DMat4FP* matStackFP = nullptr; + T3DMat4x3 tmp __attribute__((uninitialized)); for(int i = 0; i < skeleton->skeletonRef->boneCount; i++) { T3DBone *bone = &skeleton->bones[i]; - const T3DChunkBone *boneDef = &skeleton->skeletonRef->bones[i]; - if(forceUpdate && boneDef->depth <= updateLevel) { + if(forceUpdate && bone->depth <= updateLevel) { forceUpdate = false; updateLevel = -1; } - if(bone->hasChanged || forceUpdate) + const bool hasChanged = bone->hasChanged | forceUpdate; + if(hasChanged) { + // if a bone changed we need to also update any children. + // To do so, update all following bones until we hit one that has the same depth as the changed bone. + if(!forceUpdate)updateLevel = bone->depth; + forceUpdate = 1; + // only cycle through matrices if at least one bone changes. // this avoids flickering at the end of an animation, since it would cycle through the last X frames otherwise. if(matStackFP == nullptr) @@ -86,20 +169,14 @@ void t3d_skeleton_update(T3DSkeleton *skeleton) matStackFP = &skeleton->boneMatricesFP[skeleton->skeletonRef->boneCount * skeleton->currentBufferIdx]; } - // if a bone changed we need to also update any children. - // To do so, update all following bones until we hit one that has the same depth as the changed bone. - if(!forceUpdate)updateLevel = boneDef->depth; - forceUpdate = true; - - if(boneDef->parentIdx != 0xFFFF) { - T3DMat4 tmp; - t3d_mat4_from_srt(&tmp, bone->scale.v, bone->rotation.v, bone->position.v); - t3d_mat4_mul(&bone->matrix, &skeleton->bones[boneDef->parentIdx].matrix, &tmp); + if(bone->parentIdx != 0xFFFF) { + mat4x3_from_srt(&tmp, bone->scale.v, bone->rotation.v, bone->position.v); + mat4x3_mul(&bone->matrix, &skeleton->bones[bone->parentIdx].matrix, &tmp); } else { - t3d_mat4_from_srt(&bone->matrix, bone->scale.v, bone->rotation.v, bone->position.v); + mat4x3_from_srt(&bone->matrix, bone->scale.v, bone->rotation.v, bone->position.v); } - t3d_mat4_to_fixed(&matStackFP[i], &bone->matrix); + mat4x3_to_fixed(&matStackFP[i], &bone->matrix); // if a bone has changed, we need to force updating it until it reached all buffers. // otherwise once the updating stops, and we cycle through buffers still, it would flicker. diff --git a/src/t3d/t3dskeleton.h b/src/t3d/t3dskeleton.h index e004fb40..6b848675 100644 --- a/src/t3d/t3dskeleton.h +++ b/src/t3d/t3dskeleton.h @@ -19,12 +19,14 @@ extern "C" * if 'hasChanged' is set to true. */ typedef struct { - T3DMat4 matrix; + T3DMat4x3 matrix; T3DVec3 scale; T3DQuat rotation; T3DVec3 position; int32_t hasChanged; -} T3DBone; + uint16_t parentIdx; // copy of the model's bone definition, avoids touching it in 't3d_skeleton_update' + uint16_t depth; +} __attribute__((aligned(16))) T3DBone; /** * Skeleton instance, can be constructed from a model's skeleton definition.