/* Fixed point implementation of 4 dimensional matrix. Some optimizing cases are present, such as reordered matrices and assumed identity components. Column-major ordering is assumed unless stated otherwise: a e k o b f l p c g m q d h n r In memory: a b c d e f g h ... */ #ifndef COM_MAT_H #define COM_MAT_H #include "def.h" #include "fix.h" #include "vec.h" #include typedef struct { com_def_alignedas(64) com_def_vector(com_fix_t, a, 4 * 4); } com_mat_t; void com_mat_run_bench(void); #define COM_MAT_PROFILE_SINK(m_m) \ do { \ for (int i = 0; i < 16; ++i) \ COM_DEF_PROFILE_SINK((m_m).a[i]); \ } while (0) static inline com_mat_t com_mat_identity(void) { com_mat_t result = {0}; result.a[0 * 4 + 0] = COM_FIX_FRACUNIT; result.a[1 * 4 + 1] = COM_FIX_FRACUNIT; result.a[2 * 4 + 2] = COM_FIX_FRACUNIT; result.a[3 * 4 + 3] = COM_FIX_FRACUNIT; return result; } /* https://michalpitr.substack.com/p/optimizing-matrix-multiplication */ static inline com_mat_t com_mat_mul(com_mat_t a, com_mat_t b) { // com_mat_t result = {0}; // for (int c = 0; c < 4; ++c) { // for (int k = 0; k < 4; ++k) { // for (int r = 0; r < 4; ++r) { // result.a[r + c * 4] += com_fix_mul(a.a[r + k * 4], b.a[k + c * 4]); // } // } // } com_mat_t result; for (int c = 0; c < 4; ++c) { for (int r = 0; r < 4; ++r) { result.a[r + c * 4] = (((int64_t)a.a[r + 0 * 4] * b.a[0 + c * 4]) + ((int64_t)a.a[r + 1 * 4] * b.a[1 + c * 4]) + ((int64_t)a.a[r + 2 * 4] * b.a[2 + c * 4]) + ((int64_t)a.a[r + 3 * 4] * b.a[3 + c * 4])) >> COM_FIX_FRACBITS; } } return result; } /* This case might be slightly more optimized, as we can reorder one frequently * reused matrix, such as VP. */ /* Note: second a matrix is assumed to be row-major, reverse of typical. */ static inline com_mat_t com_mat_mul_reodered(com_mat_t a, com_mat_t b) { com_mat_t result; for (int c = 0; c < 4; ++c) { for (int r = 0; r < 4; ++r) { result.a[r + c * 4] = (((int64_t)a.a[0 + r * 4] * b.a[0 + c * 4]) + ((int64_t)a.a[1 + r * 4] * b.a[1 + c * 4]) + ((int64_t)a.a[2 + r * 4] * b.a[2 + c * 4]) + ((int64_t)a.a[3 + r * 4] * b.a[3 + c * 4])) >> COM_FIX_FRACBITS; } } return result; } /* Slightly optimized case of assumed identity scaling, might be useful for MVP * calculations, if model matrix does not scale. View matrix is always * unscaled as well. */ // static inline com_mat_t com_mat_mul_no_scale(com_mat_t a, com_mat_t b) { // com_mat_t result; // /* TODO: calc the rest */ // result.a[0 * 4 + 3] = 0; // result.a[1 * 4 + 3] = 0; // result.a[2 * 4 + 3] = 0; // result.a[3 * 4 + 3] = COM_FIX_FRACUNIT; // return result; // } /* Reorder between column and row major, it's also called transposing */ static inline com_mat_t com_mat_reoder(com_mat_t a) { com_mat_t result; for (int r = 0; r < 4; ++r) { for (int c = 0; c < 4; ++c) { result.a[c * 4 + r] = a.a[c + r * 4]; } } return result; } /* TODO: move to .c file */ /* Produces a view matrix needed for camera work. */ static inline com_mat_t com_mat_look_at(com_vec_t pos, com_vec_t up, com_vec_t target) { com_vec_t const r = com_vec_nrm(com_vec_crs(target, up)); com_vec_t const u = com_vec_crs(r, target); com_mat_t result; result.a[0] = r.a[0]; result.a[1] = u.a[0]; result.a[2] = -target.a[0]; result.a[3] = 0; result.a[4] = r.a[1]; result.a[5] = u.a[1]; result.a[6] = -target.a[1]; result.a[7] = 0; result.a[8] = r.a[2]; result.a[9] = u.a[2]; result.a[10] = -target.a[2]; result.a[11] = 0; result.a[12] = -com_vec_dot(r, pos); result.a[13] = -com_vec_dot(u, pos); result.a[14] = com_vec_dot(target, pos); result.a[15] = COM_FIX_FRACUNIT; return result; } /* TODO: move to .c file */ /* Produces a projection matrix needed for camera work. */ static inline com_mat_t com_mat_perspective(uint16_t rwidth, uint16_t rheight, com_fix_t nearz, com_fix_t farz, com_fix_t fov) { com_mat_t result = {0}; com_fix_t const aspect = com_fix_div(rwidth * COM_FIX_FRACUNIT, rheight * COM_FIX_FRACUNIT); com_fix_t const f = com_fix_div( COM_FIX_FRACUNIT, com_fix_tan(com_fix_mul(fov, COM_FIX_FRACHALF))); com_fix_t const fn = com_fix_div(COM_FIX_FRACUNIT, (nearz - farz)); result.a[0 * 4 + 0] = com_fix_div(f, aspect); result.a[1 * 4 + 1] = f; result.a[2 * 4 + 2] = (nearz + farz) * fn; result.a[2 * 4 + 3] = -COM_FIX_FRACUNIT; result.a[3 * 4 + 2] = (COM_FIX_FRACUNIT * 2) * com_fix_mul(com_fix_mul(nearz, farz), fn); return result; } #endif