benches, optimizations for sin and cos

This commit is contained in:
veclavtlica
2026-09-13 17:59:20 +03:00
parent 6d7fe7392a
commit 739237575a
8 changed files with 218 additions and 65 deletions
+66 -20
View File
@@ -23,6 +23,11 @@ typedef int32_t com_fixed_t;
extern const uint16_t com_fixed_sin_lut[128];
void com_fixed_print(com_fixed_t a);
void com_fixed_run_tests(void);
void com_fixed_run_bench(void);
static inline com_fixed_t com_fixed_add(com_fixed_t a, com_fixed_t b) {
return a + b;
}
@@ -47,6 +52,16 @@ static inline double com_fixed_as_float(com_fixed_t a) {
(a & 0xFFFF) * (double)(1.0 / COM_FIXED_FRACUNIT);
}
/* Branchless floor operation. */
/* TODO: Check against simpler branching implementation. */
static inline com_fixed_t com_fixed_floor(com_fixed_t const a) {
int32_t const sign = a >> 31;
int32_t const frac_mask = (1 << COM_FIXED_FRACBITS) - 1;
int32_t const has_fraction = ((a & frac_mask) != 0);
int32_t const truncated = a & ~frac_mask;
return truncated - ((sign & has_fraction) << COM_FIXED_FRACBITS);
}
/* Approximated square root by Newton-Raphson in 2 iterations over small LUT. */
com_fixed_t com_fixed_sqrt(com_fixed_t a);
@@ -55,29 +70,26 @@ com_fixed_t com_fixed_sqrt(com_fixed_t a);
/* things like camera movement can be jarring if done without.*/
/* https://namoseley.wordpress.com/2015/07/26/sincos-generation-using-table-lookup-and-iterpolation/
*/
static inline com_fixed_t com_fixed_sin(com_fixed_t a) {
int32_t sign = 1;
static inline com_fixed_t com_fixed_sin(com_fixed_t q) {
/* Wrap to (-Pi2,+Pi2) */
com_fixed_t q = a % COM_FIXED_PI2;
q %= COM_FIXED_PI2;
/* Wrap to [0,+Pi2) */
if (q < 0)
q = COM_FIXED_PI2 + q;
q += COM_FIXED_PI2;
/* Limit to [0,+1) */
q = com_fixed_div(q, COM_FIXED_PI2);
/* Handle cases of 3rd and 4th quadrant. */
/* Handle cases of 3rd and 4th quadrant separately. */
if (q >= COM_FIXED_FRACHALF) {
q -= COM_FIXED_FRACHALF;
sign = -1;
q %= COM_FIXED_FRACHALF;
uint16_t const idx = q >= COM_FIXED_FRACQRTR ? 255 - (q >> 7) : q >> 7;
return -com_fixed_sin_lut[idx];
} else {
uint16_t const idx = q >= COM_FIXED_FRACQRTR ? 255 - (q >> 7) : q >> 7;
return com_fixed_sin_lut[idx];
}
/* Finally clculate the index into LUT. */
uint16_t const idx = q >= COM_FIXED_FRACQRTR ? 255 - (q >> 7) : q >> 7;
return com_fixed_sin_lut[idx] * sign;
}
/* Implemented over sin, as to share one single LUT. */
@@ -87,14 +99,48 @@ static inline com_fixed_t com_fixed_cos(com_fixed_t a) {
return com_fixed_sin(a + COM_FIXED_PIHALF);
}
static inline void com_fixed_sincos(com_fixed_t a, com_fixed_t *restrict s,
com_fixed_t *restrict c) {
*s = com_fixed_sin(a);
*c = com_fixed_cos(a);
/* Combines calculation of both in a more optimal way. */
static inline void com_fixed_sincos(com_fixed_t q, com_fixed_t *s,
com_fixed_t *c) {
q %= COM_FIXED_PI2;
if (q < 0)
q += COM_FIXED_PI2;
q = com_fixed_div(q, COM_FIXED_PI2);
/* Sadly, compilers are dumb and branching case is determined to be
* significantly faster in profiling. */
#define CASE(m_sin_sign, m_cos_sign) \
uint16_t const idx = q >= COM_FIXED_FRACQRTR ? 255 - (q >> 7) : q >> 7; \
*s = m_sin_sign * com_fixed_sin_lut[idx]; \
*c = m_cos_sign * com_fixed_sin_lut[((uint8_t)127 - (uint8_t)idx) % 128]
/* Handle cases of 2rd and 3th quadrant separately, for minus cos. */
if (q >= COM_FIXED_FRACHALF / 2 &&
q < COM_FIXED_FRACHALF + COM_FIXED_FRACHALF / 2) {
/* Handle cases of 3rd and 4th quadrant separately, for minus sin. */
if (q >= COM_FIXED_FRACHALF) {
q %= COM_FIXED_FRACHALF;
CASE(-1, -1);
} else {
CASE(+1, -1);
}
} else {
/* Handle cases of 3rd and 4th quadrant separately, for minus sin. */
if (q >= COM_FIXED_FRACHALF) {
q %= COM_FIXED_FRACHALF;
CASE(-1, 1);
} else {
CASE(+1, 1);
}
}
#undef CASE
}
void com_fixed_print(com_fixed_t a);
void com_fixed_run_tests(void);
static inline com_fixed_t com_fixed_tan(com_fixed_t a) {
com_fixed_t s, c;
com_fixed_sincos(a, &s, &c);
return com_fixed_div(s, c);
}
#endif