fixed to fix and some more

This commit is contained in:
veclavtlica
2026-09-13 20:57:34 +03:00
parent a1745f3a52
commit 4e1fa1ed1c
12 changed files with 356 additions and 276 deletions
+2
View File
@@ -4,6 +4,8 @@
#define __USE_POSIX199309 1 #define __USE_POSIX199309 1
#include <time.h> #include <time.h>
/* TODO: Rename to posix.c */
uint64_t com_timer_count_ns(void) { uint64_t com_timer_count_ns(void) {
struct timespec time; struct timespec time;
clock_gettime(CLOCK_MONOTONIC, &time); clock_gettime(CLOCK_MONOTONIC, &time);
+12 -1
View File
@@ -13,10 +13,21 @@
/* TODO: uppercase? */ /* TODO: uppercase? */
#ifdef COM_DEF_COMPILE_MODERN #ifdef COM_DEF_COMPILE_MODERN
#define com_def_alignedas(v_as) _Alignas(v_as) #define com_def_alignedas(v_as) _Alignas(v_as)
#define com_def_vector(v_t, v_v, v_n) \
v_t v_v __attribute__((vector_size(sizeof(v_t) * v_n)))
#else #else
#define com_def_alignedas(v_as) #define com_def_alignedas(v_as)
#define com_def_vector(v_t, v_v, v_n) v_t v_v[v_n]
#endif #endif
#define COM_DEF_PROFILE_VAR volatile #define COM_DEF_VECTOR_NONE 0
#define COM_DEF_VECTOR_SSE2 1
#ifdef COM_DEF_COMPILE_MODERN
#define COM_DEF_VECTOR_IMPL COM_DEF_VECTOR_SSE2
#endif
// #define COM_DEF_PROFILE_VAR volatile /* Sadly, doesn't work in some cases */
#define COM_DEF_PROFILE_SINK(m_v) __asm__ volatile("" ::"r"(m_v) : "memory")
#endif #endif
+47 -44
View File
@@ -3,9 +3,9 @@
https://github.com/howerj/q/blob/master/q.c https://github.com/howerj/q/blob/master/q.c
*/ */
#include "fixed.h"
#include "Timer/timer.h" #include "Timer/timer.h"
#include "def.h" #include "def.h"
#include "fix.h"
#include <assert.h> #include <assert.h>
#include <stdbool.h> #include <stdbool.h>
#include <stdint.h> #include <stdint.h>
@@ -14,14 +14,14 @@
// 4-bit LUT (16 entries) for the normalized range [0.5, 2.0) // 4-bit LUT (16 entries) for the normalized range [0.5, 2.0)
// It stores the initial guess scaled to Q16.16. // It stores the initial guess scaled to Q16.16.
static const uint32_t com_fixed_sqrt_lut[16] = { static const uint32_t com_fix_sqrt_lut[16] = {
46340, 49547, 52521, 55314, 57954, 60464, 62862, 65161, 46340, 49547, 52521, 55314, 57954, 60464, 62862, 65161,
67373, 69508, 71572, 73572, 75514, 77402, 79240, 81033}; 67373, 69508, 71572, 73572, 75514, 77402, 79240, 81033};
/* TODO: Test performance of uint_fast16_t here. */ /* TODO: Test performance of uint_fast16_t here. */
/* Used for other quadrants as well as cosine eval, all from the same table. */ /* Used for other quadrants as well as cosine eval, all from the same table. */
/* Only fractional part is present, reducing the cache footprint. */ /* Only fractional part is present, reducing the cache footprint. */
const uint16_t com_fixed_sin_lut[128] = { const uint16_t com_fix_sin_lut[128] = {
0, 804, 1608, 2412, 3215, 4018, 4821, 5622, 6423, 7223, 8022, 0, 804, 1608, 2412, 3215, 4018, 4821, 5622, 6423, 7223, 8022,
8819, 9616, 10410, 11204, 11995, 12785, 13573, 14359, 15142, 15923, 16702, 8819, 9616, 10410, 11204, 11995, 12785, 13573, 14359, 15142, 15923, 16702,
17479, 18253, 19024, 19792, 20557, 21319, 22078, 22833, 23586, 24334, 25079, 17479, 18253, 19024, 19792, 20557, 21319, 22078, 22833, 23586, 24334, 25079,
@@ -35,7 +35,7 @@ const uint16_t com_fixed_sin_lut[128] = {
63943, 64115, 64276, 64428, 64571, 64703, 64826, 64939, 65043, 65136, 65220, 63943, 64115, 64276, 64428, 64571, 64703, 64826, 64939, 65043, 65136, 65220,
65294, 65358, 65412, 65457, 65491, 65516, 65531}; 65294, 65358, 65412, 65457, 65491, 65516, 65531};
com_fixed_t com_fixed_sqrt(com_fixed_t a) { com_fix_t com_fix_sqrt(com_fix_t a) {
// 1. Handle sign and edge cases // 1. Handle sign and edge cases
assert(a >= 0); assert(a >= 0);
if (a == 0) if (a == 0)
@@ -47,7 +47,7 @@ com_fixed_t com_fixed_sqrt(com_fixed_t a) {
// Calculate how much we need to shift to place the highest bit properly // Calculate how much we need to shift to place the highest bit properly
// We want the value to land squarely within an optimal window // We want the value to land squarely within an optimal window
int32_t shift = (31 - leading_zeros) - COM_FIXED_FRACBITS; int32_t shift = (31 - leading_zeros) - COM_FIX_FRACBITS;
// Normalize shift to always be even so we can cleanly pull out 2^(shift/2) // Normalize shift to always be even so we can cleanly pull out 2^(shift/2)
if (shift & 1) if (shift & 1)
@@ -62,14 +62,14 @@ com_fixed_t com_fixed_sqrt(com_fixed_t a) {
// 3. LUT Lookup using 4 MSBs of the normalized value // 3. LUT Lookup using 4 MSBs of the normalized value
// Extracted index corresponds to the interval [0.5, 2.0) // Extracted index corresponds to the interval [0.5, 2.0)
uint32_t lut_index = (normalized_a >> (COM_FIXED_FRACBITS - 3)) & 0xF; uint32_t lut_index = (normalized_a >> (COM_FIX_FRACBITS - 3)) & 0xF;
uint64_t x = com_fixed_sqrt_lut[lut_index]; uint64_t x = com_fix_sqrt_lut[lut_index];
// 4. Newton-Raphson Iterations: x = 0.5 * (x + normalized_a / x) // 4. Newton-Raphson Iterations: x = 0.5 * (x + normalized_a / x)
// We upscale to 64-bit to prevent intermediate overflow during division // We upscale to 64-bit to prevent intermediate overflow during division
x = (x + ((uint64_t)normalized_a << COM_FIXED_FRACBITS) / x) >> x = (x + ((uint64_t)normalized_a << COM_FIX_FRACBITS) / x) >>
1; // Iteration 1 1; // Iteration 1
x = (x + ((uint64_t)normalized_a << COM_FIXED_FRACBITS) / x) >> x = (x + ((uint64_t)normalized_a << COM_FIX_FRACBITS) / x) >>
1; // Iteration 2 1; // Iteration 2
// 5. Denormalize back to the target scale: result = x * 2^(shift / 2) // 5. Denormalize back to the target scale: result = x * 2^(shift / 2)
@@ -81,7 +81,7 @@ com_fixed_t com_fixed_sqrt(com_fixed_t a) {
} }
} }
void com_fixed_print(com_fixed_t a) { void com_fix_print(com_fix_t a) {
if (a < 0) { if (a < 0) {
printf("-"); printf("-");
a = -a; a = -a;
@@ -95,44 +95,49 @@ void com_fixed_print(com_fixed_t a) {
printf("%d.%04u\n", int_part, decimal_val); printf("%d.%04u\n", int_part, decimal_val);
} }
void com_fixed_run_bench(void) { void com_fix_run_bench(void) {
uint64_t start = com_timer_count_ns(); uint64_t start = com_timer_count_ns();
for (int i = 5000000; i--;) { for (int i = 5000000; i--;) {
COM_DEF_PROFILE_VAR com_fixed_t sqrt = com_fixed_sqrt(i); com_fix_t sqrt = com_fix_sqrt(i);
COM_DEF_PROFILE_SINK(sqrt);
} }
com_timer_profile(start, "fixed: sqrt"); com_timer_profile(start, "fixed: sqrt");
start = com_timer_count_ns(); start = com_timer_count_ns();
for (int i = 50000000; i--;) { for (int i = 50000000; i--;) {
COM_DEF_PROFILE_VAR com_fixed_t sin = com_fixed_sin(i); com_fix_t sin = com_fix_sin(i);
COM_DEF_PROFILE_SINK(sin);
} }
com_timer_profile(start, "fixed: sin"); com_timer_profile(start, "fixed: sin");
start = com_timer_count_ns(); start = com_timer_count_ns();
for (int i = 50000000; i--;) { for (int i = 50000000; i--;) {
COM_DEF_PROFILE_VAR com_fixed_t cos = com_fixed_cos(i); com_fix_t cos = com_fix_cos(i);
COM_DEF_PROFILE_SINK(cos);
} }
com_timer_profile(start, "fixed: cos"); com_timer_profile(start, "fixed: cos");
start = com_timer_count_ns(); start = com_timer_count_ns();
for (int i = 50000000; i--;) { for (int i = 50000000; i--;) {
com_fixed_t sin, cos; com_fix_t sin, cos;
com_fixed_sincos(i, &sin, &cos); com_fix_sincos(i, &sin, &cos);
COM_DEF_PROFILE_VAR com_fixed_t sinv = sin; com_fix_t sinv = sin;
COM_DEF_PROFILE_VAR com_fixed_t cosv = cos; com_fix_t cosv = cos;
COM_DEF_PROFILE_SINK(sinv);
COM_DEF_PROFILE_SINK(cosv);
} }
com_timer_profile(start, "fixed: sincos"); com_timer_profile(start, "fixed: sincos");
} }
void com_fixed_run_tests(void) { void com_fix_run_tests(void) {
{ {
double max_sqrt_deviation = 0.0f; double max_sqrt_deviation = 0.0f;
com_fixed_t sqrt_accumulator = 0; com_fix_t sqrt_accumulator = 0;
com_fixed_t sqrt_test_step = COM_FIXED_FRACUNIT >> 2; /* Quarter step */ com_fix_t sqrt_test_step = COM_FIX_FRACUNIT >> 2; /* Quarter step */
for (int i = 0; i < COM_FIXED_FRACUNIT << 1; ++i) { for (int i = 0; i < COM_FIX_FRACUNIT << 1; ++i) {
com_fixed_t sqrt = com_fixed_sqrt(sqrt_accumulator); com_fix_t sqrt = com_fix_sqrt(sqrt_accumulator);
double const deviation = double const deviation =
sqrt(com_fixed_as_float(sqrt_accumulator)) - com_fixed_as_float(sqrt); sqrt(com_fix_as_float(sqrt_accumulator)) - com_fix_as_float(sqrt);
if (deviation > max_sqrt_deviation) if (deviation > max_sqrt_deviation)
max_sqrt_deviation = deviation; max_sqrt_deviation = deviation;
sqrt_accumulator += sqrt_test_step; sqrt_accumulator += sqrt_test_step;
@@ -142,12 +147,12 @@ void com_fixed_run_tests(void) {
{ {
double max_sin_deviation = 0.0f; double max_sin_deviation = 0.0f;
com_fixed_t sin_accumulator = -COM_FIXED_PI * 4; com_fix_t sin_accumulator = -COM_FIX_PI * 4;
com_fixed_t sin_test_step = (COM_FIXED_PI << 1) >> 10; com_fix_t sin_test_step = (COM_FIX_PI << 1) >> 10;
for (int i = 0; i < 1024 * 32; ++i) { for (int i = 0; i < 1024 * 32; ++i) {
com_fixed_t sin = com_fixed_sin(sin_accumulator); com_fix_t sin = com_fix_sin(sin_accumulator);
double const deviation = fabs(sin(com_fixed_as_float(sin_accumulator)) - double const deviation =
com_fixed_as_float(sin)); fabs(sin(com_fix_as_float(sin_accumulator)) - com_fix_as_float(sin));
if (deviation > max_sin_deviation) if (deviation > max_sin_deviation)
max_sin_deviation = deviation; max_sin_deviation = deviation;
sin_accumulator += sin_test_step; sin_accumulator += sin_test_step;
@@ -157,12 +162,12 @@ void com_fixed_run_tests(void) {
{ {
double max_cos_deviation = 0.0f; double max_cos_deviation = 0.0f;
com_fixed_t cos_accumulator = -COM_FIXED_PI * 4; com_fix_t cos_accumulator = -COM_FIX_PI * 4;
com_fixed_t cos_test_step = (COM_FIXED_PI << 1) >> 10; com_fix_t cos_test_step = (COM_FIX_PI << 1) >> 10;
for (int i = 0; i < 1024 * 32; ++i) { for (int i = 0; i < 1024 * 32; ++i) {
com_fixed_t cos = com_fixed_cos(cos_accumulator); com_fix_t cos = com_fix_cos(cos_accumulator);
double const deviation = fabs(cos(com_fixed_as_float(cos_accumulator)) - double const deviation =
com_fixed_as_float(cos)); fabs(cos(com_fix_as_float(cos_accumulator)) - com_fix_as_float(cos));
if (deviation > max_cos_deviation) if (deviation > max_cos_deviation)
max_cos_deviation = deviation; max_cos_deviation = deviation;
cos_accumulator += cos_test_step; cos_accumulator += cos_test_step;
@@ -173,17 +178,15 @@ void com_fixed_run_tests(void) {
{ {
double max_sin_deviation = 0.0f; double max_sin_deviation = 0.0f;
double max_cos_deviation = 0.0f; double max_cos_deviation = 0.0f;
com_fixed_t sincos_accumulator = -COM_FIXED_PI * 4; com_fix_t sincos_accumulator = -COM_FIX_PI * 4;
com_fixed_t sincos_test_step = (COM_FIXED_PI << 1) >> 10; com_fix_t sincos_test_step = (COM_FIX_PI << 1) >> 10;
for (int i = 0; i < 1024 * 32; ++i) { for (int i = 0; i < 1024 * 32; ++i) {
com_fixed_t sin, cos; com_fix_t sin, cos;
com_fixed_sincos(sincos_accumulator, &sin, &cos); com_fix_sincos(sincos_accumulator, &sin, &cos);
double const sin_deviation = double const sin_deviation = fabs(
fabs(sin(com_fixed_as_float(sincos_accumulator)) - sin(com_fix_as_float(sincos_accumulator)) - com_fix_as_float(sin));
com_fixed_as_float(sin)); double const cos_deviation = fabs(
double const cos_deviation = cos(com_fix_as_float(sincos_accumulator)) - com_fix_as_float(cos));
fabs(cos(com_fixed_as_float(sincos_accumulator)) -
com_fixed_as_float(cos));
if (sin_deviation > max_sin_deviation) if (sin_deviation > max_sin_deviation)
max_sin_deviation = sin_deviation; max_sin_deviation = sin_deviation;
if (cos_deviation > max_cos_deviation) if (cos_deviation > max_cos_deviation)
+142
View File
@@ -0,0 +1,142 @@
/*
Doom-inspired Q15.16 format used for most of everything.
https://hackmd.io/9uRB9YbTSBW4Qz_2b8TyDw#Examples-2-DOOM
*/
#ifndef COM_FIX_H
#define COM_FIX_H
#include <stdint.h>
#include <stdio.h>
#define COM_FIX_FRACBITS 16
#define COM_FIX_FRACUNIT (1 << COM_FIX_FRACBITS)
#define COM_FIX_FRACHALF (COM_FIX_FRACUNIT >> 1)
#define COM_FIX_FRACQRTR (COM_FIX_FRACHALF >> 1)
/* Calculated from one number, in hopes that match will converge. */
#define COM_FIX_PI ((com_fix_t)205887)
#define COM_FIX_PI2 (COM_FIX_PI << 1)
#define COM_FIX_PIHALF (COM_FIX_PI >> 1)
#define COM_FIX_PIONEANDAHALF (COM_FIX_PI + COM_FIX_PIHALF)
typedef int32_t com_fix_t;
extern const uint16_t com_fix_sin_lut[128];
void com_fix_print(com_fix_t a);
void com_fix_run_tests(void);
void com_fix_run_bench(void);
static inline com_fix_t com_fix_add(com_fix_t a, com_fix_t b) { return a + b; }
static inline com_fix_t com_fix_sub(com_fix_t a, com_fix_t b) { return a - b; }
static inline com_fix_t com_fix_mul(com_fix_t a, com_fix_t b) {
return (com_fix_t)(((int64_t)a * (int64_t)b) >> COM_FIX_FRACBITS);
}
/* Note: this does not clamp for over/underflow cases */
static inline com_fix_t com_fix_div(com_fix_t a, com_fix_t b) {
return (com_fix_t)(((int64_t)a << COM_FIX_FRACBITS) / (int64_t)b);
}
/* Is supposed to be only used for debugging, not in real code. */
/* Because of that we default to maximum precision here. */
static inline double com_fix_as_float(com_fix_t a) {
return (a >> COM_FIX_FRACBITS) * (double)1.0 +
(a & 0xFFFF) * (double)(1.0 / COM_FIX_FRACUNIT);
}
/* Branchless floor operation. */
/* TODO: Check against simpler branching implementation. */
static inline com_fix_t com_fix_floor(com_fix_t const a) {
int32_t const sign = a >> 31;
int32_t const frac_mask = (1 << COM_FIX_FRACBITS) - 1;
int32_t const has_fraction = ((a & frac_mask) != 0);
int32_t const truncated = a & ~frac_mask;
return truncated - ((sign & has_fraction) << COM_FIX_FRACBITS);
}
/* Approximated square root by Newton-Raphson in 2 iterations over small LUT. */
com_fix_t com_fix_sqrt(com_fix_t a);
/* Approximated using LUT. */
/* TODO: test whether linearly interpolated version could be default. */
/* things like camera movement can be jarring if done without.*/
/* https://namoseley.wordpress.com/2015/07/26/sincos-generation-using-table-lookup-and-iterpolation/
*/
static inline com_fix_t com_fix_sin(com_fix_t q) {
/* Wrap to (-Pi2,+Pi2) */
q %= COM_FIX_PI2;
/* Wrap to [0,+Pi2), see: -Pi1/4 = +Pi3/4 */
if (q < 0)
q += COM_FIX_PI2;
/* Limit to [0,+1) */
q = com_fix_div(q, COM_FIX_PI2);
/* Handle cases of 3rd and 4th quadrant separately. */
if (q >= COM_FIX_FRACHALF) {
q %= COM_FIX_FRACHALF;
uint16_t const idx = q >= COM_FIX_FRACQRTR ? 255 - (q >> 7) : q >> 7;
return -com_fix_sin_lut[idx];
} else {
uint16_t const idx = q >= COM_FIX_FRACQRTR ? 255 - (q >> 7) : q >> 7;
return com_fix_sin_lut[idx];
}
}
/* Implemented over sin, as to share one single LUT. */
/* Additionally, optimizer probably can collapse some math for combined sincos
* case. */
static inline com_fix_t com_fix_cos(com_fix_t a) {
return com_fix_sin(a + COM_FIX_PIHALF);
}
/* Combines calculation of both in a more optimal way. */
static inline void com_fix_sincos(com_fix_t q, com_fix_t *s, com_fix_t *c) {
q %= COM_FIX_PI2;
if (q < 0)
q += COM_FIX_PI2;
q = com_fix_div(q, COM_FIX_PI2);
/* Sadly, compilers are dumb and branching case is determined to be
* significantly faster in profiling. */
#define CASE(m_sin_sign, m_cos_sign) \
uint16_t const idx = q >= COM_FIX_FRACQRTR ? 255 - (q >> 7) : q >> 7; \
*s = m_sin_sign * com_fix_sin_lut[idx]; \
*c = m_cos_sign * com_fix_sin_lut[((uint8_t)127 - (uint8_t)idx) % 128]
/* Handle cases of 2rd and 3th quadrant separately, for minus cos. */
if (q >= COM_FIX_FRACHALF / 2 &&
q < COM_FIX_FRACHALF + COM_FIX_FRACHALF / 2) {
/* Handle cases of 3rd and 4th quadrant separately, for minus sin. */
if (q >= COM_FIX_FRACHALF) {
q %= COM_FIX_FRACHALF;
CASE(-1, -1);
} else {
CASE(+1, -1);
}
} else {
/* Handle cases of 3rd and 4th quadrant separately, for minus sin. */
if (q >= COM_FIX_FRACHALF) {
q %= COM_FIX_FRACHALF;
CASE(-1, +1);
} else {
CASE(+1, +1);
}
}
#undef CASE
}
static inline com_fix_t com_fix_tan(com_fix_t a) {
com_fix_t s, c;
com_fix_sincos(a, &s, &c);
return com_fix_div(s, c);
}
#endif
-146
View File
@@ -1,146 +0,0 @@
/*
Doom-inspired Q15.16 format used for most of everything.
https://hackmd.io/9uRB9YbTSBW4Qz_2b8TyDw#Examples-2-DOOM
*/
#ifndef COM_FIXED_H
#define COM_FIXED_H
#include <stdint.h>
#include <stdio.h>
#define COM_FIXED_FRACBITS 16
#define COM_FIXED_FRACUNIT (1 << COM_FIXED_FRACBITS)
#define COM_FIXED_FRACHALF (COM_FIXED_FRACUNIT >> 1)
#define COM_FIXED_FRACQRTR (COM_FIXED_FRACHALF >> 1)
#define COM_FIXED_PI (com_fixed_t)205887
#define COM_FIXED_PI2 (com_fixed_t)411774
#define COM_FIXED_PIHALF (com_fixed_t)102943
#define COM_FIXED_PIONEANDAHALF (COM_FIXED_PI + COM_FIXED_PIHALF)
typedef int32_t com_fixed_t;
extern const uint16_t com_fixed_sin_lut[128];
void com_fixed_print(com_fixed_t a);
void com_fixed_run_tests(void);
void com_fixed_run_bench(void);
static inline com_fixed_t com_fixed_add(com_fixed_t a, com_fixed_t b) {
return a + b;
}
static inline com_fixed_t com_fixed_sub(com_fixed_t a, com_fixed_t b) {
return a - b;
}
static inline com_fixed_t com_fixed_mul(com_fixed_t a, com_fixed_t b) {
return (com_fixed_t)(((int64_t)a * (int64_t)b) >> COM_FIXED_FRACBITS);
}
/* Note: this does not clamp for over/underflow cases */
static inline com_fixed_t com_fixed_div(com_fixed_t a, com_fixed_t b) {
return (com_fixed_t)(((int64_t)a << COM_FIXED_FRACBITS) / (int64_t)b);
}
/* Is supposed to be only used for debugging, not in real code. */
/* Because of that we default to maximum precision here. */
static inline double com_fixed_as_float(com_fixed_t a) {
return (a >> COM_FIXED_FRACBITS) * (double)1.0 +
(a & 0xFFFF) * (double)(1.0 / COM_FIXED_FRACUNIT);
}
/* Branchless floor operation. */
/* TODO: Check against simpler branching implementation. */
static inline com_fixed_t com_fixed_floor(com_fixed_t const a) {
int32_t const sign = a >> 31;
int32_t const frac_mask = (1 << COM_FIXED_FRACBITS) - 1;
int32_t const has_fraction = ((a & frac_mask) != 0);
int32_t const truncated = a & ~frac_mask;
return truncated - ((sign & has_fraction) << COM_FIXED_FRACBITS);
}
/* Approximated square root by Newton-Raphson in 2 iterations over small LUT. */
com_fixed_t com_fixed_sqrt(com_fixed_t a);
/* Approximated using LUT. */
/* TODO: test whether linearly interpolated version could be default. */
/* things like camera movement can be jarring if done without.*/
/* https://namoseley.wordpress.com/2015/07/26/sincos-generation-using-table-lookup-and-iterpolation/
*/
static inline com_fixed_t com_fixed_sin(com_fixed_t q) {
/* Wrap to (-Pi2,+Pi2) */
q %= COM_FIXED_PI2;
/* Wrap to [0,+Pi2) */
if (q < 0)
q += COM_FIXED_PI2;
/* Limit to [0,+1) */
q = com_fixed_div(q, COM_FIXED_PI2);
/* Handle cases of 3rd and 4th quadrant separately. */
if (q >= COM_FIXED_FRACHALF) {
q %= COM_FIXED_FRACHALF;
uint16_t const idx = q >= COM_FIXED_FRACQRTR ? 255 - (q >> 7) : q >> 7;
return -com_fixed_sin_lut[idx];
} else {
uint16_t const idx = q >= COM_FIXED_FRACQRTR ? 255 - (q >> 7) : q >> 7;
return com_fixed_sin_lut[idx];
}
}
/* Implemented over sin, as to share one single LUT. */
/* Additionally, optimizer probably can collapse some math for combined sincos
* case. */
static inline com_fixed_t com_fixed_cos(com_fixed_t a) {
return com_fixed_sin(a + COM_FIXED_PIHALF);
}
/* Combines calculation of both in a more optimal way. */
static inline void com_fixed_sincos(com_fixed_t q, com_fixed_t *s,
com_fixed_t *c) {
q %= COM_FIXED_PI2;
if (q < 0)
q += COM_FIXED_PI2;
q = com_fixed_div(q, COM_FIXED_PI2);
/* Sadly, compilers are dumb and branching case is determined to be
* significantly faster in profiling. */
#define CASE(m_sin_sign, m_cos_sign) \
uint16_t const idx = q >= COM_FIXED_FRACQRTR ? 255 - (q >> 7) : q >> 7; \
*s = m_sin_sign * com_fixed_sin_lut[idx]; \
*c = m_cos_sign * com_fixed_sin_lut[((uint8_t)127 - (uint8_t)idx) % 128]
/* Handle cases of 2rd and 3th quadrant separately, for minus cos. */
if (q >= COM_FIXED_FRACHALF / 2 &&
q < COM_FIXED_FRACHALF + COM_FIXED_FRACHALF / 2) {
/* Handle cases of 3rd and 4th quadrant separately, for minus sin. */
if (q >= COM_FIXED_FRACHALF) {
q %= COM_FIXED_FRACHALF;
CASE(-1, -1);
} else {
CASE(+1, -1);
}
} else {
/* Handle cases of 3rd and 4th quadrant separately, for minus sin. */
if (q >= COM_FIXED_FRACHALF) {
q %= COM_FIXED_FRACHALF;
CASE(-1, 1);
} else {
CASE(+1, 1);
}
}
#undef CASE
}
static inline com_fixed_t com_fixed_tan(com_fixed_t a) {
com_fixed_t s, c;
com_fixed_sincos(a, &s, &c);
return com_fixed_div(s, c);
}
#endif
+58
View File
@@ -0,0 +1,58 @@
/*
Geometry rendering.
*/
#ifndef COM_GEM_H
#define COM_GEM_H
#include "mat.h"
#include <stdint.h>
/* Projects vertex position to a clip space via reordered row-major MVP matrix.
It skips calculation of z depth component, as we assume to never use it.
Instead, 3rd returned component is w value, for future projection.
We must never render from inside geometry, as it will break no Z clipping
assumption (see com_gem_clip_vis_project()).
*/
static inline com_vec_t com_gem_vec_project_clip(com_mat_t a, com_vec_t b) {
com_vec_t result;
#define CASE(m_c, m_n) \
result.a[m_c] = (((int64_t)a.a[m_n * 4 + 0] * b.s.x) + \
((int64_t)a.a[m_n * 4 + 1] * b.s.y) + \
((int64_t)a.a[m_n * 4 + 2] * b.s.z) + a.a[m_n * 4 + 3]) >> \
COM_FIX_FRACBITS;
CASE(0, 0);
CASE(1, 1);
CASE(2, 3);
#undef CASE
return result;
}
/* TODO: Move to separate render-specific file. */
/* TODO: If we clip test bounding volume of a model first we can skip
* all clipping whatsoever, which might hold true more often, than the cost of
* testing the volume. */
#define COM_GEM_VERTEX_UNCLIPPED (0 << 0)
#define COM_GEM_VERTEX_CLIPPED_X (1 << 0)
#define COM_GEM_VERTEX_CLIPPED_Y (1 << 1)
/* Attempt to project clip space vertex to NDC, reporting which component lies
* outside of view. Bit test against those per component. This is needed for
* determining new view-lying clipped triangles. */
static inline uint8_t com_gem_clip_vis_project(com_vec_t a, com_vec_t *out) {
uint8_t mask;
if (a.s.x < -a.s.z || a.s.x > a.s.z)
mask ^= COM_GEM_VERTEX_CLIPPED_X;
if (a.s.y < -a.s.z || a.s.y > a.s.z)
mask ^= COM_GEM_VERTEX_CLIPPED_Y;
out->a[0] = com_fix_div(a.a[0], a.a[2]);
out->a[1] = com_fix_div(a.a[1], a.a[2]);
return mask;
}
#endif
+19
View File
@@ -0,0 +1,19 @@
#include "mat.h"
#include "Timer/timer.h"
/* TODO: Doesn't work, clang optimizes it away. We need RNG here. */
void com_mat_run_bench(void) {
uint64_t start = com_timer_count_ns();
for (int i = 5000000; i--;) {
com_mat_t a = com_mat_identity();
com_mat_t b = com_mat_identity();
a.a[3] = i * COM_FIX_FRACUNIT;
b.a[9] = i * COM_FIX_FRACUNIT;
COM_MAT_PROFILE_SINK(a);
COM_MAT_PROFILE_SINK(b);
com_mat_t mul = com_mat_mul(a, b);
COM_MAT_PROFILE_SINK(mul);
}
com_timer_profile(start, "mat: mul");
}
+30 -43
View File
@@ -16,22 +16,31 @@
#ifndef COM_MAT_H #ifndef COM_MAT_H
#define COM_MAT_H #define COM_MAT_H
#include "fixed.h" #include "def.h"
#include "fix.h"
#include "vec.h" #include "vec.h"
#include <stdint.h> #include <stdint.h>
#include <stdio.h> #include <stdio.h>
typedef union { typedef union {
com_fixed_t com_def_alignedas(64) a[4 * 4]; com_def_alignedas(64) com_def_vector(com_fix_t, a, 4 * 4);
} com_mat_t; } com_mat_t;
void com_mat_run_bench(void);
#define COM_MAT_PROFILE_SINK(m_m) \
do { \
for (int i = 0; i < 16; ++i) \
COM_DEF_PROFILE_SINK((m_m).a[i]); \
} while (0)
static inline com_mat_t com_mat_identity(void) { static inline com_mat_t com_mat_identity(void) {
com_mat_t result = {0}; com_mat_t result = {0};
result.a[0 * 4 + 0] = COM_FIXED_FRACUNIT; result.a[0 * 4 + 0] = COM_FIX_FRACUNIT;
result.a[1 * 4 + 1] = COM_FIXED_FRACUNIT; result.a[1 * 4 + 1] = COM_FIX_FRACUNIT;
result.a[2 * 4 + 2] = COM_FIXED_FRACUNIT; result.a[2 * 4 + 2] = COM_FIX_FRACUNIT;
result.a[3 * 4 + 3] = COM_FIXED_FRACUNIT; result.a[3 * 4 + 3] = COM_FIX_FRACUNIT;
return result; return result;
} }
@@ -43,7 +52,7 @@ static inline com_mat_t com_mat_mul(com_mat_t a, com_mat_t b) {
// for (int c = 0; c < 4; ++c) { // for (int c = 0; c < 4; ++c) {
// for (int k = 0; k < 4; ++k) { // for (int k = 0; k < 4; ++k) {
// for (int r = 0; r < 4; ++r) { // for (int r = 0; r < 4; ++r) {
// result.a[r + c * 4] += com_fixed_mul(a.a[r + k * 4], b.a[k + c * 4]); // result.a[r + c * 4] += com_fix_mul(a.a[r + k * 4], b.a[k + c * 4]);
// } // }
// } // }
// } // }
@@ -56,7 +65,7 @@ static inline com_mat_t com_mat_mul(com_mat_t a, com_mat_t b) {
((int64_t)a.a[r + 1 * 4] * b.a[1 + c * 4]) + ((int64_t)a.a[r + 1 * 4] * b.a[1 + c * 4]) +
((int64_t)a.a[r + 2 * 4] * b.a[2 + c * 4]) + ((int64_t)a.a[r + 2 * 4] * b.a[2 + c * 4]) +
((int64_t)a.a[r + 3 * 4] * b.a[3 + c * 4])) >> ((int64_t)a.a[r + 3 * 4] * b.a[3 + c * 4])) >>
COM_FIXED_FRACBITS; COM_FIX_FRACBITS;
} }
} }
@@ -76,34 +85,13 @@ static inline com_mat_t com_mat_mul_reodered(com_mat_t a, com_mat_t b) {
((int64_t)a.a[1 + r * 4] * b.a[1 + c * 4]) + ((int64_t)a.a[1 + r * 4] * b.a[1 + c * 4]) +
((int64_t)a.a[2 + r * 4] * b.a[2 + c * 4]) + ((int64_t)a.a[2 + r * 4] * b.a[2 + c * 4]) +
((int64_t)a.a[3 + r * 4] * b.a[3 + c * 4])) >> ((int64_t)a.a[3 + r * 4] * b.a[3 + c * 4])) >>
COM_FIXED_FRACBITS; COM_FIX_FRACBITS;
} }
} }
return result; return result;
} }
/* Projects vertex position to a screen via reordered row-major MVP matrix,
* which implies division by w in-place */
static inline com_vec_t com_mat_vec_project(com_mat_t a, com_vec_t b) {
com_fixed_t t[4];
com_vec_t result;
for (int c = 0; c < 4; ++c) {
t[c] =
(((int64_t)a.a[c * 4 + 0] * b.s.x) + ((int64_t)a.a[c * 4 + 1] * b.s.y) +
((int64_t)a.a[c * 4 + 2] * b.s.z) + a.a[c * 4 + 3]) >>
COM_FIXED_FRACBITS;
}
/* Creates perspective effect, could be skipped for orthographic */
result.a[0] = com_fixed_div(t[0], t[3]);
result.a[1] = com_fixed_div(t[1], t[3]);
result.a[2] = com_fixed_div(t[2], t[3]);
return result;
}
/* Slightly optimized case of assumed identity scaling, might be useful for MVP /* Slightly optimized case of assumed identity scaling, might be useful for MVP
* calculations, if model matrix does not scale. View matrix is always * calculations, if model matrix does not scale. View matrix is always
* unscaled as well. * unscaled as well.
@@ -116,7 +104,7 @@ static inline com_vec_t com_mat_vec_project(com_mat_t a, com_vec_t b) {
// result.a[0 * 4 + 3] = 0; // result.a[0 * 4 + 3] = 0;
// result.a[1 * 4 + 3] = 0; // result.a[1 * 4 + 3] = 0;
// result.a[2 * 4 + 3] = 0; // result.a[2 * 4 + 3] = 0;
// result.a[3 * 4 + 3] = COM_FIXED_FRACUNIT; // result.a[3 * 4 + 3] = COM_FIX_FRACUNIT;
// return result; // return result;
// } // }
@@ -158,7 +146,7 @@ static inline com_mat_t com_mat_look_at(com_vec_t pos, com_vec_t up,
result.a[12] = -com_vec_dot(r, pos); result.a[12] = -com_vec_dot(r, pos);
result.a[13] = -com_vec_dot(u, pos); result.a[13] = -com_vec_dot(u, pos);
result.a[14] = com_vec_dot(target, pos); result.a[14] = com_vec_dot(target, pos);
result.a[15] = COM_FIXED_FRACUNIT; result.a[15] = COM_FIX_FRACUNIT;
return result; return result;
} }
@@ -166,23 +154,22 @@ static inline com_mat_t com_mat_look_at(com_vec_t pos, com_vec_t up,
/* TODO: move to .c file */ /* TODO: move to .c file */
/* Produces a projection matrix needed for camera work. */ /* Produces a projection matrix needed for camera work. */
static inline com_mat_t com_mat_perspective(uint16_t rwidth, uint16_t rheight, static inline com_mat_t com_mat_perspective(uint16_t rwidth, uint16_t rheight,
com_fixed_t nearz, com_fixed_t farz, com_fix_t nearz, com_fix_t farz,
com_fixed_t fov) { com_fix_t fov) {
com_mat_t result = {0}; com_mat_t result = {0};
com_fixed_t const aspect = com_fix_t const aspect =
com_fixed_div(rwidth * COM_FIXED_FRACUNIT, rheight * COM_FIXED_FRACUNIT); com_fix_div(rwidth * COM_FIX_FRACUNIT, rheight * COM_FIX_FRACUNIT);
com_fixed_t const f = com_fix_t const f = com_fix_div(
com_fixed_div(COM_FIXED_FRACUNIT, COM_FIX_FRACUNIT, com_fix_tan(com_fix_mul(fov, COM_FIX_FRACHALF)));
com_fixed_tan(com_fixed_mul(fov, COM_FIXED_FRACHALF))); com_fix_t const fn = com_fix_div(COM_FIX_FRACUNIT, (nearz - farz));
com_fixed_t const fn = com_fixed_div(COM_FIXED_FRACUNIT, (nearz - farz));
result.a[0 * 4 + 0] = com_fixed_div(f, aspect); result.a[0 * 4 + 0] = com_fix_div(f, aspect);
result.a[1 * 4 + 1] = f; result.a[1 * 4 + 1] = f;
result.a[2 * 4 + 2] = (nearz + farz) * fn; result.a[2 * 4 + 2] = (nearz + farz) * fn;
result.a[2 * 4 + 3] = -COM_FIXED_FRACUNIT; result.a[2 * 4 + 3] = -COM_FIX_FRACUNIT;
result.a[3 * 4 + 2] = result.a[3 * 4 + 2] =
(COM_FIXED_FRACUNIT * 2) * com_fixed_mul(com_fixed_mul(nearz, farz), fn); (COM_FIX_FRACUNIT * 2) * com_fix_mul(com_fix_mul(nearz, farz), fn);
return result; return result;
} }
+36 -31
View File
@@ -8,63 +8,63 @@
#define COM_VEC_H #define COM_VEC_H
#include "def.h" #include "def.h"
#include "fixed.h" #include "fix.h"
#include <stdint.h> #include <stdint.h>
typedef union { typedef union {
com_fixed_t com_def_alignedas(16) a[3]; com_def_alignedas(16) com_def_vector(com_fix_t, a, 3);
com_def_alignedas(16) struct { com_def_alignedas(16) struct {
com_fixed_t x; com_fix_t x;
com_fixed_t y; com_fix_t y;
com_fixed_t z; com_fix_t z;
} s; } s;
} com_vec_t; } com_vec_t;
static inline com_vec_t com_vec_identity(void) { static inline com_vec_t com_vec_identity(void) {
return (com_vec_t){.s = {.x = COM_FIXED_FRACUNIT, return (com_vec_t){.s = {.x = COM_FIX_FRACUNIT,
.y = COM_FIXED_FRACUNIT, .y = COM_FIX_FRACUNIT,
.z = COM_FIXED_FRACUNIT}}; .z = COM_FIX_FRACUNIT}};
} }
static inline com_vec_t com_vec_add(com_vec_t a, com_vec_t b) { static inline com_vec_t com_vec_add(com_vec_t a, com_vec_t b) {
return (com_vec_t){.s = {.x = com_fixed_add(a.s.x, b.s.x), return (com_vec_t){.s = {.x = com_fix_add(a.s.x, b.s.x),
.y = com_fixed_add(a.s.y, b.s.y), .y = com_fix_add(a.s.y, b.s.y),
.z = com_fixed_add(a.s.z, b.s.z)}}; .z = com_fix_add(a.s.z, b.s.z)}};
} }
static inline com_vec_t com_vec_sub(com_vec_t a, com_vec_t b) { static inline com_vec_t com_vec_sub(com_vec_t a, com_vec_t b) {
return (com_vec_t){.s = {.x = com_fixed_sub(a.s.x, b.s.x), return (com_vec_t){.s = {.x = com_fix_sub(a.s.x, b.s.x),
.y = com_fixed_sub(a.s.y, b.s.y), .y = com_fix_sub(a.s.y, b.s.y),
.z = com_fixed_sub(a.s.z, b.s.z)}}; .z = com_fix_sub(a.s.z, b.s.z)}};
} }
static inline com_vec_t com_vec_mul(com_vec_t a, com_vec_t b) { static inline com_vec_t com_vec_mul(com_vec_t a, com_vec_t b) {
return (com_vec_t){.s = {.x = com_fixed_mul(a.s.x, b.s.x), return (com_vec_t){.s = {.x = com_fix_mul(a.s.x, b.s.x),
.y = com_fixed_mul(a.s.y, b.s.y), .y = com_fix_mul(a.s.y, b.s.y),
.z = com_fixed_mul(a.s.z, b.s.z)}}; .z = com_fix_mul(a.s.z, b.s.z)}};
} }
/* Note: this does not clamp for over/underflow cases */ /* Note: this does not clamp for over/underflow cases */
static inline com_vec_t com_vec_div(com_vec_t a, com_vec_t b) { static inline com_vec_t com_vec_div(com_vec_t a, com_vec_t b) {
return (com_vec_t){.s = {.x = com_fixed_div(a.s.x, b.s.x), return (com_vec_t){.s = {.x = com_fix_div(a.s.x, b.s.x),
.y = com_fixed_div(a.s.y, b.s.y), .y = com_fix_div(a.s.y, b.s.y),
.z = com_fixed_div(a.s.z, b.s.z)}}; .z = com_fix_div(a.s.z, b.s.z)}};
} }
/* Scale vector by a fixed point number */ /* Scale vector by a fixed point number */
static inline com_vec_t com_vec_scl(com_vec_t a, com_fixed_t b) { static inline com_vec_t com_vec_scl(com_vec_t a, com_fix_t b) {
return (com_vec_t){.s = {.x = com_fixed_mul(a.s.x, b), return (com_vec_t){.s = {.x = com_fix_mul(a.s.x, b),
.y = com_fixed_mul(a.s.y, b), .y = com_fix_mul(a.s.y, b),
.z = com_fixed_mul(a.s.z, b)}}; .z = com_fix_mul(a.s.z, b)}};
} }
/* Shows how much given vectors are correlated in direction to each other */ /* Shows how much given vectors are correlated in direction to each other */
/* Resulted range depends on input, it's in -1 to 1 for normalized /* Resulted range depends on input, it's in -1 to 1 for normalized
* input and otherwise is -ab to +ab */ * input and otherwise is -ab to +ab */
static inline com_fixed_t com_vec_dot(com_vec_t a, com_vec_t b) { static inline com_fix_t com_vec_dot(com_vec_t a, com_vec_t b) {
return (((int64_t)a.s.x * b.s.x) + ((int64_t)a.s.y * b.s.y) + return (((int64_t)a.s.x * b.s.x) + ((int64_t)a.s.y * b.s.y) +
((int64_t)a.s.z * b.s.z)) >> ((int64_t)a.s.z * b.s.z)) >>
COM_FIXED_FRACBITS; COM_FIX_FRACBITS;
} }
/* Cross product produces a perpendicular for normalized vectors, or 0 for /* Cross product produces a perpendicular for normalized vectors, or 0 for
@@ -73,14 +73,19 @@ static inline com_vec_t com_vec_crs(com_vec_t a, com_vec_t b) {
int64_t const cx = ((int64_t)a.s.y * b.s.z) - ((int64_t)a.s.z - b.s.y); int64_t const cx = ((int64_t)a.s.y * b.s.z) - ((int64_t)a.s.z - b.s.y);
int64_t const cy = ((int64_t)a.s.z * b.s.x) - ((int64_t)a.s.x - b.s.z); int64_t const cy = ((int64_t)a.s.z * b.s.x) - ((int64_t)a.s.x - b.s.z);
int64_t const cz = ((int64_t)a.s.x * b.s.y) - ((int64_t)a.s.y - b.s.x); int64_t const cz = ((int64_t)a.s.x * b.s.y) - ((int64_t)a.s.y - b.s.x);
return (com_vec_t){.s = {.x = (com_fixed_t)(cx >> COM_FIXED_FRACBITS), return (com_vec_t){.s = {.x = (com_fix_t)(cx >> COM_FIX_FRACBITS),
.y = (com_fixed_t)(cy >> COM_FIXED_FRACBITS), .y = (com_fix_t)(cy >> COM_FIX_FRACBITS),
.z = (com_fixed_t)(cz >> COM_FIXED_FRACBITS)}}; .z = (com_fix_t)(cz >> COM_FIX_FRACBITS)}};
} }
/* Normalize vector, making it a unit one (of length 1). */
/* Pretty expensive, make sure you actually need it. */
static inline com_vec_t com_vec_nrm(com_vec_t a) { static inline com_vec_t com_vec_nrm(com_vec_t a) {
com_fixed_t const n = com_fixed_sqrt(com_vec_dot(a, a)); com_fix_t const n = com_fix_sqrt(com_vec_dot(a, a));
return com_vec_scl(a, com_fixed_div(COM_FIXED_FRACUNIT, n)); // return com_vec_scl(a, com_fix_div(COM_FIX_FRACUNIT, n));
return (com_vec_t){.s = {.x = com_fix_div(a.s.x, n),
.y = com_fix_div(a.s.y, n),
.z = com_fix_div(a.s.z, n)}};
} }
#endif #endif
-2
View File
@@ -4,8 +4,6 @@ import math
def generate_pi(): def generate_pi():
print(f"#define COM_FIXED_PI (com_fixed_t){int(math.pi * (1 << 16))}\n") print(f"#define COM_FIXED_PI (com_fixed_t){int(math.pi * (1 << 16))}\n")
print(f"#define COM_FIXED_PI2 (com_fixed_t){int(math.pi*2 * (1 << 16))}\n")
print(f"#define COM_FIXED_PIHALF (com_fixed_t){int(math.pi/2 * (1 << 16))}\n")
# Idea is to generate only one quadrant, as the values repeat later with different sign and order. # Idea is to generate only one quadrant, as the values repeat later with different sign and order.
def generate_sin_table(n_entries=128): def generate_sin_table(n_entries=128):
+8 -7
View File
@@ -31,18 +31,19 @@ extern int plr_display_x11_main(int argc, char *argv[]) {
com_mat_t m0 = com_mat_identity(); com_mat_t m0 = com_mat_identity();
com_mat_t m1 = {1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16}; com_mat_t m1 = {1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16};
for (int i = 0; i < 16; ++i) for (int i = 0; i < 16; ++i)
m1.a[i] *= COM_FIXED_FRACUNIT; m1.a[i] *= COM_FIX_FRACUNIT;
com_mat_t t0 = com_mat_mul(m1, m0); com_mat_t t0 = com_mat_mul(m1, m0);
com_mat_t t1 = com_mat_mul_reodered(com_mat_reoder(m1), m0); com_mat_t t1 = com_mat_mul_reodered(com_mat_reoder(m1), m0);
com_fixed_print(t0.a[4]); com_fix_print(t0.a[4]);
com_fixed_print(t1.a[4]); com_fix_print(t1.a[4]);
com_vec_t v0 = com_mat_vec_project(com_mat_reoder(m0), com_vec_identity()); // com_vec_t v0 = com_mat_vec_project(com_mat_reoder(m0), com_vec_identity());
com_fixed_print(v0.a[0]); // com_fix_print(v0.a[0]);
com_fixed_run_tests(); com_fix_run_tests();
com_fixed_run_bench(); com_fix_run_bench();
com_mat_run_bench();
Display *display = XOpenDisplay(NULL); Display *display = XOpenDisplay(NULL);
if (NULL == display) { if (NULL == display) {
+2 -2
View File
@@ -1,10 +1,10 @@
CC=clang CC=clang
# todo: Only link to libm on debug. # todo: Only link to libm on debug.
CFLAGS=-lX11 -Wall -std=c99 -g3 -O3 -msse2 -lm CFLAGS=-lX11 -Wall -std=c99 -g3 -O3 -msse2 -lm
DEPS = ../Common/lzw.h ../Common/mat.h ../Common/vec.h ../Common/fixed.h ../Common/def.h ../Common/bits.h \ DEPS = ../Common/lzw.h ../Common/mat.h ../Common/vec.h ../Common/fix.h ../Common/def.h ../Common/bits.h \
../Common/Timer/timer.h \ ../Common/Timer/timer.h \
./Display/display.h ./Display/display.h
OBJ = ../Common/lzw.o ../Common/fixed.o OBJ = ../Common/lzw.o ../Common/fix.o ../Common/mat.o
LINUX = main.o Display/x11.o ../Common/Timer/unix.o LINUX = main.o Display/x11.o ../Common/Timer/unix.o
DOS = maindos.c Display/dos.c DOS = maindos.c Display/dos.c