benches, optimizations for sin and cos
This commit is contained in:
@@ -0,0 +1,14 @@
|
||||
#ifndef COM_TIMER_H
|
||||
#define COM_TIMER_H
|
||||
|
||||
#include "../def.h"
|
||||
#include <stdint.h>
|
||||
|
||||
/* Global monotonic counter, meaning it always increases. Not applicable for
|
||||
* getting "real" time, which the game shouldn't be aware of anyways. */
|
||||
uint64_t com_timer_count_ns(void);
|
||||
|
||||
/* Output elapsed time against previous com_timer_count_ns() result. */
|
||||
void com_timer_profile(uint64_t starttime, const char *what);
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,17 @@
|
||||
#include "timer.h"
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#define __USE_POSIX199309 1
|
||||
#include <time.h>
|
||||
|
||||
uint64_t com_timer_count_ns(void) {
|
||||
struct timespec time;
|
||||
clock_gettime(CLOCK_MONOTONIC, &time);
|
||||
return time.tv_sec * 1000000000 + time.tv_nsec;
|
||||
}
|
||||
|
||||
void com_timer_profile(uint64_t starttime, const char *what) {
|
||||
uint64_t const endtime = com_timer_count_ns();
|
||||
uint64_t const diff = endtime - starttime;
|
||||
printf("profile \"%s\" took %fms\n", what, (double)diff / 1000000);
|
||||
}
|
||||
@@ -10,10 +10,13 @@
|
||||
#define COM_DEF_COMPILE_MODERN 1
|
||||
#endif
|
||||
|
||||
/* TODO: uppercase? */
|
||||
#ifdef COM_DEF_COMPILE_MODERN
|
||||
#define com_def_alignedas(v_as) _Alignas(v_as)
|
||||
#else
|
||||
#define com_def_alignedas(v_as)
|
||||
#endif
|
||||
|
||||
#define COM_DEF_PROFILE_VAR volatile
|
||||
|
||||
#endif
|
||||
|
||||
+76
-10
@@ -4,6 +4,8 @@
|
||||
*/
|
||||
|
||||
#include "fixed.h"
|
||||
#include "Timer/timer.h"
|
||||
#include "def.h"
|
||||
#include <assert.h>
|
||||
#include <stdbool.h>
|
||||
#include <stdint.h>
|
||||
@@ -16,6 +18,7 @@ static const uint32_t com_fixed_sqrt_lut[16] = {
|
||||
46340, 49547, 52521, 55314, 57954, 60464, 62862, 65161,
|
||||
67373, 69508, 71572, 73572, 75514, 77402, 79240, 81033};
|
||||
|
||||
/* TODO: Test performance of uint_fast16_t here. */
|
||||
/* Used for other quadrants as well as cosine eval, all from the same table. */
|
||||
/* Only fractional part is present, reducing the cache footprint. */
|
||||
const uint16_t com_fixed_sin_lut[128] = {
|
||||
@@ -92,35 +95,73 @@ void com_fixed_print(com_fixed_t a) {
|
||||
printf("%d.%04u\n", int_part, decimal_val);
|
||||
}
|
||||
|
||||
void com_fixed_run_bench(void) {
|
||||
uint64_t start = com_timer_count_ns();
|
||||
for (int i = 5000000; i--;) {
|
||||
COM_DEF_PROFILE_VAR com_fixed_t sqrt = com_fixed_sqrt(i);
|
||||
}
|
||||
com_timer_profile(start, "fixed: sqrt");
|
||||
|
||||
start = com_timer_count_ns();
|
||||
for (int i = 50000000; i--;) {
|
||||
COM_DEF_PROFILE_VAR com_fixed_t sin = com_fixed_sin(i);
|
||||
}
|
||||
com_timer_profile(start, "fixed: sin");
|
||||
|
||||
start = com_timer_count_ns();
|
||||
for (int i = 50000000; i--;) {
|
||||
COM_DEF_PROFILE_VAR com_fixed_t cos = com_fixed_cos(i);
|
||||
}
|
||||
com_timer_profile(start, "fixed: cos");
|
||||
|
||||
start = com_timer_count_ns();
|
||||
for (int i = 50000000; i--;) {
|
||||
com_fixed_t sin, cos;
|
||||
com_fixed_sincos(i, &sin, &cos);
|
||||
COM_DEF_PROFILE_VAR com_fixed_t sinv = sin;
|
||||
COM_DEF_PROFILE_VAR com_fixed_t cosv = cos;
|
||||
}
|
||||
com_timer_profile(start, "fixed: sincos");
|
||||
}
|
||||
|
||||
void com_fixed_run_tests(void) {
|
||||
com_fixed_print(com_fixed_sqrt(COM_FIXED_FRACUNIT * 2));
|
||||
com_fixed_print(com_fixed_sqrt(COM_FIXED_FRACUNIT * 3));
|
||||
com_fixed_print(com_fixed_sqrt(COM_FIXED_FRACUNIT * 4));
|
||||
com_fixed_print(com_fixed_sqrt(COM_FIXED_FRACUNIT * 200));
|
||||
com_fixed_print(com_fixed_sqrt(COM_FIXED_FRACUNIT * 1000));
|
||||
|
||||
com_fixed_print(COM_FIXED_PI);
|
||||
com_fixed_print(com_fixed_div(COM_FIXED_PI + 234, COM_FIXED_PI2) << 2);
|
||||
{
|
||||
double max_sqrt_deviation = 0.0f;
|
||||
com_fixed_t sqrt_accumulator = 0;
|
||||
com_fixed_t sqrt_test_step = COM_FIXED_FRACUNIT >> 2; /* Quarter step */
|
||||
for (int i = 0; i < COM_FIXED_FRACUNIT << 1; ++i) {
|
||||
com_fixed_t sqrt = com_fixed_sqrt(sqrt_accumulator);
|
||||
double const deviation =
|
||||
sqrt(com_fixed_as_float(sqrt_accumulator)) - com_fixed_as_float(sqrt);
|
||||
if (deviation > max_sqrt_deviation)
|
||||
max_sqrt_deviation = deviation;
|
||||
sqrt_accumulator += sqrt_test_step;
|
||||
}
|
||||
printf("max sqrt deviation: %f\n", max_sqrt_deviation);
|
||||
}
|
||||
|
||||
{
|
||||
double max_sin_deviation = 0.0f;
|
||||
com_fixed_t sin_accumulator = -COM_FIXED_PI * 4;
|
||||
com_fixed_t sin_test_step = (COM_FIXED_PI << 1) >> 10;
|
||||
for (int i = 0; i < 1024 * 32; ++i) {
|
||||
com_fixed_t sin = com_fixed_sin(sin_accumulator);
|
||||
double deviation = fabs(sin(com_fixed_as_float(sin_accumulator)) -
|
||||
double const deviation = fabs(sin(com_fixed_as_float(sin_accumulator)) -
|
||||
com_fixed_as_float(sin));
|
||||
if (deviation > max_sin_deviation)
|
||||
max_sin_deviation = deviation;
|
||||
sin_accumulator += sin_test_step;
|
||||
}
|
||||
printf("max sin deviation: %f\n", max_sin_deviation);
|
||||
}
|
||||
|
||||
{
|
||||
double max_cos_deviation = 0.0f;
|
||||
com_fixed_t cos_accumulator = -COM_FIXED_PI * 4;
|
||||
com_fixed_t cos_test_step = (COM_FIXED_PI << 1) >> 10;
|
||||
for (int i = 0; i < 1024 * 32; ++i) {
|
||||
com_fixed_t cos = com_fixed_cos(cos_accumulator);
|
||||
double deviation = fabs(cos(com_fixed_as_float(cos_accumulator)) -
|
||||
double const deviation = fabs(cos(com_fixed_as_float(cos_accumulator)) -
|
||||
com_fixed_as_float(cos));
|
||||
if (deviation > max_cos_deviation)
|
||||
max_cos_deviation = deviation;
|
||||
@@ -128,3 +169,28 @@ void com_fixed_run_tests(void) {
|
||||
}
|
||||
printf("max cos deviation: %f\n", max_cos_deviation);
|
||||
}
|
||||
|
||||
{
|
||||
double max_sin_deviation = 0.0f;
|
||||
double max_cos_deviation = 0.0f;
|
||||
com_fixed_t sincos_accumulator = -COM_FIXED_PI * 4;
|
||||
com_fixed_t sincos_test_step = (COM_FIXED_PI << 1) >> 10;
|
||||
for (int i = 0; i < 1024 * 32; ++i) {
|
||||
com_fixed_t sin, cos;
|
||||
com_fixed_sincos(sincos_accumulator, &sin, &cos);
|
||||
double const sin_deviation =
|
||||
fabs(sin(com_fixed_as_float(sincos_accumulator)) -
|
||||
com_fixed_as_float(sin));
|
||||
double const cos_deviation =
|
||||
fabs(cos(com_fixed_as_float(sincos_accumulator)) -
|
||||
com_fixed_as_float(cos));
|
||||
if (sin_deviation > max_sin_deviation)
|
||||
max_sin_deviation = sin_deviation;
|
||||
if (cos_deviation > max_cos_deviation)
|
||||
max_cos_deviation = cos_deviation;
|
||||
sincos_accumulator += sincos_test_step;
|
||||
}
|
||||
printf("max sincos deviations: %f %f\n", max_sin_deviation,
|
||||
max_cos_deviation);
|
||||
}
|
||||
}
|
||||
|
||||
+65
-19
@@ -23,6 +23,11 @@ typedef int32_t com_fixed_t;
|
||||
|
||||
extern const uint16_t com_fixed_sin_lut[128];
|
||||
|
||||
void com_fixed_print(com_fixed_t a);
|
||||
|
||||
void com_fixed_run_tests(void);
|
||||
void com_fixed_run_bench(void);
|
||||
|
||||
static inline com_fixed_t com_fixed_add(com_fixed_t a, com_fixed_t b) {
|
||||
return a + b;
|
||||
}
|
||||
@@ -47,6 +52,16 @@ static inline double com_fixed_as_float(com_fixed_t a) {
|
||||
(a & 0xFFFF) * (double)(1.0 / COM_FIXED_FRACUNIT);
|
||||
}
|
||||
|
||||
/* Branchless floor operation. */
|
||||
/* TODO: Check against simpler branching implementation. */
|
||||
static inline com_fixed_t com_fixed_floor(com_fixed_t const a) {
|
||||
int32_t const sign = a >> 31;
|
||||
int32_t const frac_mask = (1 << COM_FIXED_FRACBITS) - 1;
|
||||
int32_t const has_fraction = ((a & frac_mask) != 0);
|
||||
int32_t const truncated = a & ~frac_mask;
|
||||
return truncated - ((sign & has_fraction) << COM_FIXED_FRACBITS);
|
||||
}
|
||||
|
||||
/* Approximated square root by Newton-Raphson in 2 iterations over small LUT. */
|
||||
com_fixed_t com_fixed_sqrt(com_fixed_t a);
|
||||
|
||||
@@ -55,29 +70,26 @@ com_fixed_t com_fixed_sqrt(com_fixed_t a);
|
||||
/* things like camera movement can be jarring if done without.*/
|
||||
/* https://namoseley.wordpress.com/2015/07/26/sincos-generation-using-table-lookup-and-iterpolation/
|
||||
*/
|
||||
static inline com_fixed_t com_fixed_sin(com_fixed_t a) {
|
||||
int32_t sign = 1;
|
||||
|
||||
static inline com_fixed_t com_fixed_sin(com_fixed_t q) {
|
||||
/* Wrap to (-Pi2,+Pi2) */
|
||||
com_fixed_t q = a % COM_FIXED_PI2;
|
||||
q %= COM_FIXED_PI2;
|
||||
|
||||
/* Wrap to [0,+Pi2) */
|
||||
if (q < 0)
|
||||
q = COM_FIXED_PI2 + q;
|
||||
q += COM_FIXED_PI2;
|
||||
|
||||
/* Limit to [0,+1) */
|
||||
q = com_fixed_div(q, COM_FIXED_PI2);
|
||||
|
||||
/* Handle cases of 3rd and 4th quadrant. */
|
||||
/* Handle cases of 3rd and 4th quadrant separately. */
|
||||
if (q >= COM_FIXED_FRACHALF) {
|
||||
q -= COM_FIXED_FRACHALF;
|
||||
sign = -1;
|
||||
}
|
||||
|
||||
/* Finally clculate the index into LUT. */
|
||||
q %= COM_FIXED_FRACHALF;
|
||||
uint16_t const idx = q >= COM_FIXED_FRACQRTR ? 255 - (q >> 7) : q >> 7;
|
||||
|
||||
return com_fixed_sin_lut[idx] * sign;
|
||||
return -com_fixed_sin_lut[idx];
|
||||
} else {
|
||||
uint16_t const idx = q >= COM_FIXED_FRACQRTR ? 255 - (q >> 7) : q >> 7;
|
||||
return com_fixed_sin_lut[idx];
|
||||
}
|
||||
}
|
||||
|
||||
/* Implemented over sin, as to share one single LUT. */
|
||||
@@ -87,14 +99,48 @@ static inline com_fixed_t com_fixed_cos(com_fixed_t a) {
|
||||
return com_fixed_sin(a + COM_FIXED_PIHALF);
|
||||
}
|
||||
|
||||
static inline void com_fixed_sincos(com_fixed_t a, com_fixed_t *restrict s,
|
||||
com_fixed_t *restrict c) {
|
||||
*s = com_fixed_sin(a);
|
||||
*c = com_fixed_cos(a);
|
||||
/* Combines calculation of both in a more optimal way. */
|
||||
static inline void com_fixed_sincos(com_fixed_t q, com_fixed_t *s,
|
||||
com_fixed_t *c) {
|
||||
q %= COM_FIXED_PI2;
|
||||
if (q < 0)
|
||||
q += COM_FIXED_PI2;
|
||||
q = com_fixed_div(q, COM_FIXED_PI2);
|
||||
|
||||
/* Sadly, compilers are dumb and branching case is determined to be
|
||||
* significantly faster in profiling. */
|
||||
#define CASE(m_sin_sign, m_cos_sign) \
|
||||
uint16_t const idx = q >= COM_FIXED_FRACQRTR ? 255 - (q >> 7) : q >> 7; \
|
||||
*s = m_sin_sign * com_fixed_sin_lut[idx]; \
|
||||
*c = m_cos_sign * com_fixed_sin_lut[((uint8_t)127 - (uint8_t)idx) % 128]
|
||||
|
||||
/* Handle cases of 2rd and 3th quadrant separately, for minus cos. */
|
||||
if (q >= COM_FIXED_FRACHALF / 2 &&
|
||||
q < COM_FIXED_FRACHALF + COM_FIXED_FRACHALF / 2) {
|
||||
/* Handle cases of 3rd and 4th quadrant separately, for minus sin. */
|
||||
if (q >= COM_FIXED_FRACHALF) {
|
||||
q %= COM_FIXED_FRACHALF;
|
||||
CASE(-1, -1);
|
||||
} else {
|
||||
CASE(+1, -1);
|
||||
}
|
||||
} else {
|
||||
/* Handle cases of 3rd and 4th quadrant separately, for minus sin. */
|
||||
if (q >= COM_FIXED_FRACHALF) {
|
||||
q %= COM_FIXED_FRACHALF;
|
||||
CASE(-1, 1);
|
||||
} else {
|
||||
CASE(+1, 1);
|
||||
}
|
||||
}
|
||||
|
||||
void com_fixed_print(com_fixed_t a);
|
||||
#undef CASE
|
||||
}
|
||||
|
||||
void com_fixed_run_tests(void);
|
||||
static inline com_fixed_t com_fixed_tan(com_fixed_t a) {
|
||||
com_fixed_t s, c;
|
||||
com_fixed_sincos(a, &s, &c);
|
||||
return com_fixed_div(s, c);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
+13
-10
@@ -170,18 +170,21 @@ static inline com_mat_t com_mat_perspective(uint16_t rwidth, uint16_t rheight,
|
||||
com_fixed_t fov) {
|
||||
com_mat_t result = {0};
|
||||
|
||||
// com_fixed_t const aspect = com_fixed_div(rwidth, rheight);
|
||||
com_fixed_t const aspect =
|
||||
com_fixed_div(rwidth * COM_FIXED_FRACUNIT, rheight * COM_FIXED_FRACUNIT);
|
||||
com_fixed_t const f =
|
||||
com_fixed_div(COM_FIXED_FRACUNIT,
|
||||
com_fixed_tan(com_fixed_mul(fov, COM_FIXED_FRACHALF)));
|
||||
com_fixed_t const fn = com_fixed_div(COM_FIXED_FRACUNIT, (nearz - farz));
|
||||
|
||||
// const float f = 1.0f / tanf(camera->fov * 0.5f);
|
||||
// const float fn = 1.0f / (CAMERA_NEAR_Z - camera->far_z);
|
||||
result.a[0 * 4 + 0] = com_fixed_div(f, aspect);
|
||||
result.a[1 * 4 + 1] = f;
|
||||
result.a[2 * 4 + 2] = (nearz + farz) * fn;
|
||||
result.a[2 * 4 + 3] = -COM_FIXED_FRACUNIT;
|
||||
result.a[3 * 4 + 2] =
|
||||
(COM_FIXED_FRACUNIT * 2) * com_fixed_mul(com_fixed_mul(nearz, farz), fn);
|
||||
|
||||
// result.row[0].x = f / aspect;
|
||||
// result.row[1].y = f;
|
||||
// result.row[2].z = (CAMERA_NEAR_Z + camera->far_z) * fn;
|
||||
// result.row[2].w = -1.0f;
|
||||
// result.row[3].z = 2.0f * CAMERA_NEAR_Z * camera->far_z * fn;
|
||||
|
||||
// return result;
|
||||
return result;
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -42,6 +42,7 @@ extern int plr_display_x11_main(int argc, char *argv[]) {
|
||||
com_fixed_print(v0.a[0]);
|
||||
|
||||
com_fixed_run_tests();
|
||||
com_fixed_run_bench();
|
||||
|
||||
Display *display = XOpenDisplay(NULL);
|
||||
if (NULL == display) {
|
||||
|
||||
+6
-3
@@ -1,8 +1,11 @@
|
||||
CC=clang
|
||||
CFLAGS=-lX11 -Wall -std=c99 -g3 -O3 -fno-inline -msse2 -lm
|
||||
DEPS = Display/display.h ../Common/lzw.h ../Common/mat.h ../Common/vec.h ../Common/fixed.h ../Common/def.h ../Common/bits.h
|
||||
# todo: Only link to libm on debug.
|
||||
CFLAGS=-lX11 -Wall -std=c99 -g3 -O3 -msse2 -lm
|
||||
DEPS = ../Common/lzw.h ../Common/mat.h ../Common/vec.h ../Common/fixed.h ../Common/def.h ../Common/bits.h \
|
||||
../Common/Timer/timer.h \
|
||||
./Display/display.h
|
||||
OBJ = ../Common/lzw.o ../Common/fixed.o
|
||||
LINUX = main.o Display/x11.o
|
||||
LINUX = main.o Display/x11.o ../Common/Timer/unix.o
|
||||
DOS = maindos.c Display/dos.c
|
||||
|
||||
%.o: %.c $(DEPS)
|
||||
|
||||
Reference in New Issue
Block a user