diff --git a/Common/Std/libc.c b/Common/Std/libc.c index 085608d..3fd2b2d 100644 --- a/Common/Std/libc.c +++ b/Common/Std/libc.c @@ -5,8 +5,9 @@ #include #include -void *com_std_realloc(void *ptr, size_t size) { return realloc(ptr, size); } -void com_std_free(void *ptr) { free(ptr); } +void *com_std_alloc(void *ptr, size_t size) { + return size != 0 ? realloc(ptr, size) : (free(ptr), NULL); +} void com_std_assert(bool holds) { assert(holds); } void com_std_printf(char const *fmt, ...) { va_list args; diff --git a/Common/Std/linux.c b/Common/Std/linux.c new file mode 100644 index 0000000..94cd865 --- /dev/null +++ b/Common/Std/linux.c @@ -0,0 +1,33 @@ +#include "../std.h" +#include +#include + +static inline long com_std_syscall(long number, long arg1, long arg2, + long arg3) { + long ret; + __asm__ volatile("syscall" + : "=a"(ret) + : "a"(number), "D"(arg1), "S"(arg2), "d"(arg3) + : "rcx", "r11", "memory"); + return ret; +} + +void *com_std_alloc(void *ptr, size_t size) { return NULL; } +void com_std_assert(bool holds) { + /* TODO: Write line info to stderr as well. */ + if (!holds) + com_std_syscall(__NR_exit, -1, 0, 0); +} + +/* TODO: Figure out a way to make it easy to deal with. */ +void com_std_printf(char const *fmt, ...) { + // va_list args; + // va_start(args, fmt); + // vprintf(fmt, args); + // va_end(args); +} + +double com_std_sqrt(double v) { return __builtin_sqrt(v); } +double com_std_fabs(double v) { return __builtin_fabs(v); } +double com_std_sin(double v) { return __builtin_sin(v); } +double com_std_cos(double v) { return __builtin_cos(v); } diff --git a/Common/Std/stub.c b/Common/Std/stub.c index bb21cf6..6d99041 100644 --- a/Common/Std/stub.c +++ b/Common/Std/stub.c @@ -1,10 +1,10 @@ #include "../std.h" -void *com_std_realloc(void *ptr, size_t size) { return NULL; } -void com_std_free(void *ptr) {} +void *com_std_alloc(void *ptr, size_t size) { return NULL; } void com_std_assert(bool holds) {} void com_std_printf(char const *fmt, ...) {} -double com_std_sqrt(double v) { return 0; } -double com_std_fabs(double v) { return 0; } -double com_std_sin(double v) { return 0; } -double com_std_cos(double v) { return 0; } + +double com_std_sqrt(double v) { return __builtin_sqrt(v); } +double com_std_fabs(double v) { return __builtin_fabs(v); } +double com_std_sin(double v) { return __builtin_sin(v); } +double com_std_cos(double v) { return __builtin_cos(v); } diff --git a/Common/Timer/posix.c b/Common/Timer/posix.c index e67ea87..09543f3 100644 --- a/Common/Timer/posix.c +++ b/Common/Timer/posix.c @@ -1,5 +1,5 @@ #include "../std.h" -#include "timer.h" +#include "../time.h" #include #define __USE_POSIX199309 1 #include diff --git a/Common/Timer/stub.c b/Common/Timer/stub.c new file mode 100644 index 0000000..0be7d8e --- /dev/null +++ b/Common/Timer/stub.c @@ -0,0 +1,9 @@ +#include "time.h" +#include + +uint64_t com_timer_count_ns(void) { return 0; } + +void com_timer_profile(uint64_t starttime, const char *what) { + (void)starttime; + (void)what; +} diff --git a/Common/fix.c b/Common/fix.c index cfcb927..10d4832 100644 --- a/Common/fix.c +++ b/Common/fix.c @@ -4,9 +4,9 @@ */ #include "fix.h" -#include "Timer/timer.h" #include "def.h" #include "std.h" +#include "time.h" #include #include @@ -128,75 +128,75 @@ void com_fix_run_bench(void) { } void com_fix_run_tests(void) { - { - double max_sqrt_deviation = 0.0f; - com_fix_t sqrt_accumulator = 0; - com_fix_t sqrt_test_step = COM_FIX_FRACUNIT >> 2; /* Quarter step */ - for (int i = 0; i < COM_FIX_FRACUNIT << 1; ++i) { - com_fix_t sqrt = com_fix_sqrt(sqrt_accumulator); - double const deviation = - com_std_sqrt(com_fix_as_float(sqrt_accumulator)) - - com_fix_as_float(sqrt); - if (deviation > max_sqrt_deviation) - max_sqrt_deviation = deviation; - sqrt_accumulator += sqrt_test_step; - } - com_std_printf("max sqrt deviation: %f\n", max_sqrt_deviation); - } + // { + // double max_sqrt_deviation = 0.0f; + // com_fix_t sqrt_accumulator = 0; + // com_fix_t sqrt_test_step = COM_FIX_FRACUNIT >> 2; /* Quarter step */ + // for (int i = 0; i < COM_FIX_FRACUNIT << 1; ++i) { + // com_fix_t sqrt = com_fix_sqrt(sqrt_accumulator); + // double const deviation = + // com_std_sqrt(com_fix_as_float(sqrt_accumulator)) - + // com_fix_as_float(sqrt); + // if (deviation > max_sqrt_deviation) + // max_sqrt_deviation = deviation; + // sqrt_accumulator += sqrt_test_step; + // } + // com_std_printf("max sqrt deviation: %f\n", max_sqrt_deviation); + // } - { - double max_sin_deviation = 0.0f; - com_fix_t sin_accumulator = -COM_FIX_PI * 4; - com_fix_t sin_test_step = (COM_FIX_PI << 1) >> 10; - for (int i = 0; i < 1024 * 32; ++i) { - com_fix_t sin = com_fix_sin(sin_accumulator); - double const deviation = - com_std_fabs(com_std_sin(com_fix_as_float(sin_accumulator)) - - com_fix_as_float(sin)); - if (deviation > max_sin_deviation) - max_sin_deviation = deviation; - sin_accumulator += sin_test_step; - } - com_std_printf("max sin deviation: %f\n", max_sin_deviation); - } + // { + // double max_sin_deviation = 0.0f; + // com_fix_t sin_accumulator = -COM_FIX_PI * 4; + // com_fix_t sin_test_step = (COM_FIX_PI << 1) >> 10; + // for (int i = 0; i < 1024 * 32; ++i) { + // com_fix_t sin = com_fix_sin(sin_accumulator); + // double const deviation = + // com_std_fabs(com_std_sin(com_fix_as_float(sin_accumulator)) - + // com_fix_as_float(sin)); + // if (deviation > max_sin_deviation) + // max_sin_deviation = deviation; + // sin_accumulator += sin_test_step; + // } + // com_std_printf("max sin deviation: %f\n", max_sin_deviation); + // } - { - double max_cos_deviation = 0.0f; - com_fix_t cos_accumulator = -COM_FIX_PI * 4; - com_fix_t cos_test_step = (COM_FIX_PI << 1) >> 10; - for (int i = 0; i < 1024 * 32; ++i) { - com_fix_t cos = com_fix_cos(cos_accumulator); - double const deviation = - com_std_fabs(com_std_cos(com_fix_as_float(cos_accumulator)) - - com_fix_as_float(cos)); - if (deviation > max_cos_deviation) - max_cos_deviation = deviation; - cos_accumulator += cos_test_step; - } - com_std_printf("max cos deviation: %f\n", max_cos_deviation); - } + // { + // double max_cos_deviation = 0.0f; + // com_fix_t cos_accumulator = -COM_FIX_PI * 4; + // com_fix_t cos_test_step = (COM_FIX_PI << 1) >> 10; + // for (int i = 0; i < 1024 * 32; ++i) { + // com_fix_t cos = com_fix_cos(cos_accumulator); + // double const deviation = + // com_std_fabs(com_std_cos(com_fix_as_float(cos_accumulator)) - + // com_fix_as_float(cos)); + // if (deviation > max_cos_deviation) + // max_cos_deviation = deviation; + // cos_accumulator += cos_test_step; + // } + // com_std_printf("max cos deviation: %f\n", max_cos_deviation); + // } - { - double max_sin_deviation = 0.0f; - double max_cos_deviation = 0.0f; - com_fix_t sincos_accumulator = -COM_FIX_PI * 4; - com_fix_t sincos_test_step = (COM_FIX_PI << 1) >> 10; - for (int i = 0; i < 1024 * 32; ++i) { - com_fix_t sin, cos; - com_fix_sincos(sincos_accumulator, &sin, &cos); - double const sin_deviation = - com_std_fabs(com_std_sin(com_fix_as_float(sincos_accumulator)) - - com_fix_as_float(sin)); - double const cos_deviation = - com_std_fabs(com_std_cos(com_fix_as_float(sincos_accumulator)) - - com_fix_as_float(cos)); - if (sin_deviation > max_sin_deviation) - max_sin_deviation = sin_deviation; - if (cos_deviation > max_cos_deviation) - max_cos_deviation = cos_deviation; - sincos_accumulator += sincos_test_step; - } - com_std_printf("max sincos deviations: %f %f\n", max_sin_deviation, - max_cos_deviation); - } + // { + // double max_sin_deviation = 0.0f; + // double max_cos_deviation = 0.0f; + // com_fix_t sincos_accumulator = -COM_FIX_PI * 4; + // com_fix_t sincos_test_step = (COM_FIX_PI << 1) >> 10; + // for (int i = 0; i < 1024 * 32; ++i) { + // com_fix_t sin, cos; + // com_fix_sincos(sincos_accumulator, &sin, &cos); + // double const sin_deviation = + // com_std_fabs(com_std_sin(com_fix_as_float(sincos_accumulator)) - + // com_fix_as_float(sin)); + // double const cos_deviation = + // com_std_fabs(com_std_cos(com_fix_as_float(sincos_accumulator)) - + // com_fix_as_float(cos)); + // if (sin_deviation > max_sin_deviation) + // max_sin_deviation = sin_deviation; + // if (cos_deviation > max_cos_deviation) + // max_cos_deviation = cos_deviation; + // sincos_accumulator += sincos_test_step; + // } + // com_std_printf("max sincos deviations: %f %f\n", max_sin_deviation, + // max_cos_deviation); + // } } diff --git a/Common/fix.h b/Common/fix.h index db55000..4adc57b 100644 --- a/Common/fix.h +++ b/Common/fix.h @@ -32,6 +32,10 @@ static inline com_fix_t com_fix_add(com_fix_t a, com_fix_t b) { return a + b; } static inline com_fix_t com_fix_sub(com_fix_t a, com_fix_t b) { return a - b; } +static inline com_fix_t com_fix_int(com_fix_t a) { + return a >> COM_FIX_FRACBITS; +} + static inline com_fix_t com_fix_mul(com_fix_t a, com_fix_t b) { return (com_fix_t)(((int64_t)a * (int64_t)b) >> COM_FIX_FRACBITS); } diff --git a/Common/gem.h b/Common/gem.h index 7da7d9c..46eee77 100644 --- a/Common/gem.h +++ b/Common/gem.h @@ -9,7 +9,9 @@ #include "mat.h" #include "pnt.h" #include "vec.h" +#include "zord.h" #include +#include /* Projects vertex position to a clip space via reordered row-major MVP matrix. It skips calculation of z depth component, as we assume to never use it. @@ -46,17 +48,29 @@ static inline com_vec_t com_gem_vec_project_clip(com_mat_t a, com_vec_t b) { #define COM_GEM_VERTEX_UNCLIPPED (0 << 0) #define COM_GEM_VERTEX_CLIPPED_X (1 << 0) #define COM_GEM_VERTEX_CLIPPED_Y (1 << 1) -/* Attempt to project clip space vertex to NDC, reporting which component lies - * outside of view. Bit test against those per component. This is needed for - * determining new view-lying clipped triangles. */ +/* Attempt to project clip space vertex to render pixel space, reporting which + * component lies outside of view. Bit test against those per component. This is + * needed for determining new view-lying clipped triangles. */ static inline uint8_t com_gem_clip_vis_project(com_vec_t a, com_pnt_t *out) { uint8_t mask = 0; if (a.a[0] < -a.a[2] || a.a[0] > a.a[2]) mask ^= COM_GEM_VERTEX_CLIPPED_X; if (a.a[1] < -a.a[2] || a.a[1] > a.a[2]) mask ^= COM_GEM_VERTEX_CLIPPED_Y; - out->a[0] = com_fix_div(a.a[0], a.a[2]) / 480 + 240; - out->a[1] = com_fix_div(a.a[1], a.a[2]) / 640 + 320; + // out->a[0] = com_fix_mul((com_fix_div(a.a[0], a.a[2]) + COM_FIX_FRACUNIT) >> + // 1, + // 640 * COM_FIX_FRACUNIT); + // out->a[1] = com_fix_mul((com_fix_div(a.a[1], a.a[2]) + COM_FIX_FRACUNIT) >> + // 1, + // 480 * COM_FIX_FRACUNIT); + out->a[0] = + (com_fix_mul(com_fix_div(a.a[0], a.a[2]), 640 * COM_FIX_FRACUNIT) >> + COM_FIX_FRACBITS) + + 320; + out->a[1] = + (com_fix_mul(com_fix_div(a.a[1], a.a[2]), 480 * COM_FIX_FRACUNIT) >> + COM_FIX_FRACBITS) + + 240; return mask; } @@ -75,7 +89,7 @@ static inline void com_gem_draw_triangle(com_pnt_t v0, com_pnt_t v1, com_vec_t ws = com_pnt_edge_weight(v0, v1, v2, p); // If the point is inside or on all edges, draw the pixel - // (Assumes Clockwise vertex ordering) + // (Assumes Counter Clockwise vertex ordering) if (ws.a[0] >= 0 && ws.a[1] >= 0 && ws.a[2] >= 0) { out[(y * 640 + x) * 4 + 0] = 125; out[(y * 640 + x) * 4 + 1] = 125; @@ -85,10 +99,15 @@ static inline void com_gem_draw_triangle(com_pnt_t v0, com_pnt_t v1, } } -static inline void com_gem_draw_triangle_textured(com_pnt_t v0, com_pnt_t v1, - com_pnt_t v2, uint8_t *out, - com_pnt_t uv0, com_pnt_t uv1, - com_pnt_t uv2, uint8_t *tex) { +/* TODO: Implement and compare: + https://www.digipen.edu/sites/default/files/public/docs/theses/salem-haykal-digipen-master-of-science-in-computer-science-thesis-an-optimized-triangle-rasterizer.pdf + + Tile-based variant fares even worse than this implementation. + */ +static inline void +com_gem_draw_triangle_textured_edge(com_pnt_t v0, com_pnt_t v1, com_pnt_t v2, + uint8_t *out, com_pnt_t uv0, com_pnt_t uv1, + com_pnt_t uv2, uint8_t *tex) { int minX = com_vec_min(com_vec_from(v0.a[0], v1.a[0], v2.a[0])); int minY = com_vec_min(com_vec_from(v0.a[1], v1.a[1], v2.a[1])); int maxX = com_vec_max(com_vec_from(v0.a[0], v1.a[0], v2.a[0])); @@ -100,7 +119,6 @@ static inline void com_gem_draw_triangle_textured(com_pnt_t v0, com_pnt_t v1, for (int x = minX; x <= maxX; x++) { com_pnt_t p = {.a = {x, y}}; - /* TODO: Move to a separate weighting function over vec. */ // Test the pixel center against all 3 edges com_vec_t ws = com_pnt_edge_weight(v0, v1, v2, p); @@ -124,12 +142,106 @@ static inline void com_gem_draw_triangle_textured(com_pnt_t v0, com_pnt_t v1, int32_t ty = (int32_t)((v + (COM_FIX_FRACUNIT / 2)) / (COM_FIX_FRACUNIT / 16)); - out[(y * 640 + x) * 4 + 0] = tex[(ty * 32 + tx) * 3 + 0]; - out[(y * 640 + x) * 4 + 1] = tex[(ty * 32 + tx) * 3 + 1]; - out[(y * 640 + x) * 4 + 2] = tex[(ty * 32 + tx) * 3 + 2]; + // uint32_t idx = com_zord_order(tx, ty); + uint32_t idx = ty * 32 + tx; + // uint32_t bidx = com_zord_order(x, y); + uint32_t bidx = y * 640 + x; + + out[bidx * 4 + 0] = tex[idx * 3 + 0]; + out[bidx * 4 + 1] = tex[idx * 3 + 1]; + out[bidx * 4 + 2] = tex[idx * 3 + 2]; } } } } +static inline void draw_span(int32_t x_start_fp, int32_t x_end_fp, int y, + uint8_t *out, uint8_t *tex) { + int x1 = com_fix_int( + x_start_fp + (1 << (COM_FIX_FRACBITS - 1))); // Rounding to nearest pixel + int x2 = com_fix_int(x_end_fp + (1 << (COM_FIX_FRACBITS - 1))); + + if (x1 > x2) { + int tmp = x1; + x1 = x2; + x2 = tmp; + } + + for (int x = x1; x < x2; x++) { + uint32_t idx = 0 * 32 + 0; + uint32_t bidx = y * 640 + x; + out[bidx * 4 + 0] = tex[idx * 3 + 0]; + out[bidx * 4 + 1] = tex[idx * 3 + 1]; + out[bidx * 4 + 2] = tex[idx * 3 + 2]; + } +} + +static inline void +com_gem_draw_triangle_textured_span(com_pnt_t v0, com_pnt_t v1, com_pnt_t v2, + uint8_t *out, com_pnt_t uv0, com_pnt_t uv1, + com_pnt_t uv2, uint8_t *tex) { + // 1. Sort vertices by Y coordinate (v0.a[1] <= v1.a[1] <= v2.a[1]) + if (v0.a[1] > v1.a[1]) { + com_pnt_t tmp = v0; + v0 = v1; + v1 = tmp; + } + if (v0.a[1] > v2.a[1]) { + com_pnt_t tmp = v0; + v0 = v2; + v2 = tmp; + } + if (v1.a[1] > v2.a[1]) { + com_pnt_t tmp = v1; + v1 = v2; + v2 = tmp; + } + + // Guard against zero-height triangles + if ((int)v0.a[1] == (int)v2.a[1]) + return; + + // 2. Compute overall inverse slopes (dX/dY) for all edges + // Fixed-point division is performed using 64-bit integer casting to prevent + // overflow + com_fix_t dx02 = (v2.a[1] != v0.a[1]) + ? (com_fix_t)((int64_t)(v2.a[0] - v0.a[0]) + << COM_FIX_FRACBITS / (v2.a[1] - v0.a[1])) + : 0; + com_fix_t dx01 = (v1.a[1] != v0.a[1]) + ? (com_fix_t)((int64_t)(v1.a[0] - v0.a[0]) + << COM_FIX_FRACBITS / (v1.a[1] - v0.a[1])) + : 0; + com_fix_t dx12 = (v2.a[1] != v1.a[1]) + ? (com_fix_t)((int64_t)(v2.a[0] - v1.a[0]) + << COM_FIX_FRACBITS / (v2.a[1] - v1.a[1])) + : 0; + + // Determine the scanline limits (pixel discrete boundaries) + int y_start = com_fix_int(v0.a[1]); + int y_mid = com_fix_int(v1.a[1]); + int y_end = com_fix_int(v2.a[1]); + + // Initialize edge walkers starting at the top vertex (v0) + com_fix_t edge1 = v0.a[0]; // Follows the long edge (v0 -> v2) + com_fix_t edge2 = v0.a[0]; // Follows short edges (v0 -> v1, then v1 -> v2) + + // 3. Top Half Scan: Walk from top vertex to middle vertex breakpoint + for (int y = y_start; y < y_mid; y++) { + draw_span(edge1, edge2, y, out, tex); + edge1 += dx02; + edge2 += dx01; + } + + // Adjust the short edge walker directly to the middle vertex position + edge2 = v1.a[0]; + + // 4. Bottom Half Scan: Walk from middle vertex to bottom vertex + for (int y = y_mid; y < y_end; y++) { + draw_span(edge1, edge2, y, out, tex); + edge1 += dx02; + edge2 += dx12; + } +} + #endif diff --git a/Common/lzw.c b/Common/lzw.c index c6c9ddd..17aca05 100644 --- a/Common/lzw.c +++ b/Common/lzw.c @@ -44,9 +44,9 @@ struct com_lzw_table com_lzw_infer_table(const char *datain, uint32_t sizein) { if (!code_found) { if (table_size >= table_cap) { /* TODO: Ref to prev table gets missed, memory leak scenario */ - if (!(table = com_std_realloc( - table, sizeof(struct com_lzw_table_entry) * - (table_cap + COM_LZW_TABLE_CAP_GROW)))) + if (!(table = com_std_alloc(table, + sizeof(struct com_lzw_table_entry) * + (table_cap + COM_LZW_TABLE_CAP_GROW)))) goto ERR_ALLOC_INIT_TABLE; table_cap += COM_LZW_TABLE_CAP_GROW; @@ -68,14 +68,14 @@ struct com_lzw_table com_lzw_infer_table(const char *datain, uint32_t sizein) { ERR_ALLOC_INIT_TABLE: if (table_cap > 0) - com_std_free(table); + com_std_alloc(table, 0); return (struct com_lzw_table){0}; } void com_lzw_free_table(struct com_lzw_table *table) { if (!table->table) return; - com_std_free(table->table); + com_std_alloc(table->table, 0); table->size = 0; table->cap = 0; table->init_size = 0; @@ -131,8 +131,7 @@ bool com_lzw_compress(const struct com_lzw_table *table, const char *datain, if (output_size == output_cap && output_bitshift + codesize > 8) { /* TODO: catch alloc failure */ - output = - com_std_realloc(output, output_cap + COM_LZW_OUTPUT_CAP_GROWTH); + output = com_std_alloc(output, output_cap + COM_LZW_OUTPUT_CAP_GROWTH); output_cap += COM_LZW_OUTPUT_CAP_GROWTH; } diff --git a/Common/mat.c b/Common/mat.c index aee3e42..496a36b 100644 --- a/Common/mat.c +++ b/Common/mat.c @@ -1,5 +1,5 @@ #include "mat.h" -#include "Timer/timer.h" +#include "time.h" /* TODO: Doesn't work, clang optimizes it away. We need RNG here. */ diff --git a/Common/mat.h b/Common/mat.h index aae26da..9834cc4 100644 --- a/Common/mat.h +++ b/Common/mat.h @@ -150,7 +150,8 @@ static inline com_mat_t com_mat_look_at(com_vec_t pos, com_vec_t up, return result; } -/* TODO: move to .c file */ +/* TODO: Make sure it's compile time optimized, as all parameters are constant. + */ /* Produces a projection matrix needed for camera work. */ static inline com_mat_t com_mat_perspective(uint16_t rwidth, uint16_t rheight, com_fix_t nearz, com_fix_t farz, diff --git a/Common/pnt.h b/Common/pnt.h index 6ebccbc..99088e0 100644 --- a/Common/pnt.h +++ b/Common/pnt.h @@ -63,11 +63,13 @@ static inline com_fix_t com_pnt_max(com_pnt_t a) { return com_fix_max(a.a[0], a.a[1]); } -/* TODO: Safeguard by com_fix_mul()? */ /* Tests whether a point is lying right to a line produced by l0 and l1. This assumes that l0 and l1 are in clockwise order. Result is positive or zero if it holds true, otherwise it's negative. - This also produces area of a triangle! */ + This also produces area of a triangle! + + Somehow this is faster than counter clockwise ordering? In case tested. + */ static inline com_fix_t com_pnt_edge_orient(com_pnt_t l0, com_pnt_t l1, com_pnt_t p) { return (l0.a[0] - l1.a[0]) * (p.a[1] - l1.a[1]) - @@ -83,6 +85,32 @@ static inline com_vec_t com_pnt_edge_weight(com_pnt_t v0, com_pnt_t v1, com_pnt_edge_orient(v0, v1, p)}}; } +/* As terms are constant over pixel steps, we can use precalculated values. + See: (l0.a[0] - l1.a[0]) and (l0.a[1] - l1.a[1]) + + Optimizer seems to be smart enough to figure this out already? And even more + optimal. Delete this later. + */ +static inline void com_pnt_edge_weight_calc_precomp_table(com_pnt_t v0, + com_pnt_t v1, + com_pnt_t v2, + com_pnt_t out[3]) { + out[0] = com_pnt_sub(v1, v2); + out[1] = com_pnt_sub(v2, v0); + out[2] = com_pnt_sub(v0, v1); +} + +static inline com_vec_t +com_pnt_edge_weight_calc_precomp(com_pnt_t v0, com_pnt_t v1, com_pnt_t v2, + com_pnt_t p, com_pnt_t precomp[3]) { + return (com_vec_t){.a = {precomp[0].a[0] * (p.a[1] - v2.a[1]) - + precomp[0].a[1] * (p.a[0] - v2.a[0]), + precomp[1].a[0] * (p.a[1] - v0.a[1]) - + precomp[1].a[1] * (p.a[0] - v0.a[0]), + precomp[2].a[0] * (p.a[1] - v0.a[1]) - + precomp[2].a[1] * (p.a[0] - v0.a[0])}}; +} + static inline void com_pnt_print(com_pnt_t a) { com_fix_print(a.a[0]); com_fix_print(a.a[1]); diff --git a/Common/std.h b/Common/std.h index b78f430..7e8bb14 100644 --- a/Common/std.h +++ b/Common/std.h @@ -3,13 +3,16 @@ even memory mnagement. */ +#ifndef COM_STD_H +#define COM_STD_H + #include #include -extern void *com_std_realloc(void *ptr, size_t size); -extern void com_std_free(void *ptr); +/* Note: Freeing is done by passing 0 size to pointer. */ +extern void *com_std_alloc(void *ptr, size_t size); -/* Note: Functions listed below are only available for test builds. */ +/* Note: Functions listed below are only available for dev builds. */ /* Trigonometry functions are for reference testing of our own fixed point * implementations. Do not use otherwise. */ extern void com_std_assert(bool holds); @@ -18,3 +21,5 @@ extern double com_std_sqrt(double v); extern double com_std_fabs(double v); extern double com_std_sin(double v); extern double com_std_cos(double v); + +#endif diff --git a/Common/Timer/timer.h b/Common/time.h similarity index 95% rename from Common/Timer/timer.h rename to Common/time.h index 994e2e0..42330cf 100644 --- a/Common/Timer/timer.h +++ b/Common/time.h @@ -1,7 +1,7 @@ #ifndef COM_TIMER_H #define COM_TIMER_H -#include "../def.h" +#include "def.h" #include /* Global monotonic counter, meaning it always increases. Not applicable for diff --git a/Common/zord.c b/Common/zord.c new file mode 100644 index 0000000..8a1ecd3 --- /dev/null +++ b/Common/zord.c @@ -0,0 +1,33 @@ +#include "zord.h" +#include + +const uint16_t com_zord_morton_table[] = { + 0x0000, 0x0001, 0x0004, 0x0005, 0x0010, 0x0011, 0x0014, 0x0015, 0x0040, + 0x0041, 0x0044, 0x0045, 0x0050, 0x0051, 0x0054, 0x0055, 0x0100, 0x0101, + 0x0104, 0x0105, 0x0110, 0x0111, 0x0114, 0x0115, 0x0140, 0x0141, 0x0144, + 0x0145, 0x0150, 0x0151, 0x0154, 0x0155, 0x0400, 0x0401, 0x0404, 0x0405, + 0x0410, 0x0411, 0x0414, 0x0415, 0x0440, 0x0441, 0x0444, 0x0445, 0x0450, + 0x0451, 0x0454, 0x0455, 0x0500, 0x0501, 0x0504, 0x0505, 0x0510, 0x0511, + 0x0514, 0x0515, 0x0540, 0x0541, 0x0544, 0x0545, 0x0550, 0x0551, 0x0554, + 0x0555, 0x1000, 0x1001, 0x1004, 0x1005, 0x1010, 0x1011, 0x1014, 0x1015, + 0x1040, 0x1041, 0x1044, 0x1045, 0x1050, 0x1051, 0x1054, 0x1055, 0x1100, + 0x1101, 0x1104, 0x1105, 0x1110, 0x1111, 0x1114, 0x1115, 0x1140, 0x1141, + 0x1144, 0x1145, 0x1150, 0x1151, 0x1154, 0x1155, 0x1400, 0x1401, 0x1404, + 0x1405, 0x1410, 0x1411, 0x1414, 0x1415, 0x1440, 0x1441, 0x1444, 0x1445, + 0x1450, 0x1451, 0x1454, 0x1455, 0x1500, 0x1501, 0x1504, 0x1505, 0x1510, + 0x1511, 0x1514, 0x1515, 0x1540, 0x1541, 0x1544, 0x1545, 0x1550, 0x1551, + 0x1554, 0x1555, 0x4000, 0x4001, 0x4004, 0x4005, 0x4010, 0x4011, 0x4014, + 0x4015, 0x4040, 0x4041, 0x4044, 0x4045, 0x4050, 0x4051, 0x4054, 0x4055, + 0x4100, 0x4101, 0x4104, 0x4105, 0x4110, 0x4111, 0x4114, 0x4115, 0x4140, + 0x4141, 0x4144, 0x4145, 0x4150, 0x4151, 0x4154, 0x4155, 0x4400, 0x4401, + 0x4404, 0x4405, 0x4410, 0x4411, 0x4414, 0x4415, 0x4440, 0x4441, 0x4444, + 0x4445, 0x4450, 0x4451, 0x4454, 0x4455, 0x4500, 0x4501, 0x4504, 0x4505, + 0x4510, 0x4511, 0x4514, 0x4515, 0x4540, 0x4541, 0x4544, 0x4545, 0x4550, + 0x4551, 0x4554, 0x4555, 0x5000, 0x5001, 0x5004, 0x5005, 0x5010, 0x5011, + 0x5014, 0x5015, 0x5040, 0x5041, 0x5044, 0x5045, 0x5050, 0x5051, 0x5054, + 0x5055, 0x5100, 0x5101, 0x5104, 0x5105, 0x5110, 0x5111, 0x5114, 0x5115, + 0x5140, 0x5141, 0x5144, 0x5145, 0x5150, 0x5151, 0x5154, 0x5155, 0x5400, + 0x5401, 0x5404, 0x5405, 0x5410, 0x5411, 0x5414, 0x5415, 0x5440, 0x5441, + 0x5444, 0x5445, 0x5450, 0x5451, 0x5454, 0x5455, 0x5500, 0x5501, 0x5504, + 0x5505, 0x5510, 0x5511, 0x5514, 0x5515, 0x5540, 0x5541, 0x5544, 0x5545, + 0x5550, 0x5551, 0x5554, 0x5555}; diff --git a/Common/zord.h b/Common/zord.h new file mode 100644 index 0000000..dc57805 --- /dev/null +++ b/Common/zord.h @@ -0,0 +1,36 @@ +#ifndef COM_ZORD_H +#define COM_ZORD_H +#include + +// #include + +extern const uint16_t com_zord_morton_table[]; + +/* Z-filling curve to help with cache and data locality. */ +static inline uint32_t com_zord_order(uint16_t x, uint16_t y) { + return (com_zord_morton_table[y >> 8] << 17) | + (com_zord_morton_table[x >> 8] << 16) | + (com_zord_morton_table[y & 0xFF] << 1) | + (com_zord_morton_table[x & 0xFF]); + + // return _pdep_u32(x, 0x55555555) | _pdep_u32(y, 0xAAAAAAAA); +} + +/* https://en.wikipedia.org/wiki/Z-order_curve#Coordinate_values*/ +static inline uint32_t com_zord_inc_x(uint32_t z) { + return (((z | 0b10101010) + 1) & 0b01010101) | (z & 0b10101010); +} + +static inline uint32_t com_zord_dec_x(uint32_t z) { + return (((z & 0b01010101) - 1) & 0b01010101) | (z & 0b10101010); +} + +static inline uint32_t com_zord_inc_y(uint32_t z) { + return (((z | 0b01010101) + 1) & 0b10101010) | (z & 0b01010101); +} + +static inline uint32_t com_zord_dec_y(uint32_t z) { + return (((z & 0b10101010) - 1) & 0b10101010) | (z & 0b01010101); +} + +#endif diff --git a/Player/Display/x11.c b/Player/Display/x11.c index aff1d1c..e262541 100644 --- a/Player/Display/x11.c +++ b/Player/Display/x11.c @@ -1,19 +1,20 @@ #include #include #include -#include #include #include -#include -#include #include "../../Common/lzw.h" #include "../../Common/mat.h" +#include "../../Common/std.h" #include "../draw.h" #include "display.h" +/* https://hereket.com/posts/from-scratch-x11-windowing/ */ + bool quited = false; +/* TODO: Have a cfg.h configuration header setting those. */ #define WIDTH 640 #define HEIGHT 480 @@ -27,7 +28,7 @@ extern int plr_display_x11_main(int argc, char *argv[]) { bool test = com_lzw_compress(&table, test_string, sizeof(test_string), &compressed_string, &compressed_string_sz); com_lzw_free_table(&table); - assert(test); + com_std_assert(test); com_mat_t m0 = com_mat_identity(); com_mat_t m1 = {1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16}; @@ -46,98 +47,103 @@ extern int plr_display_x11_main(int argc, char *argv[]) { com_fix_run_bench(); com_mat_run_bench(); - Display *display = XOpenDisplay(NULL); - if (NULL == display) { - fprintf(stderr, "Failed to initialize display"); - return EXIT_FAILURE; + // Display *display = XOpenDisplay(NULL); + // if (NULL == display) { + // com_std_printf("Failed to initialize display"); + // return -1; + // } + + // Window root = DefaultRootWindow(display); + // if (None == root) { + // com_std_printf("No root window found"); + // XCloseDisplay(display); + // return -1; + // } + + // int screen = DefaultScreen(display); + // Visual *visual = DefaultVisual(display, screen); + // int depth = DefaultDepth(display, screen); + + // Window window = + // XCreateSimpleWindow(display, root, 0, 0, WIDTH, HEIGHT, 0, 0, + // 0xffffffff); + // if (None == window) { + // com_std_printf("Failed to create window"); + // XCloseDisplay(display); + // return -1; + // } + + // XSizeHints *hints = XAllocSizeHints(); + // if (hints == NULL) + // return -1; + + // // Pinning min and max to the same values disables resizing + // hints->flags = PMinSize | PMaxSize; + // hints->min_width = hints->max_width = WIDTH; + // hints->min_height = hints->max_height = HEIGHT; + + // XSetWMNormalHints(display, window, hints); + // XFree(hints); + + // XSelectInput(display, window, ExposureMask | KeyPressMask); + // XMapWindow(display, window); + + // GC gc = XCreateGC(display, window, 0, NULL); + + // Atom wm_delete_window = XInternAtom(display, "WM_DELETE_WINDOW", False); + // XSetWMProtocols(display, window, &wm_delete_window, 1); + + // // TODO: can't be not true + // int bytes_per_pixel = 4; + // // char *pixel_buffer = (char *)malloc(WIDTH * HEIGHT * bytes_per_pixel); + // char pixel_buffer[WIDTH * HEIGHT * bytes_per_pixel]; + + // XImage *ximage = XCreateImage(display, visual, depth, ZPixmap, 0, + // pixel_buffer, WIDTH, HEIGHT, 32, 0); + + // if (!ximage) { + // com_std_printf("Failed to create XImage\n"); + // // com_std_alloc(pixel_buffer, 0); + // XCloseDisplay(display); + // return -1; + // } + + // int frame = 0; + + // XEvent event; + // while (!quited) { + // XNextEvent(display, &event); + + // switch (event.type) { + // case ClientMessage: + // if (event.xclient.data.l[0] == wm_delete_window) { + // XDestroyWindow(display, window); + // quited = true; + // } + // break; + + // case Expose: + // frame += 2; + + // plr_draw_screen((uint8_t *)pixel_buffer); + + // // Draw the complete image onto the window when exposed + // XPutImage(display, + // window, // Target drawable + // gc, // Graphics Context + // ximage, // Source XImage + // 0, 0, // Source coordinates (x, y) + // 0, 0, // Destination coordinates (x, y) + // WIDTH, HEIGHT // Dimensions to copy + // ); + // break; + // } + // } + + // XCloseDisplay(display); + + while (1) { } - Window root = DefaultRootWindow(display); - if (None == root) { - fprintf(stderr, "No root window found"); - XCloseDisplay(display); - return EXIT_FAILURE; - } - - int screen = DefaultScreen(display); - Visual *visual = DefaultVisual(display, screen); - int depth = DefaultDepth(display, screen); - - Window window = - XCreateSimpleWindow(display, root, 0, 0, WIDTH, HEIGHT, 0, 0, 0xffffffff); - if (None == window) { - fprintf(stderr, "Failed to create window"); - XCloseDisplay(display); - return EXIT_FAILURE; - } - - XSizeHints *hints = XAllocSizeHints(); - if (hints == NULL) - return EXIT_FAILURE; - - // Pinning min and max to the same values disables resizing - hints->flags = PMinSize | PMaxSize; - hints->min_width = hints->max_width = WIDTH; - hints->min_height = hints->max_height = HEIGHT; - - XSetWMNormalHints(display, window, hints); - XFree(hints); - - XSelectInput(display, window, ExposureMask | KeyPressMask); - XMapWindow(display, window); - - GC gc = XCreateGC(display, window, 0, NULL); - - Atom wm_delete_window = XInternAtom(display, "WM_DELETE_WINDOW", False); - XSetWMProtocols(display, window, &wm_delete_window, 1); - - // TODO: can't be not true - int bytes_per_pixel = 4; - char *pixel_buffer = (char *)malloc(WIDTH * HEIGHT * bytes_per_pixel); - - XImage *ximage = XCreateImage(display, visual, depth, ZPixmap, 0, - pixel_buffer, WIDTH, HEIGHT, 32, 0); - - if (!ximage) { - fprintf(stderr, "Failed to create XImage\n"); - free(pixel_buffer); - XCloseDisplay(display); - return EXIT_FAILURE; - } - - int frame = 0; - - XEvent event; - while (!quited) { - XNextEvent(display, &event); - - switch (event.type) { - case ClientMessage: - if (event.xclient.data.l[0] == wm_delete_window) { - XDestroyWindow(display, window); - quited = true; - } - break; - - case Expose: - frame += 2; - - plr_draw_screen((uint8_t *)pixel_buffer); - - // Draw the complete image onto the window when exposed - XPutImage(display, - window, // Target drawable - gc, // Graphics Context - ximage, // Source XImage - 0, 0, // Source coordinates (x, y) - 0, 0, // Destination coordinates (x, y) - WIDTH, HEIGHT // Dimensions to copy - ); - break; - } - } - - XCloseDisplay(display); - return 0; } diff --git a/Player/Main/linux.c b/Player/Main/linux.c index 98e0d78..bc60502 100644 --- a/Player/Main/linux.c +++ b/Player/Main/linux.c @@ -1,5 +1,15 @@ #include "../Display/display.h" -extern int main(int argc, char *argv[]) { - return plr_display_x11_main(argc, argv); +__attribute__((force_align_arg_pointer)) __attribute__((noreturn)) extern void +_start(void) { + + plr_display_x11_main(0, 0); + + __asm__ volatile("mov $60, %%rax\n\t" + "xor %%rdi, %%rdi\n\t" + "syscall" + : + : + : "%rax", "%rdi"); + __builtin_unreachable(); } diff --git a/Player/Makefile b/Player/Makefile index 5abb770..1b27d1c 100644 --- a/Player/Makefile +++ b/Player/Makefile @@ -3,7 +3,7 @@ CC = clang CFLAGS = -Wall -std=c99 -g3 -O3 -flto=full DEPS = ../Common/lzw.h ../Common/mat.h ../Common/vec.h ../Common/fix.h ../Common/def.h ../Common/bits.h \ ../Common/gem.h \ - ../Common/Timer/timer.h \ + ../Common/Timer/time.h \ ./Display/display.h \ ./draw.h SRC = ../Common/lzw.c ../Common/fix.c ../Common/mat.c ./draw.c @@ -15,7 +15,7 @@ DOS = maindos.c Display/dos.c $(CC) -c -o $@ $< $(CFLAGS) # TODO: Make it work with no stdlib. -linux: CFLAGS += -msse2 -lm -lX11 # -nostdlib +linux: CFLAGS += -msse2 -mbmi2 -nostdlib linux: $(LINUX) | $(SRC) $(CC) -o player $^ $(CFLAGS) diff --git a/Player/amalgam.c b/Player/amalgam.c index fda2437..8b91c7b 100644 --- a/Player/amalgam.c +++ b/Player/amalgam.c @@ -6,11 +6,12 @@ #include "../Common/fix.c" #include "../Common/lzw.c" #include "../Common/mat.c" +#include "../Common/zord.c" #include "./draw.c" #if COM_DEF_TARGET == COM_DEF_TARGET_LINUX -#include "../Common/Std/libc.c" -#include "../Common/Timer/posix.c" +#include "../Common/Std/linux.c" +#include "../Common/Timer/stub.c" #include "./Display/x11.c" #include "./Main/linux.c" #elif COM_DEF_TARGET == COM_DEF_TARGET_WASM diff --git a/Player/draw.c b/Player/draw.c index 3d56a94..f1b9e23 100644 --- a/Player/draw.c +++ b/Player/draw.c @@ -1,6 +1,8 @@ #include "../Common/def.h" #include "../Common/gem.h" +#include "../Common/std.h" +#include "../Common/time.h" #include #if COM_DEF_TARGET == COM_DEF_TARGET_WASM @@ -33,38 +35,52 @@ com_def_export_sym("plr_draw_screen") void plr_draw_screen( com_vec_from(-COM_FIX_FRACUNIT, 0, 0)); com_mat_t mvp = com_mat_reoder(com_mat_mul(proj, view)); - com_vec_t v0 = com_vec_from(5 * COM_FIX_FRACUNIT, 0, 2 * COM_FIX_FRACUNIT); - com_vec_t v1 = com_vec_from(5 * COM_FIX_FRACUNIT, 1 * COM_FIX_FRACUNIT, - 4 * COM_FIX_FRACUNIT); - com_vec_t v2 = com_vec_from(5 * COM_FIX_FRACUNIT, 4 * COM_FIX_FRACUNIT, - 4 * COM_FIX_FRACUNIT); + com_vec_t v0 = com_vec_from(0 * COM_FIX_FRACUNIT, 0, 1 * COM_FIX_FRACUNIT); + com_vec_t v1 = com_vec_from(1 * COM_FIX_FRACUNIT, 1 * COM_FIX_FRACUNIT, + 1 * COM_FIX_FRACUNIT); + com_vec_t v2 = com_vec_from(1 * COM_FIX_FRACUNIT, 1 * COM_FIX_FRACUNIT, + 1 * COM_FIX_FRACUNIT); + com_vec_t v3 = com_vec_from(1 * COM_FIX_FRACUNIT, 1 * COM_FIX_FRACUNIT, + -1 * COM_FIX_FRACUNIT); com_pnt_t uv0 = com_pnt_from(0 * COM_FIX_FRACUNIT, 0 * COM_FIX_FRACUNIT); com_pnt_t uv1 = com_pnt_from(1 * COM_FIX_FRACUNIT, 0 * COM_FIX_FRACUNIT); com_pnt_t uv2 = com_pnt_from(1 * COM_FIX_FRACUNIT, 1 * COM_FIX_FRACUNIT); + com_pnt_t uv3 = com_pnt_from(0 * COM_FIX_FRACUNIT, 1 * COM_FIX_FRACUNIT); com_vec_t c0 = com_gem_vec_project_clip(mvp, v0); com_vec_t c1 = com_gem_vec_project_clip(mvp, v1); com_vec_t c2 = com_gem_vec_project_clip(mvp, v2); + com_vec_t c3 = com_gem_vec_project_clip(mvp, v3); - com_pnt_t p0, p1, p2; + com_pnt_t p0, p1, p2, p3; uint8_t cc0 = com_gem_clip_vis_project(c0, &p0); uint8_t cc1 = com_gem_clip_vis_project(c1, &p1); uint8_t cc2 = com_gem_clip_vis_project(c2, &p2); + uint8_t cc3 = com_gem_clip_vis_project(c3, &p3); - com_pnt_print(p0); - com_pnt_print(p1); - com_pnt_print(p2); + com_std_printf("%i, %i\n", p0.a[0], p0.a[1]); + com_std_printf("%i, %i\n", p1.a[0], p1.a[1]); + com_std_printf("%i, %i\n", p2.a[0], p2.a[1]); + com_std_printf("%i, %i\n", p3.a[0], p3.a[1]); uint8_t tex[32 * 32 * 3] = {0}; for (int y = 0; y < 32; ++y) { for (int x = 0; x < 32; ++x) { - tex[(y * 32 + x) * 3 + 0] = (x / 2 + y / 2) % 2 == 0 ? 175 : 0; - tex[(y * 32 + x) * 3 + 1] = (x / 2 + y / 2) % 2 == 0 ? 0 : 0; - tex[(y * 32 + x) * 3 + 2] = (x / 2 + y / 2) % 2 == 0 ? 175 : 0; + // uint32_t idx = com_zord_order(x, y); + uint32_t idx = y * 32 + x; + tex[idx * 3 + 0] = (x / 2 + y / 2) % 2 == 0 ? 175 : 0; + tex[idx * 3 + 1] = (x / 2 + y / 2) % 2 == 0 ? 0 : 0; + tex[idx * 3 + 2] = (x / 2 + y / 2) % 2 == 0 ? 175 : 0; } } - com_gem_draw_triangle_textured(p0, p1, p2, (uint8_t *)pixel_buffer, uv0, uv1, - uv2, tex); + uint64_t time = com_timer_count_ns(); + for (int i = 5000; i--;) { + // com_gem_draw_triangle_textured_edge(p0, p1, p2, (uint8_t *)pixel_buffer, + // uv0, uv1, uv2, tex); + // com_gem_draw_triangle_textured_span(p0, p1, p3, (uint8_t *)pixel_buffer, + // uv0, uv1, uv3, tex); + } + com_timer_profile(time, "textured triangle"); }