背景
Nuklear 是一个单文件的即时模式 GUI 库,面向嵌入式和轻量级应用场景。为了减少对外部 math 库的依赖,并针对图形渲染场景优化性能,Nuklear 内置了一组数学和几何函数。这些函数使用快速逼近、位操作、多项式拟合等计算机图形学常见技巧,在精度可接受的范围内换取更高的计算效率。
设计取舍
与标准 math 库相比,Nuklear 的数学函数有几个明确的设计取向:
效率优先 :用近似算法替代精确计算。例如 nk_inv_sqrt 直接返回平方根倒数,省去一次除法。
接口直接 :函数命名和参数设计贴合图形编程习惯,减少组合调用。
场景特定 :提供 nk_unify、nk_triangle_from_direction 等标准库中没有的几何辅助函数。
可移植性 :不依赖平台特定的 math 库实现,代码本身为纯 C,可在不同编译器下编译。
函数实现
#ifndef NK_INV_SQRT
#define NK_INV_SQRT nk_inv_sqrt
NK_LIB float
nk_inv_sqrt(float n)
{
float x2;
const float threehalfs = 1.5f;
union {nk_uint i; float f;} conv = {0};
conv.f = n;
x2 = n * 0.5f;
conv.i = 0x5f375A84 - (conv.i >> 1);
conv.f = conv.f * (threehalfs - (x2 * conv.f * conv.f));
return conv.f;
}
#endif
#ifndef NK_SIN
#define NK_SIN nk_sin
NK_LIB float
nk_sin(float x)
{
NK_STORAGE const float a0 = +1.91059300966915117e-31f;
NK_STORAGE const float a1 = +1.00086760103908896f;
NK_STORAGE const float a2 = -1.21276126894734565e-2f;
NK_STORAGE const float a3 = -1.38078780785773762e-1f;
NK_STORAGE const float a4 = -2.67353392911981221e-2f;
NK_STORAGE const float a5 = +2.08026600266304389e-2f;
NK_STORAGE const float a6 = -3.03996055049204407e-3f;
NK_STORAGE const float a7 = +1.38235642404333740e-4f;
return a0 + x*(a1 + x*(a2 + x*(a3 + x*(a4 + x*(a5 + x*(a6 + x*a7))))));
}
#endif
#ifndef NK_COS
#define NK_COS nk_cos
NK_LIB float
nk_cos(float x)
{
/* New implementation. Also generated using lolremez. */
/* Old version significantly deviated from expected results. */
NK_STORAGE const float a0 = 9.9995999154986614e-1f;
NK_STORAGE const float a1 = 1.2548995793001028e-3f;
NK_STORAGE const float a2 = -5.0648546280678015e-1f;
NK_STORAGE const float a3 = 1.2942246466519995e-2f;
NK_STORAGE const float a4 = 2.8668384702547972e-2f;
NK_STORAGE const float a5 = 7.3726485210586547e-3f;
NK_STORAGE const float a6 = -3.8510875386947414e-3f;
NK_STORAGE const float a7 = 4.7196604604366623e-4f;
NK_STORAGE const float a8 = -1.8776444013090451e-5f;
return a0 + x*(a1 + x*(a2 + x*(a3 + x*(a4 + x*(a5 + x*(a6 + x*(a7 + x*a8)))))));
}
#endif
NK_LIB nk_uint
nk_round_up_pow2(nk_uint v)
{
v--;
v |= v >> 1;
v |= v >> 2;
v |= v >> 4;
v |= v >> 8;
v |= v >> 16;
v++;
return v;
}
NK_LIB double
nk_pow(double x, int n)
{
/* check the sign of n */
double r = 1;
int plus = n >= 0;
n = (plus) ? n : -n;
while (n > 0) {
if ((n & 1) == 1)
r *= x;
n /= 2;
x *= x;
}
return plus ? r : 1.0 / r;
}
NK_LIB int
nk_ifloord(double x)
{
x = (double)((int)x - ((x < 0.0) ? 1 : 0));
return (int)x;
}
NK_LIB int
nk_ifloorf(float x)
{
x = (float)((int)x - ((x < 0.0f) ? 1 : 0));
return (int)x;
}
NK_LIB int
nk_iceilf(float x)
{
if (x >= 0) {
int i = (int)x;
return (x > i) ? i+1: i;
} else {
int t = (int)x;
float r = x - (float)t;
return (r > 0.0f) ? t+1: t;
}
}
NK_LIB int
nk_log10(double n)
{
int neg;
int ret;
int exp = 0;
neg = (n < 0) ? 1 : 0;
ret = (neg) ? (int)-n : (int)n;
while ((ret / 10) > 0) {
ret /= 10;
exp++;
}
if (neg) exp = -exp;
return exp;
}
NK_API struct nk_vec2
nk_vec2(float x, float y)
{
struct nk_vec2 ret;
ret.x = x; ret.y = y;
return ret;
}
NK_API struct nk_vec2
nk_vec2i(int x, int y)
{
struct nk_vec2 ret;
ret.x = (float)x;
ret.y = (float)y;
return ret;
}
NK_API struct nk_vec2
nk_vec2v(const float *v)
{
return nk_vec2(v[0], v[1]);
}
NK_API struct nk_vec2
nk_vec2iv(const int *v)
{
return nk_vec2i(v[0], v[1]);
}
NK_LIB void
nk_unify(struct nk_rect *clip, const struct nk_rect *a, float x0, float y0,
float x1, float y1)
{
NK_ASSERT(a);
NK_ASSERT(clip);
clip->x = NK_MAX(a->x, x0);
clip->y = NK_MAX(a->y, y0);
clip->w = NK_MIN(a->x + a->w, x1) - clip->x;
clip->h = NK_MIN(a->y + a->h, y1) - clip->y;
clip->w = NK_MAX(0, clip->w);
clip->h = NK_MAX(0, clip->h);
}
NK_API void
nk_triangle_from_direction(struct nk_vec2 *result, struct nk_rect r,
float pad_x, float pad_y, enum nk_heading direction)
{
float w_half, h_half;
NK_ASSERT(result);
r.w = NK_MAX(2 * pad_x, r.w);
r.h = NK_MAX(2 * pad_y, r.h);
r.w = r.w - 2 * pad_x;
r.h = r.h - 2 * pad_y;
r.x = r.x + pad_x;
r.y = r.y + pad_y;
w_half = r.w / 2.0f;
h_half = r.h / 2.0f;
if (direction == NK_UP) {
result[0] = nk_vec2(r.x + w_half, r.y);
result[1] = nk_vec2(r.x + r.w, r.y + r.h);
result[2] = nk_vec2(r.x, r.y + r.h);
} else if (direction == NK_RIGHT) {
result[0] = nk_vec2(r.x, r.y);
result[1] = nk_vec2(r.x + r.w, r.y + h_half);
result[2] = nk_vec2(r.x, r.y + r.h);
} else if (direction == NK_DOWN) {
result[0] = nk_vec2(r.x, r.y);
result[1] = nk_vec2(r.x + r.w, r.y);
result[2] = nk_vec2(r.x + w_half, r.y + r.h);
} else {
result[0] = nk_vec2(r.x, r.y + h_half);
result[1] = nk_vec2(r.x + r.w, r.y);
result[2] = nk_vec2(r.x + r.w, r.y + r.h);
}
}展开全部(共 205 行)
分类说明
快速逼近函数
nk_inv_sqrt :经典的快速平方根倒数算法。通过 union 将 float 的位模式重新解释为整数,用魔数 0x5f375A84 做一次位操作逼近,再经一次牛顿迭代修正。对图形渲染中的法线归一化等场景精度通常足够。
注意事项:
在 C 中,读取 union 的非 active member 是标准允许的(C11 6.5.2.3),float/int 之间的位模式重解释在 IEEE 754 + 二进制补码平台上行为一致。但在 C++ 中(C++20 之前)这属于未定义行为。若项目同时以 C++ 编译,建议改用 memcpy 或 C++20 的 std::bit_cast。
输入 n 为 0 或负数时结果无意义,调用方需保证 n > 0。
nk_sin / nk_cos :使用 Horner 形式多项式逼近。nk_sin 为 7 次多项式(8 个系数),nk_cos 为 8 次多项式(9 个系数)。nk_cos 的代码注释表明系数由 lolremez 工具生成(该工具实现 Remez 交换算法)。
注意事项:
多项式逼近的有效输入范围有限。从系数特征推断,拟合区间约为 [-π/2, π/2] 附近,超出该范围精度会显著下降。调用前需将角度归约到主值区间。
精度有限,具体最大偏差需实测确认,不适合需要双精度或更高精度的科学计算。
整数与位操作函数
nk_round_up_pow2 :将无符号整数向上取整到 2 的幂,使用经典的位或展开序列。
注意事项:
输入 v = 0 时,v-- 使无符号值回绕为 UINT_MAX,经位或展开后 v++ 再次回绕为 0,最终返回 0(不是 2 的幂)。调用方需保证 v >= 1。
该实现假设 nk_uint 为 32 位(最大移位 16 位)。若平台使用 64 位无符号整数,需补充 v |= v >> 32。
nk_pow :快速幂(二进制取幂),仅支持整数指数 n。
注意事项:
指数为 int 类型,不支持浮点指数。
底数 x 为 double,中间乘法可能溢出,调用方需确保结果在 double 表示范围内。
n = 0 时返回 1.0(包括 x = 0 的情况),与常见计算约定一致(数学上 0^0 为不定式)。
nk_ifloord / nk_ifloorf :将浮点数向下取整为 int。
注意事项:
实现中先做 (int)x 截断(向零取整),再根据符号修正为向下取整。当 x 超出 int 表示范围时,(int)x 是未定义行为。
nk_iceilf :将 float 向上取整为 int。
注意事项:
同样受 int 范围限制,超出范围时行为未定义。
对 x 恰好为整数的情况,返回 x 本身(不多加 1),符合 ceil 定义。
nk_log10 :返回 n 的十进制对数的整数部分(即位数减一),而非完整的 log10 值。
注意事项:
仅返回整数,不返回小数部分。
n = 0 时 ret = 0,循环不执行,返回 0。但 log10(0) 在数学上无定义,调用方应避免传入 0。
负数输入会返回负的位数(如 n = -100 返回 -2),语义上不太直观,需调用方自行处理。
向量与几何函数
nk_vec2 系列 :构造二维向量的便捷函数,支持 float、int 和指针参数。纯赋值操作,无精度问题。
nk_unify :计算矩形 a 与由 (x0, y0, x1, y1) 定义的矩形的交集,结果写入 clip。
注意事项:
直接修改 clip 指针指向的结构体,调用方需确保 clip 有效且非空(代码中有 NK_ASSERT 检查)。
若两矩形无交集,clip->w 和 clip->h 被钳位为 0,但 clip->x 和 clip->y 仍为计算值(两矩形左/上边缘的较大者)。调用方应检查 w > 0 && h > 0 来判断是否有实际交集。
nk_triangle_from_direction :根据方向枚举(上 / 右 / 下 / 左)生成等腰三角形的三个顶点,用于绘制箭头或指示器。
注意事项:
pad_x 和 pad_y 用于在矩形内缩进,确保三角形不超出边界。
输出为 3 个 nk_vec2,调用方需分配足够空间。
direction 参数为 enum nk_heading,若传入未定义的枚举值,会落入 else 分支(即左方向)。
适用场景与限制
这些函数适合以下场景:
嵌入式或资源受限平台,无法链接完整 math 库。
即时模式 GUI 渲染,对单次调用的延迟敏感,但精度要求不高(像素级)。
需要减少外部依赖、保持单文件可移植性的项目。
不适用于:
需要 IEEE 754 精确结果的科学计算。
输入范围超出多项式拟合区间的三角函数调用。
需要 64 位整数幂次或浮点指数的幂运算。
验证建议
对 nk_inv_sqrt,可在 n > 0 的合理范围内与 1.0f / sqrtf(n) 对比,确认误差在可接受范围。
对 nk_sin / nk_cos,在拟合区间内与标准库对比,确认最大偏差;在区间外验证精度衰减程度。
对 nk_round_up_pow2,测试 v = 1, 2, 3, 4, 5, 1023, 1024, 1025 等边界值,以及 v = 0 的回绕行为(返回 0)。
对 nk_unify,测试完全重叠、部分重叠、无交集、一个包含另一个等情形。
对 nk_pow,测试 n = 0, 1, -1 以及较大的正负指数,确认无溢出。
小结
Nuklear 的数学函数是"够用就好"的典型设计:用图形学常见的近似技巧换取速度和零依赖,接口贴合 GUI 渲染需求。阅读源码时需注意各函数的输入范围、精度边界和平台假设,避免在超出设计意图的场景中直接使用。
<!-- csdn-article-id: 133337946 -->
本文最初于 2023/9/27 发布在 CSDN 。