From 49547ece8a2e6d23fa352ca528646394bd4aee0a Mon Sep 17 00:00:00 2001 From: Vladimir Smitka Date: Mon, 21 Sep 2026 07:30:23 +0000 Subject: [PATCH 1/6] picogame: project the basis dot products without a 64-bit multiply Eight of the ten multiplies per vertex in pg.project use a camera basis component, which is at most 65536 in Q16. For that range, splitting the other operand at bit 16 gives the same result with two 32-bit multiplies instead of a call to __aeabi_lmul. The two multiplies by k stay 64-bit, because k is not bounded. The helper is not inlined: inlined it is faster, but costs 632 bytes. RP2040, 368 vertices: 1804 -> 1066 us. Flash -80 bytes. Output is unchanged. Host check against the int64 form: 60M random pairs, no mismatch. --- shared-bindings/picogame/__init__.c | 22 +++++++++++++++++++--- 1 file changed, 19 insertions(+), 3 deletions(-) diff --git a/shared-bindings/picogame/__init__.c b/shared-bindings/picogame/__init__.c index 1812c3860fa..7f60152152a 100644 --- a/shared-bindings/picogame/__init__.c +++ b/shared-bindings/picogame/__init__.c @@ -589,6 +589,18 @@ static mp_obj_t picogame_fbm1d_fx(size_t n_args, const mp_obj_t *pos, mp_map_t * static MP_DEFINE_CONST_FUN_OBJ_KW(picogame_fbm1d_fx_obj, 1, picogame_fbm1d_fx); +#if !CIRCUITPY_PICOGAME_FPU +// ((int64_t)a * b) >> 16 with two 32-bit multiplies. Exact for |b| <= 65536. +static __attribute__((noinline)) int32_t pg_fmul_basis(int32_t a, int32_t b) { + uint32_t hi = (uint32_t)(a >> 16) * (uint32_t)b; // modulo 2^32 on purpose + uint32_t l = (uint32_t)(a & 0xFFFF); + if (b >= 0) { + return (int32_t)(hi + ((l * (uint32_t)b) >> 16)); + } + return (int32_t)(hi - ((l * (uint32_t)(-b) + 0xFFFFu) >> 16)); +} +#endif + //| def project( //| cam: ReadableBuffer, //| pts: ReadableBuffer, @@ -666,10 +678,13 @@ static mp_obj_t picogame_project(size_t n_args, const mp_obj_t *args) { // the near plane - host-measured 23-34 px warps on close fly-bys at a file-browser world scale // (walls visibly broke). Correctness first: Q16 keeps the worst error a few px at any cz >= near, // for coords up to +-32k units; still ~4-5x faster than the same math in Python on the M0+. + // FMULB: same result as FMUL for |b| <= 65536 (every basis component), without + // __aeabi_lmul. The products by k keep FMUL because k is not bounded. #define FMUL(a, b) ((int32_t)(((int64_t)(a) * (b)) >> 16)) + #define FMULB(a, b) pg_fmul_basis((a), (b)) for (int i = 0; i < n; i++) { int32_t X = pts[i * 3] - ex, Y = pts[i * 3 + 1] - ey, Z = pts[i * 3 + 2] - ez; - int32_t cz = FMUL(X, fx) + FMUL(Y, fy) + FMUL(Z, fz); + int32_t cz = FMULB(X, fx) + FMULB(Y, fy) + FMULB(Z, fz); if (cz < near) { osx[i] = -32768; osy[i] = -32768; @@ -679,12 +694,13 @@ static mp_obj_t picogame_project(size_t n_args, const mp_obj_t *args) { // an int64 divide on the M0+ (no HW divide) and the lost cz precision costs <0.02 px (host- // measured). Needs FOCAL < ~250 (focal<<8 in uint32) and near >= 1/256 (cz>>8 nonzero). int32_t k = (int32_t)(((uint32_t)focal << 8) / (uint32_t)(cz >> 8)); - int32_t rr = FMUL(X, rx) + FMUL(Z, rz); - int32_t uu = FMUL(X, ux) + FMUL(Y, uy) + FMUL(Z, uz); + int32_t rr = FMULB(X, rx) + FMULB(Z, rz); + int32_t uu = FMULB(X, ux) + FMULB(Y, uy) + FMULB(Z, uz); osx[i] = (int16_t)((cx0 + FMUL(rr, k)) >> 16); osy[i] = (int16_t)((cy0 - FMUL(uu, k)) >> 16); } #undef FMUL +#undef FMULB #endif return mp_const_none; } From c738a3e948cb0e10b889b14802d39a2998fa4792 Mon Sep 17 00:00:00 2001 From: Vladimir Smitka Date: Mon, 21 Sep 2026 08:04:43 +0000 Subject: [PATCH 2/6] picogame: take the divide and the 64-bit multiplies out of the road curve Each curvature step in pg.road_edges did two modulo calls per sine lerp and four calls to __aeabi_lmul. - pg_sin_q15_lerp reduces the angle once and reads both table entries without a second modulo. - The lerp product fits in 32 bits (entries differ by less than 600, frac is below 2^16). - The amplitude multiplies use pg_mulshr15, exact for |s| <= 2^15 and any int32 amplitude. RP2040, 170 rows, curve_step 2: 490 -> 301 us. Flash +16 bytes. Output is unchanged. Host check of pg_mulshr15 against the int64 form: every Q15 sine value against a sweep of int32, no mismatch. --- shared-module/picogame/__init__.c | 36 ++++++++++++++++++++++--------- 1 file changed, 26 insertions(+), 10 deletions(-) diff --git a/shared-module/picogame/__init__.c b/shared-module/picogame/__init__.c index ee69e6ec4ac..8be10a47343 100644 --- a/shared-module/picogame/__init__.c +++ b/shared-module/picogame/__init__.c @@ -491,11 +491,8 @@ static const int16_t pg_sin_q15_quad[91] = { 32269, 32364, 32448, 32523, 32587, 32642, 32687, 32722, 32747, 32762, 32767, }; -static int32_t pg_sin_q15(int deg) { - deg %= 360; - if (deg < 0) { - deg += 360; - } +// deg must already be in 0..360 (360 == 0). +static int32_t pg_sin_q15_reduced(int deg) { if (deg <= 90) { return pg_sin_q15_quad[deg]; } @@ -507,6 +504,13 @@ static int32_t pg_sin_q15(int deg) { } return -pg_sin_q15_quad[360 - deg]; } +static int32_t pg_sin_q15(int deg) { + deg %= 360; + if (deg < 0) { + deg += 360; + } + return pg_sin_q15_reduced(deg); +} static int32_t pg_cos_q15(int deg) { return pg_sin_q15(deg + 90); } @@ -515,10 +519,22 @@ static int32_t pg_cos_q15(int deg) { // DOUBLE-integrated over ~170 rows, which amplifies whole-degree quantization into visible pixels // (host-measured 9 px); one lerp per curvature eval brings the road within 1 px of the float original. static int32_t pg_sin_q15_lerp(int64_t deg_q16) { - int d0 = (int)(deg_q16 >> 16); + // Reduce once; d0 + 1 is then at most 360, which the table handles. + int d0 = (int)(deg_q16 >> 16) % 360; + if (d0 < 0) { + d0 += 360; + } int32_t frac = (int32_t)(deg_q16 & 0xFFFF); - int32_t a = pg_sin_q15(d0); - return a + (int32_t)(((int64_t)(pg_sin_q15(d0 + 1) - a) * frac) >> 16); + int32_t a = pg_sin_q15_reduced(d0); + // Adjacent entries differ by < 600 and frac < 2^16, so this fits in 32 bits. + return a + (((pg_sin_q15_reduced(d0 + 1) - a) * frac) >> 16); +} + +// (s * amp) >> 15 with 32-bit multiplies. Exact for |s| <= 2^15 and any int32 amp. +static inline int32_t pg_mulshr15(int32_t s, int32_t amp) { + uint32_t hi = (uint32_t)(amp >> 15) * (uint32_t)s; + int32_t lo = ((amp & 0x7FFF) * s) >> 15; + return (int32_t)(hi + (uint32_t)lo); } // One racing-road frame's curve pass: the bottom-up curvature accumulator + per-row integer edges @@ -539,8 +555,8 @@ void picogame_road_edges(int16_t *rl, int16_t *rr, const int32_t *hw_q16, int n, for (int i = n - 1; i >= 0; i--) { if (cnt == 0) { int32_t d = dist + (drow - i) * wstep; - ck = (int32_t)(((int64_t)pg_sin_q15_lerp(((int64_t)d * f1) >> 4) * a1k) >> 15) - + (int32_t)(((int64_t)pg_sin_q15_lerp(((int64_t)d * f2) >> 4) * a2k) >> 15); + ck = pg_mulshr15(pg_sin_q15_lerp(((int64_t)d * f1) >> 4), a1k) + + pg_mulshr15(pg_sin_q15_lerp(((int64_t)d * f2) >> 4), a2k); cnt = cstep; } cnt--; From 43aa0361f722a3ec5586f2b0363f84b901b0a331 Mon Sep 17 00:00:00 2001 From: Vladimir Smitka Date: Mon, 21 Sep 2026 18:55:40 +0000 Subject: [PATCH 3/6] picogame: exact 32-bit forms for the raycaster and the value noise raycast: 2^32 / v is computed as (0 - v) / v + 1, one 32-bit divide. The clamp comes first (ax < 256), which also guards against v = 0. sideDist uses a split multiply that is exact for its input range. The DDA loop keeps a linear map offset and uses unsigned bounds checks. value noise: smooth16 and lerp16 are exact in 32 bits, and the four corner hashes share the coordinate multiplies (16 multiplies -> 7). RP2040: raycast dense map 717 -> 427 us, open map 1156 -> 829 us. Flash -168 bytes. Output is unchanged. Host check: the reciprocal over [256, 2^24] plus 20M random values, the split multiply over 110M pairs, no mismatch. Co-authored-by: Petr Vavrin --- shared-bindings/picogame/__init__.c | 74 ++++++++++++++++++++--------- 1 file changed, 51 insertions(+), 23 deletions(-) diff --git a/shared-bindings/picogame/__init__.c b/shared-bindings/picogame/__init__.c index 7f60152152a..5ccd1df548b 100644 --- a/shared-bindings/picogame/__init__.c +++ b/shared-bindings/picogame/__init__.c @@ -502,27 +502,42 @@ static mp_obj_t picogame_fbm1d(size_t n_args, const mp_obj_t *pos, mp_map_t *kw) static MP_DEFINE_CONST_FUN_OBJ_KW(picogame_fbm1d_obj, 1, picogame_fbm1d); #endif // float fbm2d / fbm1d +// floor((a * dd) / 65536) with 32-bit multiplies. Exact for a <= 2^16, dd <= 2^24. +static inline int32_t pg_mulshr16(uint32_t a, uint32_t dd) { + return (int32_t)(a * (dd >> 16) + ((a * (dd & 0xFFFF)) >> 16)); +} + // ---- fixed-point (Q16.16 coords, Q0.16 values) noise: the CANONICAL value-noise impl, // exposed under the plain names value2d/value1d/fbm2d/fbm1d. The inner math is integer // (float only at the Python boundary); ~1.8x faster than the retired float path. ---- -static inline uint32_t pg_nhash_raw(int32_t x, int32_t y, int32_t seed) { - uint32_t h = (uint32_t)x * 374761393u + (uint32_t)y * 668265263u + (uint32_t)seed * 362437u; + +// Mix step of the hash. The four corners of a cell differ only by constants +// in the input, so the coordinate part is computed once per cell. +static inline uint32_t pg_nhash_mix(uint32_t h) { h = (h ^ (h >> 13)) * 1274126177u; return (h ^ (h >> 16)) & 0xFFFFu; // Q0.16 in [0,1) } static inline uint32_t pg_smooth16(uint32_t t) { // t,result Q0.16: t*t*(3-2t) - uint32_t t2 = (t * t) >> 16; - uint32_t e = (3u << 16) - 2u * t; - return (uint32_t)(((uint64_t)t2 * e) >> 16); + uint32_t t2 = (t * t) >> 16; // t <= 0xFFFF, so t*t fits uint32 + uint32_t e = (3u << 16) - 2u * t; // e <= 3<<16, i.e. under 2^18 + return (uint32_t)pg_mulshr16(t2, e); // t2 <= 2^16 and e <= 2^24: exact } static inline uint32_t pg_lerp16(uint32_t a, uint32_t b, uint32_t u) { - return (uint32_t)((int32_t)a + (int32_t)(((int64_t)((int32_t)b - (int32_t)a) * (int32_t)u) >> 16)); + // Unsigned product with a sign split; a negative delta must floor like the old shift. + int32_t d = (int32_t)b - (int32_t)a; + if (d >= 0) { + return a + (((uint32_t)d * u) >> 16); + } + return a - ((((uint32_t)(-d) * u) + 0xFFFFu) >> 16); } static uint32_t pg_value2d_fx(int32_t X, int32_t Y, int32_t seed) { // X,Y Q16.16 -> Q0.16 int32_t xi = X >> 16, yi = Y >> 16; uint32_t xf = (uint32_t)(X - (xi << 16)), yf = (uint32_t)(Y - (yi << 16)); - uint32_t a = pg_nhash_raw(xi, yi, seed), b = pg_nhash_raw(xi + 1, yi, seed); - uint32_t c = pg_nhash_raw(xi, yi + 1, seed), d = pg_nhash_raw(xi + 1, yi + 1, seed); + uint32_t base = (uint32_t)xi * 374761393u + (uint32_t)yi * 668265263u + (uint32_t)seed * 362437u; + uint32_t a = pg_nhash_mix(base); // (xi, yi) + uint32_t b = pg_nhash_mix(base + 374761393u); // (xi+1, yi) + uint32_t c = pg_nhash_mix(base + 668265263u); // (xi, yi+1) + uint32_t d = pg_nhash_mix(base + 374761393u + 668265263u); uint32_t u = pg_smooth16(xf), v = pg_smooth16(yf); return pg_lerp16(pg_lerp16(a, b, u), pg_lerp16(c, d, u), v); } @@ -769,8 +784,7 @@ static MP_DEFINE_CONST_FUN_OBJ_VAR_BETWEEN(picogame_project_obj, 5, 5, picogame_ // trig and passes Q16 ray params). map: read-only bytes, mw*mh wall types (0 = empty). pos*, l*x/l*y // (leftRay, column 0), s*x/s*y (rayStep per column) are all 16.16. wcolors: uint16[(maxtype+1)*2] - // [t*2] near, [t*2+1] side colour. top/bot/col: uint16 write buffers (len>=ncols); dist: int32 write -// buffer (perpendicular distance, 16.16). The int64 divides/muls are ONLY the per-column setup -// (O(ncols)); the DDA step loop is pure 32-bit. Mirrors the Python float fallback closely. +// buffer (perpendicular distance, 16.16). All arithmetic is 32-bit. // Optional arg 17 (runs - ONE uint16 write buffer, len>=5*ncols, laid out as five ncols-long // planes [x0s | x1s | tops | bots | colors]): also emit the RLE-MERGED wall runs (adjacent equal // columns fused; x in PIXELS = column*stride) and return the run count. The planes feed @@ -778,11 +792,18 @@ static MP_DEFINE_CONST_FUN_OBJ_VAR_BETWEEN(picogame_project_obj, 5, 5, picogame_ // loop into the same C pass (measured 2-6.5 ms/frame of interpreted merge at stride=1 on RP2040). // Callers clamp the LAST run's x1 to the screen width (stride rounding can overshoot by = 1. +static inline uint32_t pg_recip32(uint32_t v) { + return (uint32_t)(0u - v) / v + 1u; +} + static mp_obj_t picogame_raycast(size_t n_args, const mp_obj_t *args) { mp_buffer_info_t mi, wi, ti, bi, ci, di; mp_get_buffer_raise(args[0], &mi, MP_BUFFER_READ); - int mw = mp_obj_get_int(args[1]); - int mh = mp_obj_get_int(args[2]); + // Non-negative: the DDA below bounds-checks with unsigned compares. + int mw = picogame_imax(mp_obj_get_int(args[1]), 0); + int mh = picogame_imax(mp_obj_get_int(args[2]), 0); int32_t posx = mp_obj_get_int(args[3]); // camera x, 16.16 int32_t posy = mp_obj_get_int(args[4]); int32_t rdx = mp_obj_get_int(args[5]); // leftRay x (column 0), 16.16 - accumulates per column @@ -821,6 +842,7 @@ static mp_obj_t picogame_raycast(size_t n_args, const mp_obj_t *args) { int half = sh >> 1; int imapx0 = posx >> 16; int imapy0 = posy >> 16; + int off0 = imapy0 * mw + imapx0; // every column starts from the same cell int32_t fracx = posx & 0xFFFF; // fractional part of pos, 16.16 int32_t fracy = posy & 0xFFFF; const int32_t DD_CAP = (int32_t)1 << 24; // cap deltaDist so a 64-step accumulation stays in int32 @@ -832,41 +854,47 @@ static mp_obj_t picogame_raycast(size_t n_args, const mp_obj_t *args) { int mapy = imapy0; int32_t ax = rdx < 0 ? -rdx : rdx; int32_t ay = rdy < 0 ? -rdy : rdy; - // deltaDist = |1/rayDir| in 16.16 = (1<<32)/|rayDir_q16| (int64; per-column setup, not per-step) - int64_t ddx64 = ax ? (((int64_t)1 << 32) / ax) : (int64_t)DD_CAP; - int64_t ddy64 = ay ? (((int64_t)1 << 32) / ay) : (int64_t)DD_CAP; - int32_t ddx = ddx64 > DD_CAP ? DD_CAP : (int32_t)ddx64; - int32_t ddy = ddy64 > DD_CAP ? DD_CAP : (int32_t)ddy64; + // deltaDist = 2^32 / |rayDir|, capped at DD_CAP. ax < 256 is exactly the capped + // range, so the test also guards against ax == 0. + int32_t ddx = (ax < 256) ? DD_CAP : (int32_t)pg_recip32((uint32_t)ax); + int32_t ddy = (ay < 256) ? DD_CAP : (int32_t)pg_recip32((uint32_t)ay); int stepx, stepy; int32_t sidex, sidey; - // sideDist to the first grid line = (fractional distance) * deltaDist, 16.16 (int64 mul, setup only) + // sideDist to the first grid line = (fractional distance) * deltaDist, 16.16 (setup only) if (rdx < 0) { stepx = -1; - sidex = (int32_t)(((int64_t)fracx * ddx) >> 16); + sidex = pg_mulshr16((uint32_t)fracx, (uint32_t)ddx); } else { stepx = 1; - sidex = (int32_t)(((int64_t)(65536 - fracx) * ddx) >> 16); + sidex = pg_mulshr16((uint32_t)(65536 - fracx), (uint32_t)ddx); } if (rdy < 0) { stepy = -1; - sidey = (int32_t)(((int64_t)fracy * ddy) >> 16); + sidey = pg_mulshr16((uint32_t)fracy, (uint32_t)ddy); } else { stepy = 1; - sidey = (int32_t)(((int64_t)(65536 - fracy) * ddy) >> 16); + sidey = pg_mulshr16((uint32_t)(65536 - fracy), (uint32_t)ddy); } int side = 0; int cell = 1; + // Linear map offset, stepped with the cell coordinates. + int off = off0; + int stepy_off = stepy * mw; for (int i = 0; i < 64; i++) { // DDA - pure 32-bit if (sidex < sidey) { sidex += ddx; mapx += stepx; + off += stepx; side = 0; } else { sidey += ddy; mapy += stepy; + off += stepy_off; side = 1; } - cell = (mapx >= 0 && mapx < mw && mapy >= 0 && mapy < mh) ? map[mapy * mw + mapx] : 1; + // Unsigned compares also reject negative coordinates. + cell = ((unsigned)mapx < (unsigned)mw && (unsigned)mapy < (unsigned)mh) + ? map[off] : 1; if (cell) { break; } From f3884847321eeeb5da9b130b92e6dd752081f5c4 Mon Sep 17 00:00:00 2001 From: Vladimir Smitka Date: Mon, 21 Sep 2026 18:56:00 +0000 Subject: [PATCH 4/6] picogame: hoist the sprite blit's row loop, and copy misaligned rows by hand - The row loop no longer recomputes the flip, transparency and effect selectors and the row addresses on every row. - An opaque RGB565 row whose source and destination differ in word alignment is copied with a halfword loop. The bootrom memcpy takes 13.9 cycles/px for those rows. - The scaled, rotated and mode7 sampler uses -1 as the key of an opaque bitmap, so it does one compare per pixel. PAL8 keys are narrowed to 8 bits in the Bitmap constructor. The unscaled loops keep the flag: GCC clones them on it, and a sentinel made opaque PAL8 33% slower. - The affine loop uses unsigned bounds checks. - The compositor skips a sprite on strips it does not touch. RP2040: RGB565 blit at an odd x 523 -> 235 us, PAL8 blit 366 -> 338 us. Output is unchanged. Co-authored-by: Petr Vavrin --- shared-bindings/picogame/Bitmap.c | 4 +- shared-module/picogame/Canvas.c | 8 +- shared-module/picogame/__init__.c | 132 ++++++++++++++++++++---------- shared-module/picogame/__init__.h | 7 +- 4 files changed, 101 insertions(+), 50 deletions(-) diff --git a/shared-bindings/picogame/Bitmap.c b/shared-bindings/picogame/Bitmap.c index 8c970a5a0c0..46a4443891a 100644 --- a/shared-bindings/picogame/Bitmap.c +++ b/shared-bindings/picogame/Bitmap.c @@ -150,7 +150,9 @@ static mp_obj_t picogame_bitmap_make_new(const mp_obj_type_t *type, size_t n_arg // palette length in entries (informational; blitter assumes indices < this - see blit contract). self->pal_entries = (uint16_t)((pal_len / 2) > 65535 ? 65535 : (pal_len / 2)); if (args[ARG_transparent].u_obj != mp_const_none) { - self->transparent = mp_obj_get_int(args[ARG_transparent].u_obj); + mp_int_t t = mp_obj_get_int(args[ARG_transparent].u_obj); + // PAL8 keys are indexes: narrow to 8 bits so every blit path agrees. + self->transparent = (format == PICOGAME_FMT_PAL8) ? (uint8_t)t : (uint16_t)t; self->has_transparent = true; } else { self->transparent = 0; diff --git a/shared-module/picogame/Canvas.c b/shared-module/picogame/Canvas.c index b51ea7084d6..e798e136226 100644 --- a/shared-module/picogame/Canvas.c +++ b/shared-module/picogame/Canvas.c @@ -179,7 +179,7 @@ typedef struct { int fmt, stride, shx, shy, mx, my, horizon, y_off; int32_t z, rx0, ry0, rsx, rsy, cam_x, cam_y; bool transp; - uint16_t key; + int32_t key; } mode7_ctx_t; static void mode7_rows(void *arg, int lo, int hi) { @@ -217,7 +217,7 @@ static void mode7_rows(void *arg, int lo, int hi) { for (int sx = 0; sx < w; sx++) { int tx = (fx >> c->shx) & c->mx, ty = (fy >> c->shy) & c->my; uint16_t val; - if (src_pixel_s(c->fmt, c->data, c->pal, c->transp, c->key, ty * c->stride + tx, &val)) { + if (src_pixel_s(c->fmt, c->data, c->pal, c->key, ty * c->stride + tx, &val)) { drow[sx] = val; } fx += stepx; @@ -247,8 +247,8 @@ void picogame_canvas_mode7(picogame_canvas_obj_t *cv, picogame_bitmap_obj_t *tex int fmt = tex->format; const uint8_t *data = tex->data; const uint16_t *pal = tex->palette; - bool transp = tex->has_transparent; - uint16_t key = tex->transparent; + bool transp = tex->has_transparent; // still gates the interp fast path above + int32_t key = picogame_key_of(tex); // sy is a row WITHIN this surface (a StripDraw view is a Canvas onto one strip); // the absolute screen row is sy + y_off, so the horizon test uses that. y_off = 0 // for a full-screen Canvas, = the strip's screen y for a StripDraw view (0-RAM floor). diff --git a/shared-module/picogame/__init__.c b/shared-module/picogame/__init__.c index 8be10a47343..3845f9f967b 100644 --- a/shared-module/picogame/__init__.c +++ b/shared-module/picogame/__init__.c @@ -115,6 +115,19 @@ static inline void picogame_fx_put(uint16_t *dst, uint16_t src, int x, int y, co } } +// Copy one row. memcpy when both pointers share word alignment; otherwise a +// halfword loop, because the bootrom memcpy is slow for misaligned copies. +static inline void picogame_row_copy(uint16_t *dst, const uint16_t *src, int n) { + if ((((uintptr_t)dst ^ (uintptr_t)src) & 2u) == 0) { + memcpy(dst, src, (size_t)n * 2u); + return; + } + #pragma GCC unroll 4 + for (int i = 0; i < n; i++) { + dst[i] = src[i]; + } +} + // Fetch one source pixel from HOISTED scalars: the caller lifts format/data/palette/transparency // out of the bitmap struct ONCE before its loop, so this does no per-pixel reload of bm fields (a // `*dst` uint16_t store would otherwise force GCC to reload bm's uint16_t members every pixel). `idx` @@ -165,8 +178,7 @@ void picogame_blit_bitmap( int t_fmt = bm->format; // hoist bm fields once (see src_pixel_s) const uint8_t *t_data = bm->data; const uint16_t *t_pal = bm->palette; - bool t_transp = bm->has_transparent; - uint16_t t_key = bm->transparent; + int32_t t_key = picogame_key_of(bm); for (int y = ys; y < ye; y++) { int ly = y - dy0; // -> source X (0..sw-1) int su = fx ? sw - 1 - ly : ly; // per-row: source column is constant across the row @@ -176,7 +188,7 @@ void picogame_blit_bitmap( int svstep = fy ? -1 : 1; for (int x = xs; x < xe; x++) { uint16_t val; - if (src_pixel_s(t_fmt, t_data, t_pal, t_transp, t_key, + if (src_pixel_s(t_fmt, t_data, t_pal, t_key, sv * stride0 + frame_col0 + su, &val)) { picogame_fx_put(dst, val, x, y, fxm); } @@ -198,6 +210,25 @@ void picogame_blit_bitmap( int frame_col = frame * sw; int stride = bm->stride; bool transp = bm->has_transparent; + int run = x_end - x_start; + int nrows = y_end - y_start; + + // Row invariants are computed once; GCC only unswitches the innermost loop. + int sy0 = y_start - dy0; + int systep = 1; + if (fy) { + sy0 = sh - 1 - sy0; + systep = -1; + } + int sx0 = x_start - dx0; + int xstep = 1; + if (fx) { + sx0 = sw - 1 - sx0; + xstep = -1; + } + int srow = sy0 * stride + frame_col; + int srow_step = systep * stride; + uint16_t *dstrow = buf + (y_start - oy) * bw + (x_start - ox); if (bm->format == PICOGAME_FMT_PAL8) { const uint8_t *data = bm->data; @@ -211,21 +242,15 @@ void picogame_blit_bitmap( // TO RESTORE FULL BOUNDS-SAFETY (at that cost) reinstate the clamp - add `unsigned pe = // bm->pal_entries;` here and `if (idx >= pe) { idx = 0; }` after each `idx = data[...]` in BOTH // loops below, and the matching guard in src_pixel() (search "blit contract"). - for (int y = y_start; y < y_end; y++) { - int sy = y - dy0; - if (fy) { - sy = sh - 1 - sy; - } - int srow = sy * stride + frame_col; - uint16_t *dst = buf + (y - oy) * bw + (x_start - ox); - int sx = x_start - dx0, xstep = 1; // hoist flip_x: walk sx +/-1, no per-pixel test - if (fx) { - sx = sw - 1 - sx; - xstep = -1; - } - if (fxm == NULL) { // plain copy (most sprites): no per-pixel fx branch/call + // + // Keep the transp flag here, not a -1 key: GCC clones these loops on the flag, + // and a sentinel made opaque PAL8 33% slower. + if (fxm == NULL) { // plain copy (most sprites): no per-pixel fx branch/call + for (int i = 0; i < nrows; i++) { + uint16_t *dst = dstrow; + int sx = sx0; #pragma GCC unroll 4 // hot path: unrolling the plain sprite blit is ~6% faster on M0+ (measured), +0.6KB - for (int x = x_start; x < x_end; x++) { + for (int x = 0; x < run; x++) { uint8_t idx = data[srow + sx]; if (!transp || idx != key) { *dst = pal[idx]; @@ -233,7 +258,14 @@ void picogame_blit_bitmap( dst++; sx += xstep; } - } else { + srow += srow_step; + dstrow += bw; + } + } else { + // The effect path keeps x/y as loop variables: DITHER uses screen coordinates. + for (int y = y_start; y < y_end; y++) { + uint16_t *dst = dstrow; + int sx = sx0; for (int x = x_start; x < x_end; x++) { uint8_t idx = data[srow + sx]; if (!transp || idx != key) { @@ -242,6 +274,8 @@ void picogame_blit_bitmap( dst++; sx += xstep; } + srow += srow_step; + dstrow += bw; } } } else { // PICOGAME_FMT_RGB565 @@ -252,26 +286,19 @@ void picogame_blit_bitmap( const uint16_t *data = (const uint16_t *)bm->data; #pragma GCC diagnostic pop uint16_t key = bm->transparent; - for (int y = y_start; y < y_end; y++) { - int sy = y - dy0; - if (fy) { - sy = sh - 1 - sy; - } - int srow = sy * stride + frame_col; - uint16_t *dst = buf + (y - oy) * bw + (x_start - ox); - int sx = x_start - dx0, xstep = 1; // hoist flip_x: walk sx +/-1, no per-pixel test - if (fx) { - sx = sw - 1 - sx; - xstep = -1; + if (fxm == NULL && !transp && !fx) { + // Opaque, not x-flipped: one block copy per row. + for (int i = 0; i < nrows; i++) { + picogame_row_copy(dstrow, &data[srow + sx0], run); + srow += srow_step; + dstrow += bw; } - if (fxm == NULL) { // plain copy (most sprites): no per-pixel fx branch/call - if (!transp && !fx) { // opaque + not x-flipped: the row is contiguous in - // both src and dst -> one memcpy (dst may be 2-byte aligned; memcpy handles that). - memcpy(dst, &data[srow + sx], (size_t)(x_end - x_start) * 2u); - continue; - } + } else if (fxm == NULL) { // plain copy (most sprites): no per-pixel fx branch/call + for (int i = 0; i < nrows; i++) { + uint16_t *dst = dstrow; + int sx = sx0; #pragma GCC unroll 4 // hot path: unrolling the plain sprite blit is ~6% faster on M0+ (measured), +0.6KB - for (int x = x_start; x < x_end; x++) { + for (int x = 0; x < run; x++) { uint16_t v = data[srow + sx]; if (!transp || v != key) { *dst = v; @@ -279,7 +306,14 @@ void picogame_blit_bitmap( dst++; sx += xstep; } - } else { + srow += srow_step; + dstrow += bw; + } + } else { + // The effect path keeps x/y as loop variables: DITHER uses screen coordinates. + for (int y = y_start; y < y_end; y++) { + uint16_t *dst = dstrow; + int sx = sx0; for (int x = x_start; x < x_end; x++) { uint16_t v = data[srow + sx]; if (!transp || v != key) { @@ -288,6 +322,8 @@ void picogame_blit_bitmap( dst++; sx += xstep; } + srow += srow_step; + dstrow += bw; } } } @@ -327,8 +363,8 @@ void picogame_blit_bitmap_scaled( int s_fmt = bm->format; // hoist bm fields once (see src_pixel_s) const uint8_t *s_data = bm->data; const uint16_t *s_pal = bm->palette; - bool s_transp = bm->has_transparent; - uint16_t s_key = bm->transparent; + bool s_transp = bm->has_transparent; // still selects the opaque fast paths below + int32_t s_key = picogame_key_of(bm); if (scale == 512 && !fx && !fy && fxm == NULL && !s_transp && s_fmt == PICOGAME_FMT_RGB565 && (((uintptr_t)s_data & 1) == 0)) { // 2x integer upscale fast path - the half-res-canvas genre's per-frame blit (a full-screen @@ -470,7 +506,7 @@ void picogame_blit_bitmap_scaled( sx = sw - 1 - sx; } uint16_t val; - if (src_pixel_s(s_fmt, s_data, s_pal, s_transp, s_key, srow + sx, &val)) { + if (src_pixel_s(s_fmt, s_data, s_pal, s_key, srow + sx, &val)) { picogame_fx_put(&drow[x - ox], val, x, y, fxm); } } @@ -664,8 +700,7 @@ void picogame_blit_bitmap_affine( int a_fmt = bm->format; // hoist bm fields once (see src_pixel_s) const uint8_t *a_data = bm->data; const uint16_t *a_pal = bm->palette; - bool a_transp = bm->has_transparent; - uint16_t a_key = bm->transparent; + int32_t a_key = picogame_key_of(bm); int x_start = picogame_imax(minx, ox), y_start = picogame_imax(miny, oy); int x_end = picogame_imin(maxx + 1, ox + bw), y_end = picogame_imin(maxy + 1, oy + bh); if (x_start >= x_end || y_start >= y_end) { @@ -701,9 +736,10 @@ void picogame_blit_bitmap_affine( int su = uacc >> 16, sv = vacc >> 16; // already the flipped source coords uacc += uxc; vacc += vxc; - if (su >= 0 && su < sw && sv >= 0 && sv < sh) { + // Unsigned compares also reject negative coordinates. + if ((unsigned)su < (unsigned)sw && (unsigned)sv < (unsigned)sh) { uint16_t val; - if (src_pixel_s(a_fmt, a_data, a_pal, a_transp, a_key, + if (src_pixel_s(a_fmt, a_data, a_pal, a_key, sv * stride + frame_col + su, &val)) { picogame_fx_put(&drow[x - ox], val, x, y, fxm); } @@ -927,6 +963,14 @@ mp_obj_t picogame_blit_strip_layers( if (!(spr->flags & PICOGAME_SPR_VISIBLE)) { continue; } + // CIRCUITPY-CHANGE: skip strips the sprite does not touch before entering the + // blitter, which picks the effect and may bake a palette first. + int ax1, ay1, ax2, ay2; + picogame_sprite_aabb(spr, &ax1, &ay1, &ax2, &ay2); + if (ay1 + ioy >= strip_top + strip_h || ay2 + ioy <= strip_top || + ax1 + iox >= x0 + region_w || ax2 + iox <= x0) { + continue; + } blit_sprite(buf, region_w, strip_h, x0, strip_top, spr, iox, ioy); } } diff --git a/shared-module/picogame/__init__.h b/shared-module/picogame/__init__.h index 412856716b5..39a0d7bc274 100644 --- a/shared-module/picogame/__init__.h +++ b/shared-module/picogame/__init__.h @@ -44,13 +44,18 @@ static inline bool src_pixel_s(int format, const uint8_t *data, const uint16_t * #pragma GCC diagnostic ignored "-Wcast-align" uint16_t v = ((const uint16_t *)data)[idx]; #pragma GCC diagnostic pop - if (transp && v == key) { + if ((int32_t)v == key) { return false; } *out = v; return true; } +// The transparent key of bm, or -1 when it is opaque. +static inline int32_t picogame_key_of(const picogame_bitmap_obj_t *bm) { + return bm->has_transparent ? (int32_t)bm->transparent : -1; +} + // Scene layer kinds (tags stored alongside items so blit/dirty can dispatch // without cross-referencing shared-bindings type objects). From b5849cdfa88e4ff39e0fe5b74e84c57caa3a368b Mon Sep 17 00:00:00 2001 From: Vladimir Smitka Date: Mon, 21 Sep 2026 18:56:22 +0000 Subject: [PATCH 5/6] picogame: word-fill the surfaces, block-fill the maps - fill565 writes four words per iteration through a pointer. The indexed loop recomputed the byte offset for every word. - Spans of 4 pixels or less are written directly, before the word path. - Tilemap.fill and the Canvas constructor use block fills. The Canvas constructor resets the dirty area before and after clear(). RP2040: Tilemap.fill 128x128 1747 -> 83 us, Canvas 240x135 2411 -> 318 us, clear 542 -> 411 us, fill_rect 466 -> 367 us. Output is unchanged. Co-authored-by: Petr Vavrin --- shared-bindings/picogame/Canvas.c | 7 +++---- shared-bindings/picogame/Tilemap.c | 10 +++++----- shared-module/picogame/Canvas.c | 13 +++++++++---- shared-module/picogame/__init__.c | 4 +--- shared-module/picogame/__init__.h | 24 ++++++++++++++++++++++-- 5 files changed, 40 insertions(+), 18 deletions(-) diff --git a/shared-bindings/picogame/Canvas.c b/shared-bindings/picogame/Canvas.c index a3efbdc0312..fde65572832 100644 --- a/shared-bindings/picogame/Canvas.c +++ b/shared-bindings/picogame/Canvas.c @@ -81,10 +81,9 @@ static mp_obj_t picogame_canvas_make_new(const mp_obj_type_t *type, size_t n_arg self->transparent = 0; self->has_transparent = false; } - uint16_t fill = self->has_transparent ? self->transparent : 0; - for (size_t i = 0; i < (size_t)w * h; i++) { - self->data[i] = fill; - } + // clear() marks the surface dirty and reads the accumulator, so reset it on both sides. + picogame_canvas_dirty_reset(self); + picogame_canvas_clear(self, self->has_transparent ? self->transparent : 0); picogame_canvas_dirty_reset(self); return MP_OBJ_FROM_PTR(self); } diff --git a/shared-bindings/picogame/Tilemap.c b/shared-bindings/picogame/Tilemap.c index 955deef1b96..a0a307d2f37 100644 --- a/shared-bindings/picogame/Tilemap.c +++ b/shared-bindings/picogame/Tilemap.c @@ -4,6 +4,8 @@ // // SPDX-License-Identifier: MIT +#include + #include "py/runtime.h" #include "shared-module/picogame/pg_compat.h" #include "shared-bindings/picogame/Tilemap.h" @@ -175,11 +177,9 @@ static mp_obj_t picogame_tilemap_fill(mp_obj_t self_in, mp_obj_t value_in) { picogame_tilemap_obj_t *self = MP_OBJ_TO_PTR(self_in); uint8_t v = mp_obj_get_int(value_in) & 0xff; size_t total = (size_t)self->map_w * self->map_h; - for (size_t i = 0; i < total; i++) { - self->map[i] = v; - if (self->orient) { - self->orient[i] = 0; // a plain fill clears any per-cell orientation - } + memset(self->map, v, total); + if (self->orient) { + memset(self->orient, 0, total); // a plain fill clears any per-cell orientation } int x1, y1, x2, y2; picogame_tilemap_extent(self, &x1, &y1, &x2, &y2); diff --git a/shared-module/picogame/Canvas.c b/shared-module/picogame/Canvas.c index e798e136226..618cadb8a69 100644 --- a/shared-module/picogame/Canvas.c +++ b/shared-module/picogame/Canvas.c @@ -62,10 +62,18 @@ static __attribute__((noinline)) void put(picogame_canvas_obj_t *cv, int x, int // so it stays safe on Cortex-M0+ (RP2040), which faults on an unaligned 32-bit access - a StripDraw // view's rows into the render strip can start on an odd pixel. This is the per-frame path for // view.clear / Sky / HUD-bar / Fade fills, so the word-fill is worth it. +// Not in SRAM: every caller is in flash, and the long-branch veneer made it slower. static void fill565(uint16_t *p, int n, uint16_t color) { if (n <= 0) { return; } + // Short spans (triangle rows, wall runs) skip the word path's setup. + if (n <= 4) { + do { + *p++ = color; + } while (--n); + return; + } if (color == 0) { memset(p, 0, (size_t)n * 2); return; @@ -79,10 +87,7 @@ static void fill565(uint16_t *p, int n, uint16_t color) { #pragma GCC diagnostic ignored "-Wcast-align" uint32_t *w32 = (uint32_t *)p; // now 4-byte aligned #pragma GCC diagnostic pop - int nw = n >> 1; - for (int i = 0; i < nw; i++) { - w32[i] = w; - } + picogame_fill_words(w32, n >> 1, w); if (n & 1) { // trailing odd pixel p[n - 1] = color; } diff --git a/shared-module/picogame/__init__.c b/shared-module/picogame/__init__.c index 3845f9f967b..8988d8b8a9c 100644 --- a/shared-module/picogame/__init__.c +++ b/shared-module/picogame/__init__.c @@ -867,9 +867,7 @@ mp_obj_t picogame_blit_strip_layers( uint32_t *w32 = (uint32_t *)(buf + i); // now 4-byte aligned #pragma GCC diagnostic pop int nw = (npix - i) >> 1; - for (int k = 0; k < nw; k++) { - w32[k] = w; - } + picogame_fill_words(w32, nw, w); i += nw << 1; if (i < npix) { // odd trailing pixel buf[i] = background; diff --git a/shared-module/picogame/__init__.h b/shared-module/picogame/__init__.h index 39a0d7bc274..12868ae274a 100644 --- a/shared-module/picogame/__init__.h +++ b/shared-module/picogame/__init__.h @@ -27,13 +27,33 @@ #endif #endif +// Fill nw 32-bit words with w, four per iteration. w32 must be 4-byte aligned. +static inline void picogame_fill_words(uint32_t *w32, int nw, uint32_t w) { + for (int b = nw >> 2; b > 0; b--) { + w32[0] = w; + w32[1] = w; + w32[2] = w; + w32[3] = w; + w32 += 4; + } + if (nw & 2) { + w32[0] = w; + w32[1] = w; + w32 += 2; + } + if (nw & 1) { + w32[0] = w; + } +} + // Sample one texel as wire RGB565; false = transparent (skip). Shared by the sprite/canvas // blit paths so they inline one copy (see the blit contract: PAL8 indices must be < palette len). +// key is the transparent value, or -1 for an opaque bitmap (never matches). static inline bool src_pixel_s(int format, const uint8_t *data, const uint16_t *pal, - bool transp, uint16_t key, int idx, uint16_t *out) { + int32_t key, int idx, uint16_t *out) { if (format == PICOGAME_FMT_PAL8) { uint8_t i = data[idx]; - if (transp && i == (uint8_t)key) { + if ((int32_t)i == key) { return false; } *out = pal[i]; // indices must be < palette length (see blit contract) From e253861caf322ea03325e23ca8a102c41f796ca9 Mon Sep 17 00:00:00 2001 From: Vladimir Smitka Date: Mon, 21 Sep 2026 18:56:23 +0000 Subject: [PATCH 6/6] picogame: cheaper text, triangle setup, particles and bounds tests - canvas_text copies the atlas fields to locals, so they are not reloaded for every pixel. - fill_triangle does the int64 seed multiplies only when the triangle is clipped at the top. edge_slope is not inlined, so 6 divide call sites become 2. - Particles are rejected on Y first. The colour scale uses one divide instead of three (at most 1 LSB darker). - put() and the affine sampler use unsigned bounds checks. RP2040: 32 triangles 3367 -> 3075 us, text 294 -> 282 us. Output is unchanged. Co-authored-by: Petr Vavrin --- shared-module/picogame/Canvas.c | 32 +++++++++++++++++++++++------- shared-module/picogame/Particles.c | 23 +++++++++++++++------ 2 files changed, 42 insertions(+), 13 deletions(-) diff --git a/shared-module/picogame/Canvas.c b/shared-module/picogame/Canvas.c index 618cadb8a69..2e694e20cab 100644 --- a/shared-module/picogame/Canvas.c +++ b/shared-module/picogame/Canvas.c @@ -52,7 +52,8 @@ static void mark(picogame_canvas_obj_t *cv, int lx1, int ly1, int lx2, int ly2) // 8 calls/iteration). Inlining bloated them (circle was ~1.4 KB); a real call keeps // them small. Shapes aren't the hot path (the sprite/tilemap blits don't use put). static __attribute__((noinline)) void put(picogame_canvas_obj_t *cv, int x, int y, uint16_t c) { - if (x >= 0 && y >= 0 && x < cv->w && y < cv->h) { + // Unsigned compares also reject negative coordinates. + if ((unsigned)x < (unsigned)cv->w && (unsigned)y < (unsigned)cv->h) { cv->data[y * cv->w + x] = c; } } @@ -306,7 +307,8 @@ void picogame_canvas_line(picogame_canvas_obj_t *cv, int x0, int y0, int x1, int // Clamp a row span to the surface and word-fill it (the span-pass idiom shared by the filled // shapes; the per-pixel put() loops it replaced clipped and indexed every pixel). -static inline int64_t edge_slope(int32_t dx, int32_t dy) { +// Not inlined: each inlined copy carried both divides. +static __attribute__((noinline)) int64_t edge_slope(int32_t dx, int32_t dy) { if (dx >= -32768 && dx <= 32767) { return (int32_t)(dx << 16) / dy; } @@ -420,8 +422,15 @@ void picogame_canvas_fill_triangle(picogame_canvas_obj_t *cv, // top half: rows [Y0, Y1) walk edges A->C and A->B int ys = Y[0] < 0 ? 0 : Y[0]; int ye = (Y[1] - 1) < (h - 1) ? (Y[1] - 1) : (h - 1); - int64_t accAC = ((int64_t)X[0] << 16) + sAC * (ys - Y[0]); - int64_t acc2 = ((int64_t)X[0] << 16) + sAB * (ys - Y[0]); + // skip is 0 unless the triangle is clipped at the top, so the int64 multiplies + // usually do not run. + int skip = ys - Y[0]; + int64_t accAC = (int64_t)X[0] << 16; + int64_t acc2 = accAC; + if (skip) { + accAC += sAC * skip; + acc2 += sAB * skip; + } for (int y = ys; y <= ye; y++) { int xac = (int)(accAC >> 16); int xsh = (int)(acc2 >> 16); @@ -441,8 +450,12 @@ void picogame_canvas_fill_triangle(picogame_canvas_obj_t *cv, // bottom half: rows [Y1, Y2] walk edges A->C and B->C (a flat bottom degenerates to sBC=0) ys = Y[1] < 0 ? 0 : Y[1]; ye = Y[2] < (h - 1) ? Y[2] : (h - 1); - accAC = ((int64_t)X[0] << 16) + sAC * (ys - Y[0]); - acc2 = ((int64_t)X[1] << 16) + sBC * (ys - Y[1]); + accAC = ((int64_t)X[0] << 16) + sAC * (ys - Y[0]); // spans the whole top half: rarely 0 + acc2 = (int64_t)X[1] << 16; + skip = ys - Y[1]; + if (skip) { + acc2 += sBC * skip; + } for (int y = ys; y <= ye; y++) { int xac = (int)(accAC >> 16); int xsh = (int)(acc2 >> 16); @@ -581,6 +594,11 @@ void picogame_canvas_text(picogame_canvas_obj_t *cv, int x, int y, const char *t bool onebit = (sheet->bits_per_value == 1); // terminalio.FONT is 1-bpp; other fonts take the fallback const uint8_t *sdata = (const uint8_t *)sheet->data; int sstride_b = sheet->stride * 4; // atlas row stride in BYTES (stride counts uint32) + // Copy to locals: the uint16_t store may alias sheet->bitmask, so the fields + // would be reloaded for every pixel. + int sx_shift = sheet->x_shift; + size_t sx_mask = sheet->x_mask; + uint16_t sbitmask = sheet->bitmask; for (const uint8_t *p = (const uint8_t *)text; *p; p++) { uint8_t gi = fontio_builtinfont_get_glyph_index(f, *p); if (gi != 0xff) { // 0xff = no glyph -> blank advance @@ -596,7 +614,7 @@ void picogame_canvas_text(picogame_canvas_obj_t *cv, int x, int y, const char *t const uint8_t *srow = sdata + (size_t)sy * sstride_b; for (int gx = gx0; gx < gx1; gx++) { int sx = tx + gx; - if ((srow[sx >> sheet->x_shift] >> (sheet->x_mask - (sx & sheet->x_mask))) & sheet->bitmask) { + if ((srow[sx >> sx_shift] >> (sx_mask - ((size_t)sx & sx_mask))) & sbitmask) { drow[gx] = fg; } else if (has_bg) { drow[gx] = bg; diff --git a/shared-module/picogame/Particles.c b/shared-module/picogame/Particles.c index 87e6917f479..401c382b0ee 100644 --- a/shared-module/picogame/Particles.c +++ b/shared-module/picogame/Particles.c @@ -34,9 +34,11 @@ static void swap_remove(picogame_particles_obj_t *ps, int i) { // Dim a wire-order RGB565 color to num/den of its brightness (per channel). static inline uint16_t scale_wire565(uint16_t wire, int num, int den) { uint16_t c = (uint16_t)((wire >> 8) | (wire << 8)); // wire -> native - int r = ((c >> 11) & 0x1F) * num / den; - int g = ((c >> 5) & 0x3F) * num / den; - int b = (c & 0x1F) * num / den; + // One Q8 reciprocal for all three channels (0 <= q <= 256). At most 1 LSB darker. + int q = (num << 8) / den; + int r = (((c >> 11) & 0x1F) * q) >> 8; + int g = (((c >> 5) & 0x3F) * q) >> 8; + int b = ((c & 0x1F) * q) >> 8; uint16_t out = (uint16_t)((r << 11) | (g << 5) | b); return (uint16_t)((out >> 8) | (out << 8)); // native -> wire } @@ -139,9 +141,18 @@ void picogame_blit_particles( int sz = ps->size; int rx2 = x0 + region_w; int ry2 = strip_top + strip_h; - for (int i = 0; i < ps->count; i++) { - int sx = (ps->px[i] >> 8) + ox; - int sy = (ps->py[i] >> 8) + oy; + // Reject on Y first: a strip spans the full width, so X rarely rejects. + const int32_t *pxs = ps->px, *pys = ps->py; + int ylo = strip_top - sz - oy; // py >> 8 <= ylo -> entirely above + int yhi = ry2 - oy; // py >> 8 >= yhi -> entirely below + int count = ps->count; + for (int i = 0; i < count; i++) { + int sy8 = pys[i] >> 8; + if (sy8 <= ylo || sy8 >= yhi) { + continue; + } + int sx = (pxs[i] >> 8) + ox; + int sy = sy8 + oy; int xs = picogame_imax(sx, x0); int ys = picogame_imax(sy, strip_top); int xe = picogame_imin(sx + sz, rx2);