Skip to content

Commit e2d97af

Browse files
committed
displayio: step the TileGrid tile terms along the row
TileGrid.fill_area computed the tile index, the position inside the tile and the tile's place in the bitmap with nine integer divides per pixel. All divisors are constant for the call, and five of the divides depend only on the row. On cores without a divide instruction each one is a library call. - The row terms are computed once per row. - The x terms are stepped along the row. - The tile's position in the bitmap is recomputed only when the tile changes. - Bitmap.get_pixel computes values_per_byte only for bitmaps below 8 bits per pixel, so 8 and 16 bpp reads no longer divide. RP2040 (pajenicko_picopad), full-screen 320x240 repaint of an 8 bpp bitmap: 297 -> 203 ms. Flash +104 bytes. Output is unchanged: 12 cases (tiles of 16x16 and 12x10, partly off-screen, flip_x, flip_y, transpose, scale 2 and 3, transparency with overlap, 1/2/4/8/16 bpp, ColorConverter, dither) checksum identically through BusDisplay.fill_row.
1 parent ac3ad80 commit e2d97af

2 files changed

Lines changed: 38 additions & 8 deletions

File tree

‎shared-module/displayio/Bitmap.c‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -92,8 +92,8 @@ uint32_t common_hal_displayio_bitmap_get_pixel(displayio_bitmap_t *self, int16_t
9292
int32_t row_start = y * self->stride;
9393
uint32_t *row = self->data + row_start;
9494
uint8_t bytes_per_value = self->bits_per_value / 8;
95-
uint8_t values_per_byte = 8 / self->bits_per_value;
9695
if (bytes_per_value < 1) {
96+
uint8_t values_per_byte = 8 / self->bits_per_value;
9797
uint8_t bits = ((uint8_t *)row)[x >> self->x_shift];
9898
uint8_t bit_position = (values_per_byte - (x & self->x_mask) - 1) * self->bits_per_value;
9999
return (bits >> bit_position) & self->bitmask;

‎shared-module/displayio/TileGrid.c‎

Lines changed: 37 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -515,10 +515,38 @@ bool displayio_tilegrid_fill_area(displayio_tilegrid_t *self,
515515
displayio_input_pixel_t input_pixel;
516516
displayio_output_pixel_t output_pixel;
517517

518+
uint16_t scale = self->absolute_transform->scale;
519+
uint16_t tile_width = self->tile_width;
520+
uint16_t width_in_tiles = self->width_in_tiles;
521+
518522
for (input_pixel.y = start_y; input_pixel.y < end_y; ++input_pixel.y) {
519523
int16_t row_start = start + (input_pixel.y - start_y + y_shift) * y_stride; // in pixels
520-
int16_t local_y = input_pixel.y / self->absolute_transform->scale;
524+
int16_t local_y = input_pixel.y / scale;
525+
uint16_t y_tile_index = (local_y / self->tile_height + self->top_left_y) % self->height_in_tiles;
526+
uint16_t row_tile_location = y_tile_index * width_in_tiles;
527+
int16_t y_in_tile = local_y % self->tile_height;
528+
529+
// The x terms are stepped along the row instead of divided for every pixel.
530+
int16_t local_x = start_x / scale;
531+
uint16_t x_in_scale = start_x % scale;
532+
int16_t x_in_tile = local_x % tile_width;
533+
uint16_t x_tile_index = (local_x / tile_width + self->top_left_x) % width_in_tiles;
534+
int32_t cached_tile = -1;
535+
int16_t tile_base_x = 0;
536+
int16_t tile_base_y = 0;
537+
521538
for (input_pixel.x = start_x; input_pixel.x < end_x; ++input_pixel.x) {
539+
if (input_pixel.x != start_x && ++x_in_scale == scale) {
540+
x_in_scale = 0;
541+
local_x++;
542+
if (++x_in_tile == tile_width) {
543+
x_in_tile = 0;
544+
if (++x_tile_index == width_in_tiles) {
545+
x_tile_index = 0;
546+
}
547+
}
548+
}
549+
522550
// Compute the destination pixel in the buffer and mask based on the transformations.
523551
int16_t offset = row_start + (input_pixel.x - start_x + x_shift) * x_stride; // in pixels
524552

@@ -531,18 +559,20 @@ bool displayio_tilegrid_fill_area(displayio_tilegrid_t *self,
531559
if ((mask[offset / 32] & (1 << (offset % 32))) != 0) {
532560
continue;
533561
}
534-
int16_t local_x = input_pixel.x / self->absolute_transform->scale;
535-
uint16_t x_tile_index = (local_x / self->tile_width + self->top_left_x) % self->width_in_tiles;
536-
uint16_t y_tile_index = (local_y / self->tile_height + self->top_left_y) % self->height_in_tiles;
537-
uint16_t tile_location = y_tile_index * self->width_in_tiles + x_tile_index;
562+
uint16_t tile_location = row_tile_location + x_tile_index;
538563

539564
if (self->tiles_in_bitmap > 255) {
540565
input_pixel.tile = ((uint16_t *)tiles)[tile_location];
541566
} else {
542567
input_pixel.tile = ((uint8_t *)tiles)[tile_location];
543568
}
544-
input_pixel.tile_x = (input_pixel.tile % self->bitmap_width_in_tiles) * self->tile_width + local_x % self->tile_width;
545-
input_pixel.tile_y = (input_pixel.tile / self->bitmap_width_in_tiles) * self->tile_height + local_y % self->tile_height;
569+
if (input_pixel.tile != cached_tile) {
570+
cached_tile = input_pixel.tile;
571+
tile_base_x = (input_pixel.tile % self->bitmap_width_in_tiles) * tile_width;
572+
tile_base_y = (input_pixel.tile / self->bitmap_width_in_tiles) * self->tile_height;
573+
}
574+
input_pixel.tile_x = tile_base_x + x_in_tile;
575+
input_pixel.tile_y = tile_base_y + y_in_tile;
546576

547577
output_pixel.pixel = 0;
548578
input_pixel.pixel = 0;

0 commit comments

Comments
 (0)