Add fast path in Banshee overlay render; if overlay is unfiltered and

unscaled then just write straight to destination buffer.
This commit is contained in:
SarahW 2020-09-28 20:02:38 +01:00
commit f6406c77e5

View file

@ -1535,7 +1535,7 @@ void banshee_hwcursor_draw(svga_t *svga, int displine)
int g = (data >> 5) & 0x3f; \ int g = (data >> 5) & 0x3f; \
int b = data >> 11; \ int b = data >> 11; \
\ \
banshee->overlay_buffer[buf][wp++] = (r << 3) | (g << 10) | (b << 19); \ buf[wp++] = (r << 3) | (g << 10) | (b << 19); \
src += 2; \ src += 2; \
} \ } \
} while (0) } while (0)
@ -1553,7 +1553,7 @@ void banshee_hwcursor_draw(svga_t *svga, int displine)
int g = (data >> 5) & 0x3f; \ int g = (data >> 5) & 0x3f; \
int b = data >> 11; \ int b = data >> 11; \
\ \
banshee->overlay_buffer[buf][wp++] = (r << 3) | (g << 10) | (b << 19); \ buf[wp++] = (r << 3) | (g << 10) | (b << 19); \
} \ } \
} while (0) } while (0)
@ -1586,7 +1586,7 @@ void banshee_hwcursor_draw(svga_t *svga, int displine)
CLAMP(g); \ CLAMP(g); \
b = y1 + dB; \ b = y1 + dB; \
CLAMP(b); \ CLAMP(b); \
banshee->overlay_buffer[buf][wp++] = r | (g << 8) | (b << 16); \ buf[wp++] = r | (g << 8) | (b << 16); \
\ \
r = y2 + dR; \ r = y2 + dR; \
CLAMP(r); \ CLAMP(r); \
@ -1594,7 +1594,7 @@ void banshee_hwcursor_draw(svga_t *svga, int displine)
CLAMP(g); \ CLAMP(g); \
b = y2 + dB; \ b = y2 + dB; \
CLAMP(b); \ CLAMP(b); \
banshee->overlay_buffer[buf][wp++] = r | (g << 8) | (b << 16); \ buf[wp++] = r | (g << 8) | (b << 16); \
} \ } \
} while (0) } while (0)
@ -1627,7 +1627,7 @@ void banshee_hwcursor_draw(svga_t *svga, int displine)
CLAMP(g); \ CLAMP(g); \
b = y1 + dB; \ b = y1 + dB; \
CLAMP(b); \ CLAMP(b); \
banshee->overlay_buffer[buf][wp++] = r | (g << 8) | (b << 16); \ buf[wp++] = r | (g << 8) | (b << 16); \
\ \
r = y2 + dR; \ r = y2 + dR; \
CLAMP(r); \ CLAMP(r); \
@ -1635,7 +1635,7 @@ void banshee_hwcursor_draw(svga_t *svga, int displine)
CLAMP(g); \ CLAMP(g); \
b = y2 + dB; \ b = y2 + dB; \
CLAMP(b); \ CLAMP(b); \
banshee->overlay_buffer[buf][wp++] = r | (g << 8) | (b << 16); \ buf[wp++] = r | (g << 8) | (b << 16); \
} \ } \
} while (0) } while (0)
@ -1691,13 +1691,21 @@ static void banshee_overlay_draw(svga_t *svga, int displine)
// fatal("overlay out of range!\n"); // fatal("overlay out of range!\n");
p = &((uint32_t *)buffer32->line[displine])[svga->overlay_latch.x + 32]; p = &((uint32_t *)buffer32->line[displine])[svga->overlay_latch.x + 32];
OVERLAY_SAMPLE(0); if ((banshee->vidProcCfg & VIDPROCCFG_FILTER_MODE_MASK) != VIDPROCCFG_FILTER_MODE_BILINEAR &&
!(banshee->vidProcCfg & VIDPROCCFG_H_SCALE_ENABLE))
{
/*No scaling or filtering required, just write straight to output buffer*/
OVERLAY_SAMPLE(p);
}
else
{
OVERLAY_SAMPLE(banshee->overlay_buffer[0]);
switch (banshee->vidProcCfg & VIDPROCCFG_FILTER_MODE_MASK) switch (banshee->vidProcCfg & VIDPROCCFG_FILTER_MODE_MASK)
{ {
case VIDPROCCFG_FILTER_MODE_BILINEAR: case VIDPROCCFG_FILTER_MODE_BILINEAR:
src = &svga->vram[src_addr2 & svga->vram_mask]; src = &svga->vram[src_addr2 & svga->vram_mask];
OVERLAY_SAMPLE(1); OVERLAY_SAMPLE(banshee->overlay_buffer[1]);
if (banshee->vidProcCfg & VIDPROCCFG_H_SCALE_ENABLE) if (banshee->vidProcCfg & VIDPROCCFG_H_SCALE_ENABLE)
{ {
for (x = 0; x < svga->overlay_latch.xsize; x++) for (x = 0; x < svga->overlay_latch.xsize; x++)
@ -1765,6 +1773,7 @@ static void banshee_overlay_draw(svga_t *svga, int displine)
} }
break; break;
} }
}
if (banshee->vidProcCfg & VIDPROCCFG_V_SCALE_ENABLE) if (banshee->vidProcCfg & VIDPROCCFG_V_SCALE_ENABLE)
voodoo->overlay.src_y += voodoo->overlay.vidOverlayDvdy; voodoo->overlay.src_y += voodoo->overlay.vidOverlayDvdy;