pico9918-core 1.3.0
TMS9918A / F18A video display processor emulation in C99
Loading...
Searching...
No Matches
pico9918.c
Go to the documentation of this file.
1/**
2 * \file
3 * \brief pico9918-core - Core interface
4 *
5 * Copyright (c) 2021 Troy Schrapel
6 *
7 * This code is licensed under the MIT license
8 *
9 * https://github.com/visrealm/pico9918-core
10 *
11 */
12
13
14#include "impl/pico9918_priv.h"
15/* pico9918_gpu_service: the register writes below are where a GPU program is armed,
16 and a host that handed the library the GPU wants it run there. */
18#include "overlay/splash.h" /* pico9918_reset re-arms the splash, as a host reset does */
19
20#include <string.h>
21
22
23/* The mode emitters stay out of pico9918_scan_line. That function is well past the Thumb-1 b.n
24 reach, so anything added to it relaxes branches and shuffles registers inside whichever emitters
25 are inlined there. Keeping them out of line costs nothing on either board. */
26#define EMITTER_NOINLINE PICO9918_NOINLINE
27
28#ifdef PICO_BUILD
29/* The DMA channel numbers are compile-time macros in the pico platform header, not
30 variables here: an extern channel number costs three instructions at every access.
31 The copy channel's two configs do live here. */
32PICO9918_COPY_STATE()
33#else
34/* Off-target the fills and the copy are plain structs. */
35pico9918_fill32_t pico9918_fill_border = {NULL, 0};
36pico9918_fill32_t pico9918_fill_masks = {NULL, 0};
37pico9918_fill32_t pico9918_fill_line = {NULL, 0};
38pico9918_copy32_t pico9918_copy = {NULL, NULL, 0};
39#endif
40
41/* not .scratch_x: that is where the fill writes */
42PICO9918_SECTION_SCRATCH_Y(buffer) static uint32_t bg;
43
44/* Where a scanline is arbitrated: the fill, the sprites, the bitmap layer and the composite all land
45 here. Its own bank, so the fill writes it from `bg`'s and a caller reads it against striped SRAM. */
46static PICO9918_SECTION_SCRATCH_X(buffer) uint8_t __aligned(4) scanlineBuffer[SCANLINE_BUFFER_BYTES];
47
48
49/* Declared on the impl surface, with the TEXT80_WIDE_ROW macro and the inline readers that go
50 with it, so the frame module does not call across the TU boundary for them every line. */
51pico9918_mode_t pico9918_cached_mode = TMS_MODE_GRAPHICS_I;
52
53/* Configured and claimed once, before the host brings up anything that shares the
54 DMA. Defined below, next to the tables it fills. */
55void initLookups(void);
56
57#if PICO9918_SINGLE_INSTANCE
58
59// VRAM is intentionally never zeroed at boot (pico9918_reset() below
60// leaves it alone - matches real TMS9918 hardware, whose VRAM content is
61// undefined at power-on); every other field is explicitly written by
62// pico9918_reset()/vdpRegisterReset() before anything reads it - so this
63// doesn't need the crt0 .bss zero-fill, and skipping it saves boot time.
64//
65// The 256-byte alignment is required: the GPU guards vram.bytes[0x8000] and
66// the palette with MPU regions that are whole 256-byte pages, and neither
67// range may cross a page boundary. A 256-aligned instance is what fixes where
68// inside its page each range lands.
69static pico9918_t __aligned(256) PICO9918_UNINITIALIZED(tms9918Inst);
70
71/* const so the instance address is a link-time constant rather than a pointer the
72 emitters have to load and keep live: every field offset then folds into its own
73 literal, which is what makes vram's offset within the struct cost nothing. */
74pico9918_t* const tms9918 = &tms9918Inst;
75
76/** \brief initialize the TMS9918 library in single-instance mode */
78void __time_critical_func(pico9918_init)(void)
79{
80 tms9918->vdpBase = PICO9918_BASE_TMS9918;
81#if PICO9918_BUILD_RUNTIME_CHIP
83#endif
84 initLookups();
86}
87
88/** \brief see the header. The same pointer every implicit-instance entry point uses. */
89PICO9918_DLLEXPORT pico9918_t* pico9918_instance(void)
90{
91 return tms9918;
92}
93
94#else
95
96#include <stdlib.h>
97
98/** \brief create a new TMS9918 */
99PICO9918_DLLEXPORT pico9918_t* pico9918_new(void)
100{
101 pico9918_t* tms9918 = (pico9918_t*)calloc(1, sizeof(pico9918_t));
102 if (tms9918 != NULL)
103 {
104 tms9918->vdpBase = PICO9918_BASE_TMS9918; /* see pico9918_init */
105#if PICO9918_BUILD_RUNTIME_CHIP
107#endif
108 initLookups();
109 pico9918_reset(tms9918);
110 }
111
112 return tms9918;
113}
114
115#endif
116
117/** \brief see the header. What a versioned save/restore copies from the instance base. */
119{
120 return sizeof(pico9918_t);
121}
122
123/** \brief see the header. The latch itself, not the personality that could set it. */
125{
126 return PICO9918_UNLOCKED(tms9918);
127}
128
129/* host /INT hook - see the header for the contract; only the storage differs by build */
130#if PICO9918_SINGLE_INSTANCE
131static struct
132{
134 void* userdata;
135} interruptCb;
136#define INTERRUPT_CB interruptCb
137#else
138#define INTERRUPT_CB tms9918->interrupt
139#endif
140
142 void* userdata)
143{
144 INTERRUPT_CB.fn = cb;
145 INTERRUPT_CB.userdata = userdata;
146}
147
148/** \brief see impl. What the desktop PICO9918_HOST_SET_INT expands to. */
150{
151 if (INTERRUPT_CB.fn) INTERRUPT_CB.fn(tms9918, active, INTERRUPT_CB.userdata);
152}
153
154/* Here rather than beside the macros: pico9918.h reaches neither PICO9918_STATIC_ASSERT
155 nor the private map type. */
156PICO9918_STATIC_ASSERT(offsetof(pico9918_mem_map_t, pram) == PICO9918_MAP_PRAM,
157 "PICO9918_MAP_PRAM does not match the memory map");
158PICO9918_STATIC_ASSERT(offsetof(pico9918_mem_map_t, registers) == PICO9918_MAP_REGISTERS,
159 "PICO9918_MAP_REGISTERS does not match the memory map");
160PICO9918_STATIC_ASSERT(offsetof(pico9918_mem_map_t, scanline) == PICO9918_MAP_SCANLINE,
161 "PICO9918_MAP_SCANLINE does not match the memory map");
162PICO9918_STATIC_ASSERT(offsetof(pico9918_mem_map_t, status) == PICO9918_MAP_STATUS,
163 "PICO9918_MAP_STATUS does not match the memory map");
164
165
166static const pico9918_mode_t r1Modes[] = {TMS_MODE_GRAPHICS_I, TMS_MODE_MULTICOLOR, TMS_MODE_TEXT,
167 TMS_MODE_GRAPHICS_I};
168
169static inline pico9918_mode_t tmsMode(pico9918_t* tms9918)
170{
171 /* The pre-A part leaves M3 undecoded, so the bit selects nothing and M1/M2 still do. */
172 if ((TMS_REGISTER(tms9918, TMS_REG_0) & TMS_R0_MODE_GRAPHICS_II) && PICO9918_GM2(tms9918))
173 {
174 return TMS_MODE_GRAPHICS_II;
175 }
176 /* An F18A honours M4 while still locked, so the test is the personality, not the lock. */
177 else if (PICO9918_M4(tms9918))
178 {
179 return TMS_MODE_TEXT80;
180 }
181 else
182 {
183 return r1Modes[(TMS_REGISTER(tms9918, TMS_REG_1) & (TMS_R1_MODE_MULTICOLOR | TMS_R1_MODE_TEXT)) >> 3];
184 }
185}
186
187/** \brief sprite size (8 or 16) */
188static inline uint8_t tmsSpriteSize(pico9918_t* tms9918)
189{
190 return TMS_REGISTER(tms9918, TMS_REG_1) & TMS_R1_SPRITE_16 ? 16 : 8;
191}
192
193/** \brief sprite size (0 = 1x, 1 = 2x) */
194static inline bool tmsSpriteMag(pico9918_t* tms9918)
195{
196 return TMS_REGISTER(tms9918, TMS_REG_1) & TMS_R1_SPRITE_MAG2;
197}
198
199/** \brief name table base address */
200static inline uint16_t tmsNameTableAddr(pico9918_t* tms9918)
201{
202 return (TMS_REGISTER(tms9918, TMS_REG_NAME_TABLE) & 0x0f) << 10;
203}
204
205/** \brief name table base address */
206static inline uint16_t tmsNameTable2Addr(pico9918_t* tms9918)
207{
208 return (TMS_REGISTER(tms9918, PICO9918_REG_NAME_TABLE2) & 0x0f) << 10;
209}
210
211/** \brief color table base address */
212static inline uint16_t tmsColorTableAddr(pico9918_t* tms9918)
213{
214 const uint8_t mask = (pico9918_cached_mode == TMS_MODE_GRAPHICS_II) ? 0x80 : 0xff;
215
216 return (TMS_REGISTER(tms9918, TMS_REG_COLOR_TABLE) & mask) << 6;
217}
218
219/** \brief color table base address */
220static inline uint16_t tmsColorTable2Addr(pico9918_t* tms9918)
221{
222 const uint8_t mask = (pico9918_cached_mode == TMS_MODE_GRAPHICS_II) ? 0x80 : 0xff;
223
224 return (TMS_REGISTER(tms9918, PICO9918_REG_COLOR_TABLE2) & mask) << 6;
225}
226
227/** \brief pattern table base address */
228static inline uint16_t tmsPatternTableAddr(pico9918_t* tms9918)
229{
230 const uint8_t mask = (pico9918_cached_mode == TMS_MODE_GRAPHICS_II) ? 0x04 : 0x07;
231
232 return (TMS_REGISTER(tms9918, TMS_REG_PATTERN_TABLE) & mask) << 11;
233}
234
235/** \brief sprite attribute table base address */
236static inline uint16_t tmsSpriteAttrTableAddr(pico9918_t* tms9918)
237{
238 return (TMS_REGISTER(tms9918, TMS_REG_SPRITE_ATTR_TABLE) & 0x7f) << 7;
239}
240
241/** \brief sprite pattern table base address */
242static inline uint16_t tmsSpritePatternTableAddr(pico9918_t* tms9918)
243{
244 return (TMS_REGISTER(tms9918, TMS_REG_SPRITE_PATT_TABLE) & 0x07) << 11;
245}
246
247/** \brief background color */
248static inline pico9918_color_t tmsMainBgColor(pico9918_t* tms9918)
249{
250 return TMS_REGISTER(tms9918, TMS_REG_FG_BG_COLOR) & 0x0f;
251}
252
253/** \brief foreground color */
254static inline pico9918_color_t tmsMainFgColor(pico9918_t* tms9918)
255{
256 const pico9918_color_t c = (pico9918_color_t)(TMS_REGISTER(tms9918, TMS_REG_FG_BG_COLOR) >> 4);
257 return c == TMS_TRANSPARENT ? tmsMainBgColor(tms9918) : c;
258}
259
260/** \brief foreground color */
261static inline pico9918_color_t tmsFgColor(pico9918_t* tms9918, uint8_t colorByte)
262{
263 const pico9918_color_t c = (pico9918_color_t)(colorByte >> 4);
264 return c == TMS_TRANSPARENT ? tmsMainBgColor(tms9918) : c;
265}
266
267/** \brief background color */
268static inline pico9918_color_t tmsBgColor(pico9918_t* tms9918, uint8_t colorByte)
269{
270 const pico9918_color_t c = (pico9918_color_t)(colorByte & 0x0f);
271 return c == TMS_TRANSPARENT ? tmsMainBgColor(tms9918) : c;
272}
273
274
275// default palette 0xARGB
276static const uint16_t defaultPalette[] = {
277 //-- Palette 0, default TMS9918A palette
278 0x0000, 0xF000, 0xF2C3, 0xF5D6, 0xF54F, 0xF76F, 0xFD54, 0xF4EF, 0xFF54, 0xFF76, 0xFDC3, 0xFED6, 0xF2B2,
279 0xFC5C, 0xFCCC, 0xFFFF,
280 //-- Palette 1, ECM1 (0 index is always 000) version of palette 0
281 0x0000, 0xF2C3, 0xF000, 0xF54F, 0xF000, 0xFD54, 0xF000, 0xF4EF, 0xF000, 0xFCCC, 0xF000, 0xFDC3, 0xF000,
282 0xFC5C, 0xF000, 0xFFFF,
283 //-- Palette 2, CGA colors
284 0x0000, 0xF00A, 0xF0A0, 0xF0AA, 0xFA00, 0xFA0A, 0xFA50, 0xFAAA, 0xF555, 0xF55F, 0xF5F5, 0xF5FF, 0xFF55,
285 0xFF5F, 0xFFF5, 0xFFFF,
286 //-- Palette 3, ECM1 (0 index is always 000) version of palette 2
287 0x0000, 0xF555, 0xF000, 0xF00A, 0xF000, 0xF0A0, 0xF000, 0xF0AA, 0xF000, 0xFA00, 0xF000, 0xFA0A, 0xF000,
288 0xFA50, 0xF000, 0xFFFF};
289
290static PICO9918_NOINLINE void vdpRegisterReset(pico9918_t* tms9918)
291{
292 tms9918->isUnlocked = false;
293 tms9918->restart = 0;
294 tms9918->unlockCount = 0;
295 tms9918->lockedMask = 0x07;
296 memset(&TMS_REGISTER(tms9918, TMS_REG_0), 0, TMS_REGISTERS);
297 TMS_REGISTER(tms9918, TMS_REG_1) = 0x40;
298 TMS_REGISTER(tms9918, TMS_REG_3) = 0x10;
299 TMS_REGISTER(tms9918, TMS_REG_4) = 0x01;
300 TMS_REGISTER(tms9918, TMS_REG_5) = 0x0A;
301 TMS_REGISTER(tms9918, TMS_REG_6) = 0x02;
302 TMS_REGISTER(tms9918, TMS_REG_7) = 0xF2;
303 TMS_REGISTER(tms9918, PICO9918_REG_MAX_SCAN_SPRITES) = PICO9918_SCAN_SPRITE_LIMIT(tms9918);
304 TMS_REGISTER(tms9918, PICO9918_REG_VRAM_INC) = 1; // vram address increment register
305 TMS_REGISTER(tms9918, PICO9918_REG_MAX_SPRITES) = MAX_SPRITES; // Sprites to process
306 TMS_REGISTER(tms9918, PICO9918_REG_GPU_PC_MSB) = 0x40;
307}
308
309
310#if PICO9918_BUILD_RUNTIME_CHIP
311
312/** \brief the feature bits a personality answers to - the ladder, in one place */
313static uint8_t chipFeatures(pico9918_chip_t chip)
314{
315 switch (chip)
316 {
318 return PICO9918_FEAT_UNLOCK | PICO9918_FEAT_CONFIG | PICO9918_FEAT_OVERLAY |
319 PICO9918_FEAT_BITMAP | PICO9918_FEAT_WIDE_T80 | PICO9918_FEAT_GPU_RAM;
321 return PICO9918_FEAT_UNLOCK | PICO9918_FEAT_CONFIG | PICO9918_FEAT_OVERLAY |
322 PICO9918_FEAT_BITMAP | PICO9918_FEAT_GPU_RAM;
323 case PICO9918_CHIP_F18A: return PICO9918_FEAT_UNLOCK | PICO9918_FEAT_BITMAP | PICO9918_FEAT_WIDE_T80;
324 case PICO9918_CHIP_TMS9918A: return PICO9918_FEAT_BITMAP | PICO9918_FEAT_VRAM_4K;
325 default: return PICO9918_FEAT_VRAM_4K;
326 }
327}
328
329/** \brief select which chip this instance answers as */
331{
332 /* unsigned, so a value below the bottom of the ladder clamps here too rather than being stored */
333 if ((unsigned)chip > (unsigned)PICO9918_CHIP_MAX)
334 {
335 chip = PICO9918_CHIP_MAX;
336 }
337
338 tms9918->chip = (uint8_t)chip;
339 tms9918->features = chipFeatures(chip);
340
341 /* only a personality with no settings block takes its limit here - the block owns VR30 */
342 if (!PICO9918_HAS(tms9918, PICO9918_FEAT_CONFIG))
343 {
344 TMS_REGISTER(tms9918, PICO9918_REG_MAX_SCAN_SPRITES) = PICO9918_SCAN_SPRITE_LIMIT(tms9918);
345 }
346
347#if !PICO9918_NO_SPLASH
349#endif
350
351 /* The wide line is a different palette layout, so the tier is a palette change */
352 tms9918->palDirty = 1;
353
354 /* the new personality either seeds its effects from the block or lets go of them */
355 tms9918->configDirty = true;
356 tms9918->configVdpDirty = true;
357
358 if (!PICO9918_HAS(tms9918, PICO9918_FEAT_UNLOCK))
359 {
360 tms9918->isUnlocked = false;
361 tms9918->unlockCount = 0;
362 tms9918->lockedMask = 0x07;
363 TMS_REGISTER(tms9918, PICO9918_REG_GPU_CONTROL) = 0;
364 }
365
366 TMS_STATUS(tms9918, PICO9918_SR_IDENT) = PICO9918_SR1_ID(tms9918);
367}
368
369/** \brief which chip this instance answers as */
374
375#endif // PICO9918_BUILD_RUNTIME_CHIP
376
377/** \brief reset the new TMS9918 */
379{
380 tms9918->regWriteStage0Value = 0;
381 tms9918->currentAddress = 0;
382 tms9918->gpuAddress = 0xFFFF; // "Odd" don't start value
383 tms9918->regWriteStage = 0;
384
385 tms9918->palWriteStage = 0;
386 tms9918->palWriteStage0Value = 0;
387 tms9918->flash = 0;
388 memset(&TMS_STATUS(tms9918, PICO9918_SR_STATUS), 0, TMS_STATUS_REGISTERS);
390 TMS_STATUS(tms9918, PICO9918_SR_IDENT) = PICO9918_SR1_ID(tms9918);
391 TMS_STATUS(tms9918, PICO9918_SR_VERSION) = 0x1A; // Version
392 tms9918->readAheadBuffer = 0;
393
394 vdpRegisterReset(tms9918);
395 TMS_REGISTER(tms9918, TMS_REG_1) = 0x00; // turn display off
396 TMS_REGISTER(tms9918, TMS_REG_7) = 0x00;
397 pico9918_cached_mode = TMS_MODE_GRAPHICS_I;
398
399 // set up default palettes (arm is little-endian, tms9900 is big-endian)
400 for (int i = 0; i < sizeof(defaultPalette) / sizeof(uint16_t); ++i)
401 {
402 tms9918->vram.map.pram[i] = __builtin_bswap16(defaultPalette[i]);
403 }
404
405 /* row-30 progressive has no border line, so nothing else invalidates the derived LUT */
406 tms9918->palDirty = 1;
407
408 pico9918_frame_reset_count_impl(PICO9918_INST_ONLY);
410
411 /* ram intentionally left in unknown state */
412}
413
414
415/**
416 * \brief destroy a TMS9918
417 *
418 * tms9918: tms9918 object to destroy / clean up
419 */
421{
422#if !PICO9918_SINGLE_INSTANCE
423 free(tms9918);
424 tms9918 = NULL;
425#endif
426}
427
428/**
429 * \brief write an address (mode = 1) to the tms9918
430 *
431 * data: the data (DB0 -> DB7) to send
432 */
433PICO9918_DLLEXPORT void __time_critical_func(pico9918_write_addr)(PICO9918_INST_ARG uint8_t data)
434{
436}
437
438/** \brief read from the status register */
443
444/** \brief read from the status register without resetting it */
449
450/**
451 * \brief write data (mode = 0) to the tms9918
452 *
453 * data: the data (DB0 -> DB7) to send
454 */
455PICO9918_DLLEXPORT void __time_critical_func(pico9918_write_data)(PICO9918_INST_ARG uint8_t data)
456{
458}
459
460
461/** \brief read data (mode = 0) from the tms9918 */
466
467/** \brief read data (mode = 0) from the tms9918 */
472
473/** \brief return true if both INT status and INT control set */
478
479/** \brief raise the INT status flag */
484
485/** \brief set status flag */
487void __time_critical_func(pico9918_set_status)(PICO9918_INST_ARG uint8_t status)
488{
490}
491
492static const uint32_t zeroWord = 0;
493
494/* Sprites and the bitmap layer are always on the 256-pixel grid, so their masks are one
495 bit per grid pixel whatever the mode. A tile layer's own line is not: 80 columns at eight bits a
496 pixel are 512 pixels and want a bit for each, which is what lets a layer be selected per pixel.
497 The two are different lengths *and* different units, and the same function must not take both. */
498typedef uint32_t BitMask[9];
499typedef uint32_t TileMask[SCANLINE_MASK_WORDS];
500
501/* A mask walked one `<<= 1` a pixel has to be unsigned: shifting a negative signed value left is
502 undefined, and a compiler may then fold the sign test away. `-fsanitize=shift-base` catches it. */
503#define MASK_NEXT_PIXEL 0x80000000u
504
505
506/* one object, so the per-scanline clear is a single transfer and the layout cannot drift */
507static PICO9918_SECTION_SCRATCH_X(lookup) struct
508{
509 BitMask rowBits; /* pixel mask */
510 BitMask rowTransparentSpriteBits; /* transparent sprite pixels */
511 BitMask rowSpriteBits; /* collision mask */
512} __aligned(4) rowMasks;
513
514/** \brief Test and update the sprite collision mask. */
515static inline uint32_t tmsTestCollisionMask(const uint32_t xPos, const uint32_t spritePixels,
516 const uint32_t spriteWidth)
517{
518 uint32_t rowSpriteBitsWord = xPos >> 5;
519 uint32_t rowSpriteBitsWordBit = xPos & 0x1f;
520
521 uint32_t validPixels =
522 (~rowMasks.rowSpriteBits[rowSpriteBitsWord]) & (spritePixels >> rowSpriteBitsWordBit);
523 rowMasks.rowSpriteBits[rowSpriteBitsWord] |= validPixels;
524 validPixels <<= rowSpriteBitsWordBit;
525
526 rowSpriteBitsWordBit = 32 - rowSpriteBitsWordBit;
527 if (rowSpriteBitsWordBit < spriteWidth)
528 {
529 uint32_t right = (~rowMasks.rowSpriteBits[++rowSpriteBitsWord]) & (spritePixels << rowSpriteBitsWordBit);
530 rowMasks.rowSpriteBits[rowSpriteBitsWord] |= right;
531 validPixels |= (right >> rowSpriteBitsWordBit);
532 }
533
534 return validPixels;
535}
536
537
538/** \brief set the transparent sprite mask. */
539static inline void tmsSetTransparentSpriteMask(const uint32_t xPos, const uint32_t spritePixels,
540 const uint32_t spriteWidth)
541{
542 uint32_t rowSpriteBitsWord = xPos >> 5;
543 uint32_t rowSpriteBitsWordBit = xPos & 0x1f;
544
545 rowMasks.rowTransparentSpriteBits[rowSpriteBitsWord] |= spritePixels >> rowSpriteBitsWordBit;
546
547 rowSpriteBitsWordBit = 32 - rowSpriteBitsWordBit;
548 if (rowSpriteBitsWordBit < spriteWidth)
549 {
550 rowMasks.rowTransparentSpriteBits[rowSpriteBitsWord + 1] |= spritePixels << rowSpriteBitsWordBit;
551 }
552}
553
554
555/** \brief Clear the row pixels bit mask. */
556static inline void tmsClearRowBitsMask(const uint32_t xPos, const uint32_t tilePixels,
557 const uint32_t tileWidth, BitMask rowBitsMask)
558{
559 uint32_t rowBitsWord = xPos >> 5;
560 uint32_t rowBitsWordBit = xPos & 0x1f;
561
562 uint32_t validPixels = tilePixels >> rowBitsWordBit;
563 rowBitsMask[rowBitsWord] &= ~validPixels;
564
565 rowBitsWordBit = 32 - rowBitsWordBit;
566 if (rowBitsWordBit < tileWidth)
567 {
568 ++rowBitsWord;
569 uint32_t right = (tilePixels << rowBitsWordBit);
570 rowBitsMask[rowBitsWord] &= ~right;
571 }
572}
573
574/** \brief Update the row pixels bit mask (aligned - no word boundary crossing). */
575static inline void tmsUpdateRowBitsMaskAligned(const uint32_t xPos, const uint32_t tilePixels,
576 BitMask rowBitsMask)
577{
578 rowBitsMask[xPos >> 5] |= tilePixels >> (xPos & 0x1f);
579}
580
581/** \brief Test against the row pixels bit mask (aligned - no word boundary crossing). */
582static inline uint32_t tmsTestRowBitsMaskAligned(const uint32_t xPos, const uint32_t tilePixels,
583 const BitMask rowBitsMask)
584{
585 return tilePixels & ~(rowBitsMask[xPos >> 5] << (xPos & 0x1f));
586}
587
588/* Out of line on purpose: two calls a line, and inlined it puts a second copy of the whole
589 straight-line word copy into the scanline body, which costs the board more than the call. */
590static PICO9918_NOINLINE void tmsCopyAlignMask(TileMask dstMask, const TileMask srcMask, int pixelShift)
591{
592 if (pixelShift == 0)
593 {
594 /* straight-line, not a loop: -O3 rewrites the loop form into a bootrom memcpy call */
595 dstMask[0] = srcMask[0];
596 dstMask[1] = srcMask[1];
597 dstMask[2] = srcMask[2];
598 dstMask[3] = srcMask[3];
599 dstMask[4] = srcMask[4];
600 dstMask[5] = srcMask[5];
601 dstMask[6] = srcMask[6];
602 dstMask[7] = srcMask[7];
603 dstMask[8] = srcMask[8];
604#if SCANLINE_MASK_WORDS > 9
605 dstMask[9] = srcMask[9];
606 dstMask[10] = srcMask[10];
607 dstMask[11] = srcMask[11];
608 dstMask[12] = srcMask[12];
609 dstMask[13] = srcMask[13];
610 dstMask[14] = srcMask[14];
611 dstMask[15] = srcMask[15];
612 dstMask[16] = srcMask[16];
613#endif
614 return;
615 }
616
617 if (pixelShift > 0)
618 {
619 // Right shift - carry flows left to right (low to high index)
620 uint32_t carry = 0;
621 for (int i = 0; i < SCANLINE_MASK_WORDS; i++) // LOW to HIGH
622 {
623 uint32_t word = srcMask[i];
624 dstMask[i] = (word >> pixelShift) | carry;
625 carry = word << (32 - pixelShift);
626 }
627 }
628 else
629 {
630 // Left shift - carry flows right to left (high to low index)
631 pixelShift = -pixelShift;
632 uint32_t carry = 0;
633 for (int i = SCANLINE_MASK_WORDS - 1; i >= 0; i--) // HIGH to LOW
634 {
635 uint32_t word = srcMask[i];
636 dstMask[i] = (word << pixelShift) | carry;
637 carry = word >> (32 - pixelShift);
638 }
639 }
640}
641
642/* The tile2 mask's trip home. Unshifted it is a copy out and straight back, and nothing
643 between the two writes either mask - the row emitters touch layerSelectionMask only on a
644 tile2 pass - so there is nothing to bring home. Under any scroll the round trip nets a
645 shift of -t2Scroll and loses the bits it carries off the end, so it must run.
646
647 The test lives here rather than at the call site because the scanline body is at the size
648 where one more branch in it re-plans the whole function. */
649static PICO9918_NOINLINE void tmsRestoreAlignMask(TileMask dstMask, const TileMask srcMask,
650 int pixelShift, int otherShift)
651{
652 if (pixelShift | otherShift) tmsCopyAlignMask(dstMask, srcMask, pixelShift);
653}
654
655
656/** \brief Test and update the row pixels bit mask. */
657static inline uint32_t tmsTestRowBitsMask(const uint32_t xPos, const uint32_t tilePixels,
658 const uint32_t tileWidth, const bool update, const bool test,
659 const bool testColl)
660{
661 uint32_t rowBitsWord = xPos >> 5;
662 uint32_t rowBitsWordBit = xPos & 0x1f;
663
664 uint32_t validPixels = tilePixels >> rowBitsWordBit;
665 if (testColl) validPixels &= ~rowMasks.rowSpriteBits[rowBitsWord];
666 if (test) validPixels &= ~rowMasks.rowBits[rowBitsWord];
667 if (update) rowMasks.rowBits[rowBitsWord] |= validPixels;
668 if (test || testColl) validPixels <<= rowBitsWordBit;
669
670 rowBitsWordBit = 32 - rowBitsWordBit;
671 if (rowBitsWordBit < tileWidth)
672 {
673 ++rowBitsWord;
674 uint32_t right = (tilePixels << rowBitsWordBit);
675
676 if (testColl) right &= ~rowMasks.rowSpriteBits[rowBitsWord];
677 if (test) right &= ~rowMasks.rowBits[rowBitsWord];
678
679 if (update) rowMasks.rowBits[rowBitsWord] |= right;
680 if (test || testColl) validPixels |= (right >> rowBitsWordBit);
681 }
682
683 return (test || testColl) ? validPixels : tilePixels;
684}
685
686
687/* lookup for combining ecm nibbles, returning 4 pixels.
688 *
689 * Deliberately a full table in striped SRAM. Plane 3 only ever lands in bit 2 of a pixel and
690 * nothing else does at any level (the palette note below), so a 256-entry two-plane table plus a
691 * 16-entry plane 3 mask would give the same words in a fraction of the space. That was built and
692 * rejected: it costs a load and an OR on every ECM3 quad, and the tile path takes far more lookups
693 * a line than sprites do. Revisit it only when something else needs the room. It cannot live in
694 * .scratch_x either way - core 1's stack has the top half of that bank.
695 */
696/* Every entry is written by ecmLookupInit() before `lookupsReady` is ever set, so it does not
697 need the crt0 .bss zero-fill either. */
698static uint32_t __aligned(8) PICO9918_UNINITIALIZED(ecmLookup)[16 * 16 * 16];
699
700static uint8_t PICO9918_IN_FLASH_FUNC(ecmByte)(bool h, bool m, bool l)
701{
702 return (h << 2) | (m << 1) | l;
703}
704
705/* lookup from bit planes: 333322221111 to merged palette values for four pixels
706 * NOTE: The left-most pixel is stored in the least significant byte of the result
707 * because it's more efficient to offload them that way
708 */
709static void PICO9918_IN_FLASH_FUNC(ecmLookupInit)(void)
710{
711 for (uint16_t i = 0; i < 16 * 16 * 16; ++i)
712 {
713 ecmLookup[i] =
714 (ecmByte(i & 0x800, i & 0x080, i & 0x008)) | (ecmByte(i & 0x400, i & 0x040, i & 0x004) << 8) |
715 (ecmByte(i & 0x200, i & 0x020, i & 0x002) << 16) | (ecmByte(i & 0x100, i & 0x010, i & 0x001) << 24);
716 }
717}
718
719/* random note about how palettes are applied:
720 * PR Address bit: 0 1 2 3 4 5
721 * --------------------------------------
722 * original mode: ps0 ps1 cs0 cs1 cs2 cs3
723 * 1-bit (ECM1) : ps0 cs0 cs1 cs2 cs3 px0
724 * 2-bit (ECM2) : cs0 cs1 cs2 cs3 px1 px0
725 * 3-bit (ECM3) : cs0 cs1 cs2 px2 px1 px0
726*/
727
728
729/*
730 * to generate the doubled pixels required when the sprite MAG flag is set,
731 * use a lookup table. generate the doubledBits lookup table when we need it
732 * using doubledBitsNibble.
733 */
734static uint8_t __aligned(4) doubledBitsNibble[16] = {0x00, 0x03, 0x0c, 0x0f, 0x30, 0x33, 0x3c, 0x3f,
735 0xc0, 0xc3, 0xcc, 0xcf, 0xf0, 0xf3, 0xfc, 0xff};
736
737/* lookup for doubling pixel patterns in mag mode */
738static PICO9918_SECTION_SCRATCH_X(lookup) uint16_t __aligned(4) doubledBits[256];
739static void PICO9918_IN_FLASH_FUNC(doubledBitsInit)(void)
740{
741 for (int i = 0; i < 256; ++i)
742 {
743 doubledBits[i] = (doubledBitsNibble[(i & 0xf0) >> 4] << 8) | doubledBitsNibble[i & 0x0f];
744 }
745}
746
747/* reversed bits in a byte */
748static PICO9918_SECTION_SCRATCH_X(lookup) uint8_t __aligned(4) reversedBits[256];
749
750static uint8_t PICO9918_IN_FLASH_FUNC(reverseBits)(uint8_t byte)
751{
752 byte = (byte & 0xf0) >> 4 | (byte & 0x0f) << 4;
753 byte = (byte & 0xcc) >> 2 | (byte & 0x33) << 2;
754 return (byte & 0xaa) >> 1 | (byte & 0x55) << 1;
755}
756
757/* the same reversal a text cell wants: six bits, so the two pixels it never shows fall off the
758 bottom and the mirror lands back at bit 7. Folding the shift into the table saves a shift and a
759 truncation on each of the three planes and the mask - paid only by a flipped cell, and a row of
760 those is the most expensive row a text mode has. */
761static PICO9918_SECTION_SCRATCH_X(lookup) uint8_t __aligned(4) reversedBits6[256];
762
763static void PICO9918_IN_FLASH_FUNC(reversedBitsInit)(void)
764{
765 for (int i = 0; i < 256; ++i)
766 {
767 reversedBits[i] = reverseBits(i);
768 reversedBits6[i] = reverseBits(i) << 2;
769 }
770}
771
772/* a 6-bit palette index applied to all four bytes of a uint32_t, which is one multiply and wants no
773 table at all.
774
775 Spelling it as a multiply is what makes that true. Left to itself GCC expands the constant
776 multiply into a run of shifts and adds, because materialising the constant that way costs nothing
777 - the right call for a cold caller and the wrong one for three hot ones. `mul` rather than
778 `muls`: GCC wraps inline asm in `.syntax divided`, where the Thumb-1 multiply takes two operands
779 and always sets the flags. The C arm keeps the host build and the init-time constant folding at
780 ecm0PaletteInit. */
781static inline uint32_t repeatedPalette(const uint32_t index)
782{
783#ifdef PICO_BUILD
784 uint32_t repeated = index;
785 __asm__("mul %0, %1" : "+l"(repeated) : "l"(0x01010101u));
786 return repeated;
787#else
788 return index * 0x01010101u;
789#endif
790}
791
792/* The same value, except that colour 0 of each sub-palette holds what a tile writes where it draws
793 nothing - so this one cannot be arithmetic. ECM0 tiles index it, and transparency then costs them
794 no test: `pal` is always a multiple of 16, so those four entries are exactly the zero colours. */
795static PICO9918_SECTION_SCRATCH_X(lookup) uint32_t __aligned(4) ecm0Palette[64];
796
797static void PICO9918_IN_FLASH_FUNC(ecm0PaletteInit)(void)
798{
799 for (int i = 0; i < 64; ++i)
800 {
801 ecm0Palette[i] = repeatedPalette(i);
802 }
803}
804
805/* What a tile layer writes where it draws nothing. Hardware marks a zero tile colour as not-a-pixel
806 and falls through to the backdrop; our layer buffer carries no such bit, so the backdrop colour
807 goes in directly. The exception is a non-priority bitmap layer, where zero is the composite's own
808 transparency marker and letting the layer show through matters more. Decided once per scanline. */
809static uint32_t transparentPixels[2];
810
811/* a lookup from a 4-bit mask to a word of 8-bit masks (reversed byte order) */
812static PICO9918_SECTION_SCRATCH_X(lookup) uint32_t __aligned(4) maskExpandNibbleToWordRev[16] = {
813 0x00000000, 0xff000000, 0x00ff0000, 0xffff0000, 0x0000ff00, 0xff00ff00, 0x00ffff00, 0xffffff00,
814 0x000000ff, 0xff0000ff, 0x00ff00ff, 0xffff00ff, 0x0000ffff, 0xff00ffff, 0x00ffffff, 0xffffffff};
815
816/* A 2bpp bitmap-layer nibble as its two pixels, low byte leftmost. Two of these make one
817 source byte's four pixels into one output word, which is the whole point.
818 nibble pixels value
819 0b00_00 0, 0 0x0000
820 0b01_10 1, 2 0x0201
821 0b11_11 3, 3 0x0303 */
822static PICO9918_SECTION_SCRATCH_X(lookup) uint16_t __aligned(4) bmlExpand2bpp[16] = {
823 0x0000, 0x0100, 0x0200, 0x0300, 0x0001, 0x0101, 0x0201, 0x0301,
824 0x0002, 0x0102, 0x0202, 0x0302, 0x0003, 0x0103, 0x0203, 0x0303};
825
826bool lookupsReady = false;
827void PICO9918_IN_FLASH_FUNC(initLookups)(void)
828{
829 if (lookupsReady) return;
830
831 PICO9918_DMA_CLAIM();
832
833 ecmLookupInit();
834 doubledBitsInit();
835 reversedBitsInit();
836 ecm0PaletteInit();
837
838 /* every fill is configured here: one triggered before a lazy init reached it would run unconfigured */
839 PICO9918_FILL32_INIT(PICO9918_FILL_LINE, &bg);
840 PICO9918_FILL32_SET_COUNT(PICO9918_FILL_LINE, TMS9918_PIXELS_X / 4);
841
842 PICO9918_FILL32_INIT(PICO9918_FILL_MASKS, &zeroWord);
843 PICO9918_FILL32_SET_COUNT(PICO9918_FILL_MASKS, sizeof(rowMasks) / sizeof(uint32_t));
844
845 PICO9918_FILL32_INIT(PICO9918_FILL_BORDER, &pico9918_border_bg);
846
847 PICO9918_COPY_INIT(PICO9918_COPY);
848
849 lookupsReady = true;
850}
851
852/* a tile plane's byte split into its two quads: the high nibble - the cell's left four pixels - at
853 * bit 16, the low nibble at bit 0. Three of these OR together into one accumulator holding both of
854 * the cell's ecmLookup indices, `index >> 16` and `(uint16_t)index`.
855 */
856static inline uint32_t ecmSplitQuads(const uint32_t patt)
857{
858 return (patt | (patt << 12)) & 0x000f000fu;
859}
860
861/* the index into ecmLookup for four sprite pixels, from the three left-aligned plane words: each
862 * plane's top nibble is this quad's bit for that plane, `sb0` being plane 1. The tile path's `patt`
863 * numbers them the other way, plane 3 first, and indexes the same table.
864 *
865 * Not for correctness - the planes above `ecm` are zero and only ever shifted - but for shape:
866 * `ecm` is a scanline invariant, so GCC unswitches the emit loops on it and each level gets a
867 * straight-line body with the unused planes' terms dead.
868 */
869static inline uint32_t calculateEcmIndex(const uint32_t ecm, const uint32_t sb0, const uint32_t sb1,
870 const uint32_t sb2)
871{
872 uint32_t ecmIndex = 0;
873 switch (ecm)
874 {
875 case 3:
876 ecmIndex = sb2 >> 28;
877 // fallthrough
878 case 2:
879 ecmIndex = (ecmIndex << 4) | (sb1 >> 28);
880 // fallthrough
881 default: ecmIndex = (ecmIndex << 4) | (sb0 >> 28);
882 }
883 return ecmIndex;
884}
885
886static inline void loadSpriteData(const uint8_t* vram, uint32_t* spriteBits, uint32_t pattOffset,
887 uint32_t* pattMask, const uint32_t ecm, const uint32_t ecmOffset,
888 const bool flipX, const bool sprite16)
889{
890 int i = 0;
891 do // do-while since behavior for ecm=0 and ecm==1 is the same
892 {
893 uint32_t patt = vram[pattOffset];
894 if (flipX) patt = reversedBits[patt];
895 uint32_t bits = patt << ((flipX && sprite16) ? 16 : 24);
896
897 if (sprite16)
898 {
899 patt = vram[pattOffset + PATTERN_BYTES * 2];
900 if (flipX) patt = reversedBits[patt];
901 bits |= patt << (flipX ? 24 : 16);
902 }
903 spriteBits[i] = bits;
904 *pattMask |= bits;
905 pattOffset += ecmOffset;
906 } while (++i < ecm);
907}
908
909
910/* The sprites this scanline draws, in list order. Each carries its attribute with the row inside
911 the pattern in place of the y, which the drawing pass does not need - the collect pass has the
912 whole word in a register for the zero test anyway, so keeping it spares that pass three reads of
913 VRAM, which shares its bank with the DMA and the PIO. The index is only ever read to report a
914 fifth sprite, so it sits apart rather than widening the record every sprite pays for. */
915static PICO9918_SECTION_SCRATCH_X(lookup) uint32_t spriteAttrRows[MAX_SPRITES];
916static PICO9918_SECTION_SCRATCH_X(lookup) uint8_t spriteIndices[MAX_SPRITES];
917
918/**
919 * \brief Which sprites this scanline draws at all, and where in each pattern it starts.
920 *
921 * The y tests want the scanline, the wrap threshold and the table bounds. The drawing pass wants
922 * the ECM settings, the palette and the pattern table. Neither wants the other's, and together
923 * they are more than the eight registers hold - so the list is walked once here and the drawing
924 * pass reads a row at a time instead of carrying both sets through every sprite.
925 */
926static uint32_t __time_critical_func(collectSpriteRows)(PICO9918_INST_ARG uint16_t y)
927{
928 const uint32_t unlockedMask = -(uint32_t)PICO9918_UNLOCKED(tms9918);
929 const uint32_t row30Mode =
930 (TMS_REGISTER(tms9918, PICO9918_REG_ENHANCED1) & PICO9918_R49_ROW30) & unlockedMask;
931
932 /* the wrap threshold and the row both carry the YPOS -1 offset, so the walk stays in raw YPOS */
933 const int32_t realY = (TMS_REGISTER(tms9918, PICO9918_REG_ENHANCED1) & PICO9918_R49_Y_REAL) ? 0 : 1;
934 const int32_t maxY = (row30Mode ? 0xf0 : 0xe0) - realY;
935 const int32_t yAdj = (int32_t)y - realY;
936
937 uint32_t maxSprites = TMS_REGISTER(tms9918, PICO9918_REG_MAX_SPRITES);
938 if (maxSprites > MAX_SPRITES) maxSprites = MAX_SPRITES;
939
940 const uint8_t* spriteAttr = tms9918->vram.bytes + tmsSpriteAttrTableAddr(tms9918);
941 uint32_t count = 0;
942
943 for (uint32_t spriteIdx = 0; spriteIdx < maxSprites; ++spriteIdx, spriteAttr += SPRITE_ATTR_BYTES)
944 {
945 int32_t yPos = spriteAttr[SPRITE_ATTR_Y];
946
947 /* stop processing when yPos == LAST_SPRITE_YPOS */
948 if (yPos == LAST_SPRITE_YPOS && !row30Mode)
949 {
950 break;
951 }
952
953 /* check if sprite position is in the -31 to 0 range and move back to top */
954 if (yPos > maxY) yPos -= 256;
955
956 const int32_t pattRow = yAdj - yPos;
957 if ((uint32_t)pattRow > 31)
958 {
959 continue;
960 }
961
962 const uint32_t attr = *(const uint32_t*)spriteAttr;
963
964 if (attr == 0 && unlockedMask)
965 {
966 continue;
967 }
968
969 spriteAttrRows[count] = attr;
970 ((uint8_t*)&spriteAttrRows[count])[SPRITE_ATTR_Y] = (uint8_t)pattRow;
971 spriteIndices[count] = (uint8_t)spriteIdx;
972 ++count;
973 }
974
975 return count;
976}
977
978/** \brief the sprite ECM level, zero on a locked device - the one term a clone can pin */
979static inline uint32_t spriteEcm(PICO9918_INST_ONLY_ARG)
980{
981 return (TMS_REGISTER(tms9918, PICO9918_REG_ENHANCED1) & PICO9918_R49_ECM_SPRITE) &
982 -(uint32_t)PICO9918_UNLOCKED(tms9918);
983}
984
985/** \brief Output Sprites to a scanline. ecm0 pins the ECM level to zero, which folds the
986 * plane loop to one pass, the colour shift to nothing and the whole ECM emit arm away.
987 */
988static inline uint8_t __time_critical_func(renderSprites)(PICO9918_INST_ARG const uint32_t spriteCount,
989 const bool spriteMag, const bool wide,
990 const bool ecm0,
991 uint8_t pixels[TMS9918_PIXELS_X])
992{
993 const uint32_t unlockedMask = -(uint32_t)PICO9918_UNLOCKED(tms9918);
994 const uint8_t* const vram = tms9918->vram.bytes;
995 bool hasSprites = false;
996 const uint8_t spriteSize = tmsSpriteSize(tms9918);
997 const bool sprite16 = spriteSize == 16;
998 const uint8_t spriteIdxMask = sprite16 ? 0xfc : 0xff;
999 const uint8_t spriteColorMask = 0x8f | unlockedMask;
1000 const uint8_t spriteSizePx = spriteSize << spriteMag;
1001 const uint16_t spritePatternAddr = tmsSpritePatternTableAddr(tms9918);
1002 uint32_t spritesShown = 0;
1003
1004 /* the sprite-number field reads zero unless a fifth sprite latches one in */
1005 uint8_t tempStatus = 0;
1006 uint32_t transparentCount = 0;
1007
1008 // ecm settings
1009 const uint32_t ecm = ecm0 ? 0 : spriteEcm(PICO9918_INST_ONLY);
1010 const uint32_t ecmColorOffset = (ecm == 3) ? 2 : ecm;
1011 const uint32_t ecmColorMask = (ecm == 3) ? 0x0e : 0x0f;
1012 const uint32_t ecmOffset =
1013 0x800 >> ((TMS_REGISTER(tms9918, PICO9918_REG_PAGE_SIZE) & PICO9918_R29_SPRITE_STRIDE) >> 6);
1014
1015 uint8_t pal = (TMS_REGISTER(tms9918, PICO9918_REG_PALETTE_SELECT) & PICO9918_R24_SPRITE_PS) & unlockedMask;
1016 if (ecm == 1)
1017 {
1018 pal &= 0x20;
1019 }
1020 else if (ecm)
1021 {
1022 pal = 0;
1023 }
1024
1025 const uint32_t scanlineSprites = TMS_REGISTER(tms9918, PICO9918_REG_MAX_SCAN_SPRITES);
1026 const uint32_t unlimited =
1027 (TMS_REGISTER(tms9918, PICO9918_REG_ENHANCED2) & PICO9918_R50_REPORT_MAX) & unlockedMask;
1028
1029 for (uint32_t n = 0; n < spriteCount; ++n)
1030 {
1031 const uint8_t* spriteAttr = (const uint8_t*)&spriteAttrRows[n];
1032
1033 int32_t pattRow = spriteAttr[SPRITE_ATTR_Y] >> spriteMag;
1034
1035 uint8_t thisSpriteSize = spriteSize;
1036 bool thisSprite16 = sprite16;
1037 uint8_t thisSpriteIdxMask = spriteIdxMask;
1038 uint8_t thisSpriteSizePx = spriteSizePx;
1039 uint8_t spriteAttrColor = spriteAttr[SPRITE_ATTR_COLOR] & spriteColorMask;
1040 bool opaq = false;
1041
1042 if (spriteAttrColor & 0x10)
1043 {
1044 if (sprite16)
1045 {
1046 // PICO9918-specific. If all sprites are 16px anyway, this bit is used to have opaque sprites
1047 opaq = true;
1048 }
1049 else
1050 {
1051 thisSpriteSize = 16;
1052 thisSprite16 = true;
1053 thisSpriteIdxMask = 0xfc;
1054 thisSpriteSizePx = thisSpriteSize << spriteMag;
1055 }
1056 }
1057
1058 /* check if sprite is visible on this line */
1059 if (pattRow >= thisSpriteSize)
1060 {
1061 continue;
1062 }
1063
1064 /* have we exceeded the scanline sprite limit? */
1065 if (++spritesShown > MAX_SCANLINE_SPRITES)
1066 {
1067 if (((tempStatus & PICO9918_SR0_5S) == 0) && (!unlimited || spritesShown > scanlineSprites))
1068 {
1069 tempStatus |= PICO9918_SR0_5S | spriteIndices[n];
1070 }
1071
1072 if (spritesShown > scanlineSprites) break;
1073 }
1074
1075 const int32_t earlyClockOffset = (spriteAttrColor & 0x80) ? -32 : 0;
1076 int32_t xPos = (int32_t)(spriteAttr[SPRITE_ATTR_X]) + earlyClockOffset;
1077 if ((xPos > TMS9918_PIXELS_X) || (-xPos > thisSpriteSizePx))
1078 {
1079 continue;
1080 }
1081
1082 if (spriteAttrColor & 0x20) pattRow = thisSpriteSize - pattRow - 1; // flip Y?
1083
1084 /* sprite is visible on this line */
1085 uint8_t spriteColor = (spriteAttrColor & ecmColorMask) << ecmColorOffset;
1086 const uint8_t pattIdx = spriteAttr[SPRITE_ATTR_NAME] & thisSpriteIdxMask;
1087 uint16_t pattOffset = spritePatternAddr + pattIdx * PATTERN_BYTES + (uint16_t)pattRow;
1088
1089
1090 uint32_t pattMask = 0;
1091 uint32_t spriteBits[3] = {0};
1092 const bool flipX = spriteAttrColor & 0x40;
1093
1094 loadSpriteData(vram, spriteBits, pattOffset, &pattMask, ecm, ecmOffset, flipX, thisSprite16);
1095
1096 if (opaq) pattMask = 0xffff0000;
1097
1098 /* bail early if no bits to draw */
1099 if (!pattMask)
1100 {
1101 continue;
1102 }
1103
1104 if (spriteMag)
1105 {
1106 pattMask = ((uint32_t)doubledBits[pattMask >> 24] << 16) | doubledBits[(pattMask >> 16) & 0xff];
1107 }
1108
1109 /* perform clipping operations */
1110 if (xPos < 0)
1111 {
1112 int32_t absX = -xPos;
1113 uint32_t offset = absX >> spriteMag;
1114 spriteBits[2] <<= offset;
1115 spriteBits[1] <<= offset;
1116 spriteBits[0] <<= offset;
1117 pattMask <<= absX;
1118
1119 /* bail early if no bits to draw */
1120 if (!pattMask)
1121 {
1122 continue;
1123 }
1124
1125 thisSpriteSizePx += xPos;
1126 xPos = 0;
1127 }
1128
1129 int pixelsLeft = TMS9918_PIXELS_X - xPos;
1130 if (pixelsLeft < thisSpriteSizePx)
1131 {
1132 thisSpriteSizePx = pixelsLeft;
1133 pattMask &= ~((1u << (32 - pixelsLeft)) - 1);
1134 }
1135
1136 /* test and update the collision mask */
1137 uint32_t validPixels = tmsTestCollisionMask(xPos, pattMask, thisSpriteSizePx);
1138
1139 /* if the result is different, we collided */
1140 if (validPixels != pattMask)
1141 {
1142 tempStatus |= PICO9918_SR0_COLLISION;
1143 }
1144
1145 /* LOAD-BEARING: a suppressed sprite takes the transparent arm rather than skipping,
1146 which is what leaves the collision and fifth-sprite bits above it reported. */
1147 if (PICO9918_DRAWS(tms9918, PICO9918_SUPPRESS_SPRITES) &&
1148 (ecm || (spriteColor != TMS_TRANSPARENT)))
1149 {
1150 hasSprites = true;
1151 spriteColor |= pal;
1152 if (ecm)
1153 {
1154
1155 uint32_t quadPal = repeatedPalette(spriteColor);
1156
1157 if (spriteMag)
1158 {
1159 uint8_t* p = pixels + (wide ? xPos * 2 : xPos);
1160 const uint32_t step = wide ? 2 : 1;
1161 uint32_t bits = validPixels;
1162
1163 while (bits)
1164 {
1165 if (bits >> 24)
1166 {
1167 const uint32_t ecmIndex = calculateEcmIndex(ecm, spriteBits[0], spriteBits[1], spriteBits[2]);
1168 uint32_t quad = ecmLookup[ecmIndex] | quadPal;
1169
1170 for (int n = 0; n < 4; ++n)
1171 {
1172 const uint8_t v = (uint8_t)quad;
1173 if (bits & MASK_NEXT_PIXEL)
1174 {
1175 if (wide)
1176 *(uint16_t*)p = v | (v << 8);
1177 else
1178 p[0] = v;
1179 }
1180 bits <<= 1;
1181 if (bits & MASK_NEXT_PIXEL)
1182 {
1183 if (wide)
1184 *(uint16_t*)(p + 2) = v | (v << 8);
1185 else
1186 p[1] = v;
1187 }
1188 bits <<= 1;
1189 p += 2 * step;
1190 quad >>= 8;
1191 }
1192 }
1193 else
1194 {
1195 bits <<= 8;
1196 p += 8 * step;
1197 }
1198 spriteBits[2] <<= 4;
1199 spriteBits[1] <<= 4;
1200 spriteBits[0] <<= 4;
1201 }
1202 }
1203 else if (wide)
1204 {
1205 uint32_t x = xPos;
1206
1207 while (validPixels)
1208 {
1209 uint32_t chunkMask = validPixels >> 28;
1210 if (chunkMask)
1211 {
1212 const uint32_t ecmIndex = calculateEcmIndex(ecm, spriteBits[0], spriteBits[1], spriteBits[2]);
1213 uint32_t color = ecmLookup[ecmIndex] | quadPal;
1214 uint8_t* q = pixels + x * 2;
1215
1216 for (int n = 0; n < 4; ++n)
1217 {
1218 if (chunkMask & 0x8)
1219 {
1220 const uint8_t v = (uint8_t)color;
1221 *(uint16_t*)q = v | (v << 8);
1222 }
1223 chunkMask <<= 1;
1224 color >>= 8;
1225 q += 2;
1226 }
1227 }
1228 spriteBits[2] <<= 4;
1229 spriteBits[1] <<= 4;
1230 spriteBits[0] <<= 4;
1231 x += 4;
1232 validPixels <<= 4;
1233 }
1234 }
1235 else // regular ecm sprite (8 or 16px, non-magnified)
1236 {
1237
1238 // get him to be word aligned so we can smash out 4 pixels at a time
1239 uint32_t quadOffset = xPos >> 2;
1240 const uint32_t pixOffset = xPos & 0x3;
1241 validPixels >>= pixOffset;
1242 spriteBits[2] >>= pixOffset;
1243 spriteBits[1] >>= pixOffset;
1244 spriteBits[0] >>= pixOffset;
1245
1246 uint32_t* quadPixels = (uint32_t*)pixels;
1247
1248 while (validPixels)
1249 {
1250 uint32_t chunkMask = validPixels >> 28;
1251 if (chunkMask == 0x0f)
1252 {
1253 const uint32_t ecmIndex = calculateEcmIndex(ecm, spriteBits[0], spriteBits[1], spriteBits[2]);
1254 quadPixels[quadOffset] = ecmLookup[ecmIndex] | quadPal;
1255 }
1256 else if (chunkMask)
1257 {
1258 const uint32_t maskQuad = maskExpandNibbleToWordRev[chunkMask];
1259 const uint32_t ecmIndex = calculateEcmIndex(ecm, spriteBits[0], spriteBits[1], spriteBits[2]);
1260 const uint32_t color = ecmLookup[ecmIndex] | quadPal;
1261 quadPixels[quadOffset] = (quadPixels[quadOffset] & ~maskQuad) | (color & maskQuad);
1262 }
1263 spriteBits[2] <<= 4;
1264 spriteBits[1] <<= 4;
1265 spriteBits[0] <<= 4;
1266 ++quadOffset;
1267 validPixels <<= 4;
1268 }
1269 }
1270 }
1271 else // non-ecm single-color sprite
1272 {
1273 if (!wide && pico9918_cached_mode == TMS_MODE_TEXT80) spriteColor |= spriteColor << 4;
1274
1275 while (validPixels)
1276 {
1277 if ((int32_t)validPixels < 0)
1278 {
1279 if (wide)
1280 *(uint16_t*)(pixels + xPos * 2) = spriteColor | (spriteColor << 8);
1281 else
1282 pixels[xPos] = spriteColor;
1283 }
1284 validPixels <<= 1;
1285 ++xPos;
1286 }
1287 }
1288 }
1289 else
1290 {
1291 // keep track of the transparent sprites, we remove them from the sprite mask later
1292 tmsSetTransparentSpriteMask(xPos, validPixels, thisSpriteSizePx);
1293 ++transparentCount;
1294 }
1295 }
1296
1297 tms9918->scanlineHasSprites = hasSprites;
1298
1299 // remove the transparent sprite pixels if there are any
1300 if (transparentCount)
1301 {
1302 for (int i = 0; i < 9; ++i)
1303 {
1304 rowMasks.rowSpriteBits[i] ^= rowMasks.rowTransparentSpriteBits[i];
1305 }
1306 }
1307
1308
1309 return tempStatus;
1310}
1311
1312static EMITTER_NOINLINE uint8_t
1313__time_critical_func(pico9918_output_sprites)(PICO9918_INST_ARG uint16_t y, uint8_t pixels[TMS9918_PIXELS_X])
1314{
1315 const bool spriteMag = tmsSpriteMag(tms9918);
1316
1317 if (TMS_REGISTER(tms9918, TMS_REG_0) &
1318 TMS_R0_DOUBLE_ROWS) // double rows (high-res)? still only have low-res sprites
1319 y >>= 1;
1320
1321 const uint32_t spriteCount = collectSpriteRows(PICO9918_INST y);
1322
1323 /* LOAD-BEARING: pico9918_scan_line clears scanlineHasSprites before it dispatches, and
1324 * renderSprites would only store that same false back, so nothing is owed on this path. */
1325 if (spriteCount == 0) return 0;
1326
1327#if PICO9918_TEXT80_8BPP
1328 /* the store width is inside the emit loop, so it rides a clone parameter rather than a test */
1329 if (TEXT80_WIDE_ROW)
1330 {
1331 return spriteMag ? renderSprites(PICO9918_INST spriteCount, true, true, false, pixels)
1332 : renderSprites(PICO9918_INST spriteCount, false, true, false, pixels);
1333 }
1334#endif
1335
1336 if (spriteMag)
1337 {
1338 return renderSprites(PICO9918_INST spriteCount, true, false, false, pixels);
1339 }
1340
1341 /* every locked device and every ECM0 scene lands here, so it earns a clone of its own */
1342 if (spriteEcm(PICO9918_INST_ONLY) == 0)
1343 {
1344 return renderSprites(PICO9918_INST spriteCount, false, false, true, pixels);
1345 }
1346
1347 return renderSprites(PICO9918_INST spriteCount, false, false, false, pixels);
1348}
1349
1350/* What a tile row needs that the mode, rather than the layer, decides. The name and colour
1351 * addresses stay out of it: the row loop walks them as it crosses a page boundary.
1352 */
1353typedef struct
1354{
1355 const uint8_t* pattern; /* pattern table + this row; index with name * PATTERN_BYTES */
1356 int8_t flipY; /* what the ECM attribute's Y flip adds, or 0 where it is inert */
1357 uint8_t nameMask;
1358} TileRowAddr;
1359
1360/**
1361 * \brief the per-mode half of a tile row's addressing, once per layer per scanline.
1362 *
1363 * `y` is the scrolled raster row and `rawY` the unscrolled one. Multicolor is the only mode that
1364 * needs both, and it needs them apart: its name address uses the scrolled row while its pattern
1365 * byte comes from the raw one, so a vertical scroll changes which
1366 * tiles are fetched but not which four-line block of each is shown.
1367 */
1368static inline void tileRowAddr(PICO9918_INST_ARG const uint16_t y, const uint16_t rawY, const uint8_t colorReg,
1369 const bool gm2, const bool mcm, TileRowAddr* addr, uint16_t* colorTableAddr)
1370{
1371 uint16_t pageOffset = 0;
1372 const uint8_t pattRow = mcm ? (((rawY >> 2) & 0x01) + ((rawY >> 3) & 0x03) * 2) : (y & 0x07);
1373
1374 addr->nameMask = 0xff;
1375
1376 /* Multicolor's pattern address never reads the scrolled row, so Y flip is inert there */
1377 addr->flipY = mcm ? 0 : (7 - 2 * pattRow);
1378
1379 if (gm2)
1380 {
1381 pageOffset = (((y >> 6) & 0x03) & (TMS_REGISTER(tms9918, TMS_REG_PATTERN_TABLE) & 0x03)) << 11;
1382 addr->nameMask = ((colorReg & 0x7f) << 3) | 0x07;
1383 *colorTableAddr += (pageOffset & ((colorReg & 0x60) << 6)) + pattRow;
1384 }
1385
1386 addr->pattern = tms9918->vram.bytes + tmsPatternTableAddr(tms9918) + pageOffset + pattRow;
1387}
1388
1389/* Which cell a scrolled text row starts on. Graphics cells are eight pixels wide so the scroll
1390 register splits by shifting; six does not divide, so hardware multiplies by the reciprocal
1391 instead, exactly floor(h/6) for every value the register holds. The
1392 pixel within that cell is what is left: h - 6 * cell. 80 columns doubles h first, its cells
1393 being half as wide, which is why its offset is only ever 0, 2 or 4 (:741). */
1394static inline uint32_t textScrollCell(const uint32_t hscrollPixels)
1395{
1396 return (hscrollPixels * 342) >> 11;
1397}
1398
1399/* 80 columns double the register before dividing, their cells being half as wide, and what is left
1400 inside the first cell is an even number of pixels - a whole byte at four bits a pixel, which is
1401 what lets that depth place the offset by moving the destination. Both depths and both
1402 column counts come through here so the emitter and the composite cannot disagree about it. */
1403/* Cell in the low half, the pixel within it in the high. One split a layer a line, handed to the
1404 emitter whole, so the nine wide bodies below do not each carry a copy of the division. */
1405#define TEXT_SCROLL_CELL(s) ((s) & 0xffffu)
1406#define TEXT_SCROLL_OFFSET(s) ((s) >> 16)
1407
1408static inline uint32_t textScrollSplit(const uint32_t hscroll, const bool wide)
1409{
1410 /* = wide ? hscroll * 2 : hscroll, less the branch a run-time `wide` would cost */
1411 const uint32_t h = hscroll << wide;
1412 const uint32_t cell = textScrollCell(h);
1413 const uint32_t offset = h - cell * 6;
1414
1415 /* LOAD-BEARING: the line is handed out at this offset and a caller may read it a word at a time,
1416 which an odd one costs the zero-copy path entirely. Two is the only offset six-pixel cells can
1417 leave that a word cannot start on, so it backs up a cell to eight - buffer slack covers it. */
1418 if (wide && offset == 2) return (cell ? cell - 1 : TEXT80_NUM_COLS - 1) | (8u << 16);
1419 return cell | (offset << 16);
1420}
1421
1422static inline uint32_t textPixelOffset(const uint32_t hscroll, const bool wide)
1423{
1424 return TEXT_SCROLL_OFFSET(textScrollSplit(hscroll, wide));
1425}
1426
1427static inline int scrollOffset(const uint32_t hscroll, const bool text, const bool wide)
1428{
1429 return text ? (int)textPixelOffset(hscroll, wide) : (int)(hscroll & 0x07);
1430}
1431
1432typedef struct
1433{
1434 uint8_t vertScrollReg;
1435 uint8_t yPageSwapMask;
1436 uint8_t paletteShift;
1437 uint8_t paletteMask;
1438 uint8_t startPattReg;
1439 uint8_t hpSizeMask;
1440 uint8_t priorityReg;
1441 uint8_t priorityMask;
1442 uint8_t colorTableReg;
1443 bool isTile2;
1444 uint16_t (*nameTableAddrFunc)(pico9918_t*);
1445 uint16_t (*colorTableAddrFunc)(pico9918_t*);
1447
1448static const TileLayerConfig T1_CONFIG = {.vertScrollReg = 0x1c,
1449 .yPageSwapMask = 0x01,
1450 .paletteShift = 4,
1451 .paletteMask = 0x03,
1452 .startPattReg = 0x1b,
1453 .hpSizeMask = 0x02,
1454 .priorityReg = 0, // T1 has no priority control
1455 .priorityMask = 0,
1456 .colorTableReg = TMS_REG_COLOR_TABLE,
1457 .isTile2 = false,
1458 .nameTableAddrFunc = tmsNameTableAddr,
1459 .colorTableAddrFunc = tmsColorTableAddr};
1460
1461static const TileLayerConfig T2_CONFIG = {.vertScrollReg = 0x1a,
1462 .yPageSwapMask = 0x10,
1463 .paletteShift = 2,
1464 .paletteMask = 0x0c,
1465 .startPattReg = 0x19,
1466 .hpSizeMask = 0x20,
1467 .priorityReg = 0x32,
1468 .priorityMask = 0x01,
1469 .colorTableReg = 11,
1470 .isTile2 = true,
1471 .nameTableAddrFunc = tmsNameTable2Addr,
1472 .colorTableAddrFunc = tmsColorTable2Addr};
1473
1474/* Where a scrolled 80-column row starts. Hardware doubles the scroll register before dividing by
1475 * six, T80 cells being half as wide, so the offset it leaves inside the first cell is only ever 0,
1476 * 2 or 4 pixels - a whole number of bytes at 4bpp, and never a nibble.
1477 *
1478 * `startCol` and `cells` are for the aligned emitter, which stores three words per four cells and
1479 * so cannot begin part-way through one: it backs up `bytes` cells and the caller moves the
1480 * destination four bytes for each, leaving the picture where it should be and the cells backed over
1481 * in the side border.
1482 */
1483typedef struct
1484{
1485 uint16_t startCell; /* the first cell the row shows */
1486 uint16_t startCol; /* the first cell it emits */
1487 uint8_t cells;
1488 uint8_t bytes; /* how far into the first cell the row starts */
1489 bool scrolled;
1490} TextScroll;
1491
1492static inline TextScroll textScroll80(PICO9918_INST_ARG const TileLayerConfig* config)
1493{
1494 const uint32_t hscroll =
1495 PICO9918_UNLOCKED(tms9918) ? TMS_REGISTER(tms9918, config->startPattReg) * 2 : 0;
1496 const uint32_t cell = textScrollCell(hscroll);
1497 const uint32_t bytes = (hscroll - cell * 6) >> 1;
1498
1499 TextScroll s;
1500 s.startCell = cell;
1501 s.startCol = (cell >= bytes) ? (cell - bytes) : (cell + TEXT80_NUM_COLS - bytes);
1502 s.cells = bytes ? TEXT80_NUM_COLS + 4 : TEXT80_NUM_COLS;
1503 s.bytes = bytes;
1504 s.scrolled = hscroll != 0;
1505 return s;
1506}
1507
1508/**
1509 * \brief where one layer reads this scanline from: the vertical scroll and its page swap, the name and
1510 * colour rows, and the mode's pattern table. Every mode and both layers come through here, which
1511 * is what stops the scroll from being written a fourth time.
1512 *
1513 * Position attributes decide the attribute row offset for every mode alike - by name when they are
1514 * off, ECM0 or not. `textMode` selects only what a text row does not have: the page bits, in
1515 * either direction.
1516 */
1517static inline void tileLayerAddr(PICO9918_INST_ARG const uint16_t rawY, const TileLayerConfig* config,
1518 const uint8_t numCols, const uint8_t nameAddrMask, const bool textMode,
1519 const bool gm2, const bool mcm, const bool attrPerPos, TileRowAddr* addr,
1520 uint16_t* namesAddr, uint16_t* colorAddr)
1521{
1522 uint16_t y = rawY;
1523 bool swapYPage = false;
1524
1525 if (TMS_REGISTER(tms9918, config->vertScrollReg))
1526 {
1527 int virtY = y + TMS_REGISTER(tms9918, config->vertScrollReg);
1528 const int maxY =
1529 ((TMS_REGISTER(tms9918, PICO9918_REG_ENHANCED1) & PICO9918_R49_ROW30) ? (8 * 30) : (8 * 24))
1530 << (bool)(TMS_REGISTER(tms9918, TMS_REG_0) & TMS_R0_DOUBLE_ROWS);
1531
1532 if (virtY >= maxY)
1533 {
1534 virtY -= maxY;
1535
1536 /* a text row's address carries no page bit, so the size bit does nothing there */
1537 swapYPage = !textMode && (TMS_REGISTER(tms9918, PICO9918_REG_PAGE_SIZE) & config->yPageSwapMask);
1538 }
1539
1540 y = virtY;
1541 }
1542
1543 const uint16_t rowOffset = (y >> 3) * numCols;
1544
1545 *namesAddr = (config->nameTableAddrFunc(tms9918) & (nameAddrMask << 10)) + rowOffset;
1546 if (swapYPage) *namesAddr ^= 0x800;
1547
1548 *colorAddr = config->colorTableAddrFunc(tms9918);
1549 if (attrPerPos)
1550 {
1551 if (!textMode) *colorAddr += *namesAddr & 0xc00;
1552 *colorAddr = (*colorAddr + rowOffset) & VRAM_MASK;
1553 }
1554
1555 tileRowAddr(PICO9918_INST y, rawY, TMS_REGISTER(tms9918, config->colorTableReg), gm2, mcm, addr, colorAddr);
1556}
1557
1558#define TEXT80_COLOR_WORD(n) ((uint32_t)((n) * 0x111111u))
1559#define TEXT80_MASK_WORD(bits) \
1560 (((uint32_t)((bits) & 0x20 ? 0x0000F0u : 0x0)) | ((uint32_t)((bits) & 0x10 ? 0x00000Fu : 0x0)) | \
1561 ((uint32_t)((bits) & 0x08 ? 0x00F000u : 0x0)) | ((uint32_t)((bits) & 0x04 ? 0x000F00u : 0x0)) | \
1562 ((uint32_t)((bits) & 0x02 ? 0xF00000u : 0x0)) | ((uint32_t)((bits) & 0x01 ? 0x0F0000u : 0x0)))
1563
1564static const uint32_t text80ColorWord[16] = {
1565 TEXT80_COLOR_WORD(0x0), TEXT80_COLOR_WORD(0x1), TEXT80_COLOR_WORD(0x2), TEXT80_COLOR_WORD(0x3),
1566 TEXT80_COLOR_WORD(0x4), TEXT80_COLOR_WORD(0x5), TEXT80_COLOR_WORD(0x6), TEXT80_COLOR_WORD(0x7),
1567 TEXT80_COLOR_WORD(0x8), TEXT80_COLOR_WORD(0x9), TEXT80_COLOR_WORD(0xa), TEXT80_COLOR_WORD(0xb),
1568 TEXT80_COLOR_WORD(0xc), TEXT80_COLOR_WORD(0xd), TEXT80_COLOR_WORD(0xe), TEXT80_COLOR_WORD(0xf)};
1569
1570/* the low two pattern bits never reach the screen, so indexing by the raw pattern byte and
1571 repeating each entry four times spends table space to save a shift on every cell */
1572#define TEXT80_MASK_WORD4(bits) \
1573 TEXT80_MASK_WORD(bits), TEXT80_MASK_WORD(bits), TEXT80_MASK_WORD(bits), TEXT80_MASK_WORD(bits)
1574
1575static const uint32_t text80MaskWord[256] = {
1576 TEXT80_MASK_WORD4(0x00), TEXT80_MASK_WORD4(0x01), TEXT80_MASK_WORD4(0x02), TEXT80_MASK_WORD4(0x03),
1577 TEXT80_MASK_WORD4(0x04), TEXT80_MASK_WORD4(0x05), TEXT80_MASK_WORD4(0x06), TEXT80_MASK_WORD4(0x07),
1578 TEXT80_MASK_WORD4(0x08), TEXT80_MASK_WORD4(0x09), TEXT80_MASK_WORD4(0x0a), TEXT80_MASK_WORD4(0x0b),
1579 TEXT80_MASK_WORD4(0x0c), TEXT80_MASK_WORD4(0x0d), TEXT80_MASK_WORD4(0x0e), TEXT80_MASK_WORD4(0x0f),
1580 TEXT80_MASK_WORD4(0x10), TEXT80_MASK_WORD4(0x11), TEXT80_MASK_WORD4(0x12), TEXT80_MASK_WORD4(0x13),
1581 TEXT80_MASK_WORD4(0x14), TEXT80_MASK_WORD4(0x15), TEXT80_MASK_WORD4(0x16), TEXT80_MASK_WORD4(0x17),
1582 TEXT80_MASK_WORD4(0x18), TEXT80_MASK_WORD4(0x19), TEXT80_MASK_WORD4(0x1a), TEXT80_MASK_WORD4(0x1b),
1583 TEXT80_MASK_WORD4(0x1c), TEXT80_MASK_WORD4(0x1d), TEXT80_MASK_WORD4(0x1e), TEXT80_MASK_WORD4(0x1f),
1584 TEXT80_MASK_WORD4(0x20), TEXT80_MASK_WORD4(0x21), TEXT80_MASK_WORD4(0x22), TEXT80_MASK_WORD4(0x23),
1585 TEXT80_MASK_WORD4(0x24), TEXT80_MASK_WORD4(0x25), TEXT80_MASK_WORD4(0x26), TEXT80_MASK_WORD4(0x27),
1586 TEXT80_MASK_WORD4(0x28), TEXT80_MASK_WORD4(0x29), TEXT80_MASK_WORD4(0x2a), TEXT80_MASK_WORD4(0x2b),
1587 TEXT80_MASK_WORD4(0x2c), TEXT80_MASK_WORD4(0x2d), TEXT80_MASK_WORD4(0x2e), TEXT80_MASK_WORD4(0x2f),
1588 TEXT80_MASK_WORD4(0x30), TEXT80_MASK_WORD4(0x31), TEXT80_MASK_WORD4(0x32), TEXT80_MASK_WORD4(0x33),
1589 TEXT80_MASK_WORD4(0x34), TEXT80_MASK_WORD4(0x35), TEXT80_MASK_WORD4(0x36), TEXT80_MASK_WORD4(0x37),
1590 TEXT80_MASK_WORD4(0x38), TEXT80_MASK_WORD4(0x39), TEXT80_MASK_WORD4(0x3a), TEXT80_MASK_WORD4(0x3b),
1591 TEXT80_MASK_WORD4(0x3c), TEXT80_MASK_WORD4(0x3d), TEXT80_MASK_WORD4(0x3e), TEXT80_MASK_WORD4(0x3f)};
1592
1593
1594/**
1595 * \brief one 40- or 80-column text row, six pixels a cell at one byte each
1596 *
1597 * `colorStride` is 0 when the whole row shares one colour pair. At ECM1-3 a cell is an
1598 * ordinary ECM tile six pixels wide and the fg/bg pair goes inert.
1599 */
1600PICO9918_INLINE_HOT void
1601renderTextRow(PICO9918_INST_ARG const uint8_t* __restrict rowNames, const TileRowAddr* __restrict addr,
1602 const uint8_t* __restrict rowColors, const uint32_t colorStride, uint32_t pal,
1603 uint8_t* __restrict dest, const uint32_t scroll, const bool alwaysOnTop,
1604 const uint32_t numCols, const bool isTile2, const uint32_t ecm, const bool blend)
1605{
1606 const uint8_t* __restrict patternTable = addr->pattern;
1607 const bool wide = numCols == TEXT80_NUM_COLS;
1608 const uint32_t padding = wide ? TEXT80_PADDING_PX : TEXT_PADDING_PX;
1609 const uint32_t startCell = TEXT_SCROLL_CELL(scroll);
1610 const uint32_t pixelOffset = TEXT_SCROLL_OFFSET(scroll);
1611 const uint32_t numCells = numCols + (pixelOffset ? 2 : 0);
1612
1613 uint32_t* pix32 = (uint32_t*)PICO9918_ASSUME_ALIGNED(dest, 4);
1614 uint32_t xPos = ecm ? (padding - pixelOffset) : padding;
1615
1616 const uint8_t* __restrict names = rowNames + startCell;
1617 const uint8_t* __restrict colors = rowColors + startCell * colorStride;
1618 const uint32_t colorWrap = numCols * colorStride;
1619 uint32_t col = startCell;
1620
1621 const uint32_t nameAttrMask = (ecm && !colorStride) ? 0xff : 0;
1622 const uint32_t spritePriMask = tms9918->scanlineHasSprites ? 0x80 : 0;
1623 const uint32_t spritePriForced = (isTile2 && alwaysOnTop) ? spritePriMask : 0;
1624
1625 /* the row masks are uint32_t too, so a store through one forces this reload unless it is held */
1626 const uint32_t clear = transparentPixels[0];
1627 const int32_t flipY = addr->flipY;
1628 uint32_t ecmOffset = 0, ecmColorMask = 0, ecmColorOffset = 0;
1629 const uint32_t* __restrict palette = ecm0Palette + pal;
1630 if (ecm)
1631 {
1632 ecmColorOffset = (ecm == 3) ? 2 : ecm;
1633 ecmColorMask = (ecm == 3) ? 0x0e : 0x0f;
1634 ecmOffset = 0x800 >> ((TMS_REGISTER(tms9918, PICO9918_REG_PAGE_SIZE) & PICO9918_R29_TILE_STRIDE) >> 2);
1635 pal = (ecm == 1) ? (pal & 0x20) : 0;
1636 }
1637
1638 uint8_t lastColor = 0;
1639 uint32_t bgWord = palette[0], diffWord = 0;
1640 uint32_t coverBg = 0, coverDiff = 0;
1641 uint32_t lo = 0, hi = 0, m0 = 0, m1 = 0;
1642
1643#define TEXT40_NEXT_CELL() \
1644 const uint32_t name = *names++; \
1645 const uint8_t color = colors[name & nameAttrMask]; \
1646 colors += colorStride;
1647
1648#define TEXT40_CELL(wrapping) \
1649 { \
1650 TEXT40_NEXT_CELL() \
1651 if (wrapping) \
1652 { \
1653 /* text has no page size bits, so a start cell past the last reads on into the next row */ \
1654 if (++col == numCols) \
1655 { \
1656 col = 0; \
1657 names -= numCols; \
1658 colors -= colorWrap; \
1659 } \
1660 } \
1661 const uint32_t patt = patternTable[name * PATTERN_BYTES]; \
1662 if (color != lastColor) \
1663 { \
1664 const uint8_t bgColor = color & 0xf; \
1665 const uint8_t fgColor = color >> 4; \
1666 bgWord = palette[bgColor]; \
1667 diffWord = bgWord ^ palette[fgColor]; \
1668 lastColor = color; \
1669 if (isTile2) \
1670 { \
1671 /* = bgColor ? ~0u : 0u, and that xor the same for fgColor */ \
1672 coverBg = (uint32_t)(-(int32_t)bgColor >> 31); \
1673 coverDiff = coverBg ^ (uint32_t)(-(int32_t)fgColor >> 31); \
1674 } \
1675 } \
1676 /* 0x0c, not 0x0f: six pixels a cell, so the byte's two spare bits are dropped at the index */ \
1677 const uint32_t maskLo = maskExpandNibbleToWordRev[patt >> 4]; \
1678 const uint32_t maskHi = maskExpandNibbleToWordRev[patt & 0x0c]; \
1679 lo = bgWord ^ (diffWord & maskLo); \
1680 hi = bgWord ^ (diffWord & maskHi); \
1681 if (isTile2) \
1682 { \
1683 if (blend) \
1684 { \
1685 /* cover has the shape colour does, so it selects through the masks already in hand */ \
1686 m0 = coverBg ^ (coverDiff & maskLo); \
1687 m1 = coverBg ^ (coverDiff & maskHi); \
1688 } \
1689 else \
1690 { \
1691 /* = ((fg ? bits : 0) | (bg ? ~bits : 0)) & 0x3f, off the pair memoised above */ \
1692 const uint32_t cover = ((coverBg ^ (coverDiff & (patt >> 2))) & 0x3f) << 26; \
1693 /* rolled every cell, drawn or not, or the bit position stops tracking */ \
1694 coverAcc |= cover >> coverBit; \
1695 coverBit += 6; \
1696 if (coverBit >= 32) \
1697 { \
1698 *coverWord++ |= coverAcc; \
1699 coverBit -= 32; \
1700 coverAcc = cover << (6 - coverBit); \
1701 } \
1702 } \
1703 } \
1704 }
1705
1706#define TEXT40_ECM_CELL() \
1707 { \
1708 TEXT40_NEXT_CELL() \
1709 const uint8_t* pattData = patternTable + name * PATTERN_BYTES + ((color & 0x20) ? flipY : 0); \
1710 uint32_t pattMask = (color & 0x10) ? 0 : 0xff; \
1711 uint32_t cover = 0; \
1712 uint8_t patt[3] = {0}; \
1713 switch (ecm) \
1714 { \
1715 case 3: patt[0] = pattData[ecmOffset * 2]; pattMask |= patt[0]; \
1716 case 2: patt[1] = pattData[ecmOffset]; pattMask |= patt[1]; \
1717 default: patt[2] = *pattData; pattMask |= patt[2]; \
1718 } \
1719 if (pattMask) \
1720 { \
1721 if (color & 0x40) \
1722 { \
1723 patt[0] = reversedBits6[patt[0]]; \
1724 patt[1] = reversedBits6[patt[1]]; \
1725 patt[2] = reversedBits6[patt[2]]; \
1726 pattMask = reversedBits6[pattMask]; \
1727 } \
1728 cover = (pattMask & 0xfc) << 24; \
1729 if ((color | spritePriForced) & spritePriMask) \
1730 tmsClearRowBitsMask(xPos, cover, 6, rowMasks.rowSpriteBits); \
1731 uint32_t index = 0; \
1732 switch (ecm) \
1733 { \
1734 case 3: index = ecmSplitQuads(patt[0]) << 8; \
1735 case 2: index |= ecmSplitQuads(patt[1]) << 4; \
1736 default: index |= ecmSplitQuads(patt[2]); \
1737 } \
1738 const uint32_t cellPal = repeatedPalette(pal | ((color & ecmColorMask) << ecmColorOffset)); \
1739 lo = ecmLookup[index >> 16] | cellPal; \
1740 hi = ecmLookup[(uint16_t)index] | cellPal; \
1741 if (!isTile2 && (color & 0x10)) \
1742 { \
1743 lo = clear ^ ((lo ^ clear) & maskExpandNibbleToWordRev[pattMask >> 4]); \
1744 hi = clear ^ ((hi ^ clear) & maskExpandNibbleToWordRev[pattMask & 0x0f]); \
1745 } \
1746 } \
1747 else \
1748 { \
1749 lo = hi = clear; \
1750 } \
1751 if (isTile2) \
1752 { \
1753 /* every cell rolls the six-bit accumulator, drawn or not, or the bit position stops tracking */ \
1754 coverAcc |= cover >> coverBit; \
1755 coverBit += 6; \
1756 if (coverBit >= 32) \
1757 { \
1758 *coverWord++ |= coverAcc; \
1759 coverBit -= 32; \
1760 coverAcc = cover << (6 - coverBit); \
1761 } \
1762 } \
1763 xPos += 6; \
1764 }
1765
1766 if (ecm)
1767 {
1768 /* do not hoist out of the branch: shared, these stay live across both and the ECM row drops rows */
1769 uint32_t* coverWord = tms9918->layerSelectionMask + (padding >> 5);
1770 uint32_t coverAcc = 0, coverBit = padding & 0x1f;
1771
1772 uint8_t* p = dest;
1773 uint32_t remaining = numCells;
1774 uint32_t run = (startCell < numCols) ? (numCols - startCell) : numCells;
1775
1776 while (remaining)
1777 {
1778 if (run > remaining) run = remaining;
1779 remaining -= run;
1780
1781 while (run--)
1782 {
1783 TEXT40_ECM_CELL();
1784 *(uint16_t*)(p) = lo;
1785 *(uint16_t*)(p + 2) = lo >> 16;
1786 *(uint16_t*)(p + 4) = hi;
1787 p += 6;
1788 }
1789
1790 names = rowNames;
1791 colors = rowColors;
1792 run = numCols;
1793 }
1794
1795 if (isTile2) *coverWord |= coverAcc;
1796 }
1797 else if (blend)
1798 {
1799 /* named, not used: the cell macro's other arm still has to compile here */
1800 uint32_t* coverWord = tms9918->layerSelectionMask;
1801 uint32_t coverAcc = 0, coverBit = 0;
1802
1803 uint16_t* p = (uint16_t*)dest;
1804 uint32_t remaining = numCells;
1805 uint32_t run = (startCell < numCols) ? (numCols - startCell) : numCells;
1806
1807 while (remaining)
1808 {
1809 if (run > remaining) run = remaining;
1810 remaining -= run;
1811
1812 while (run--)
1813 {
1814 TEXT40_CELL(false);
1815
1816 // a cell covering nothing merges nothing: all three of these are provably no-ops
1817 if (m0 | m1)
1818 {
1819 uint32_t d;
1820 d = p[0];
1821 p[0] = d ^ ((d ^ lo) & m0);
1822 d = p[1];
1823 p[1] = d ^ ((d ^ (lo >> 16)) & (m0 >> 16));
1824 d = p[2];
1825 p[2] = d ^ ((d ^ hi) & m1);
1826 }
1827 p += 3;
1828 }
1829
1830 names = rowNames;
1831 colors = rowColors;
1832 run = numCols;
1833 }
1834 }
1835 else
1836 {
1837 /* do not hoist out of the branch: shared, these stay live across both and the ECM row drops rows */
1838 uint32_t* coverWord = tms9918->layerSelectionMask + (padding >> 5);
1839 uint32_t coverAcc = 0, coverBit = padding & 0x1f;
1840
1841 for (uint32_t tileX = 0; tileX < numCells; tileX += 2)
1842 {
1843 uint16_t* pix16 = (uint16_t*)pix32;
1844
1845 TEXT40_CELL(true);
1846 pix32[0] = lo;
1847 pix16[2] = hi;
1848 TEXT40_CELL(true);
1849 pix16[3] = lo;
1850 pix32[2] = (lo >> 16) | (hi << 16);
1851 pix32 += 3;
1852 }
1853
1854 if (isTile2) *coverWord |= coverAcc;
1855 }
1856
1857#undef TEXT40_ECM_CELL
1858#undef TEXT40_CELL
1859#undef TEXT40_NEXT_CELL
1860}
1861
1862/* One body per (layer, ecm). 80 columns arrives here on the 8bpp tier, where a text row is the same
1863 byte-per-pixel shape and only the count differs. At 4bpp it cannot use a layer buffer at all, so
1864 RP2040 keeps the blend-in-place emitter below. */
1865#define TEXT_ROW_PARAMS \
1866 PICO9918_INST_ARG const uint8_t *rowNames, const TileRowAddr *addr, const uint8_t *rowColors, \
1867 const uint32_t colorStride, const uint32_t pal, uint8_t *dest, const uint32_t scroll, \
1868 const bool alwaysOnTop
1869#define TEXT_ROW_CLONE(name, cols, t2, e) \
1870 static EMITTER_NOINLINE void __time_critical_func(name)(TEXT_ROW_PARAMS) \
1871 { \
1872 renderTextRow(PICO9918_INST rowNames, addr, rowColors, colorStride, pal, dest, scroll, alwaysOnTop, \
1873 cols, t2, e, false); \
1874 }
1875
1876TEXT_ROW_CLONE(text40RowT1, TEXT_NUM_COLS, false, 0)
1877TEXT_ROW_CLONE(text40RowT1Ecm1, TEXT_NUM_COLS, false, 1)
1878TEXT_ROW_CLONE(text40RowT1Ecm2, TEXT_NUM_COLS, false, 2)
1879TEXT_ROW_CLONE(text40RowT1Ecm3, TEXT_NUM_COLS, false, 3)
1880TEXT_ROW_CLONE(text40RowT2, TEXT_NUM_COLS, true, 0)
1881TEXT_ROW_CLONE(text40RowT2Ecm1, TEXT_NUM_COLS, true, 1)
1882TEXT_ROW_CLONE(text40RowT2Ecm2, TEXT_NUM_COLS, true, 2)
1883TEXT_ROW_CLONE(text40RowT2Ecm3, TEXT_NUM_COLS, true, 3)
1884
1885static void (*const textRowClones[2][4])(TEXT_ROW_PARAMS) = {
1886 {text40RowT1, text40RowT1Ecm1, text40RowT1Ecm2, text40RowT1Ecm3},
1887 {text40RowT2, text40RowT2Ecm1, text40RowT2Ecm2, text40RowT2Ecm3}};
1888
1889#if PICO9918_TEXT80_8BPP
1890/* 80 columns are the same body at a different count: one byte a pixel, six pixels a cell, the same
1891 word-and-halfword stores at the same alignments, and a layer buffer the composite can arbitrate
1892 because a mask bit now covers one pixel rather than two */
1893TEXT_ROW_CLONE(text80RowT1, TEXT80_NUM_COLS, false, 0)
1894TEXT_ROW_CLONE(text80RowT1Ecm1, TEXT80_NUM_COLS, false, 1)
1895TEXT_ROW_CLONE(text80RowT1Ecm2, TEXT80_NUM_COLS, false, 2)
1896TEXT_ROW_CLONE(text80RowT1Ecm3, TEXT80_NUM_COLS, false, 3)
1897TEXT_ROW_CLONE(text80RowT2, TEXT80_NUM_COLS, true, 0)
1898TEXT_ROW_CLONE(text80RowT2Ecm1, TEXT80_NUM_COLS, true, 1)
1899TEXT_ROW_CLONE(text80RowT2Ecm2, TEXT80_NUM_COLS, true, 2)
1900TEXT_ROW_CLONE(text80RowT2Ecm3, TEXT80_NUM_COLS, true, 3)
1901
1902/* Layer 2 folded into layer 1's line as it emits, instead of into a coverage mask for the
1903 composite to arbitrate. Only an ECM0 line with layer 1 present may do it - above ECM0 priority is
1904 attr(0) per tile - so it is a body of its own rather than what the layer always does. */
1905static EMITTER_NOINLINE void __time_critical_func(text80RowT2Blend)(TEXT_ROW_PARAMS)
1906{
1907 renderTextRow(PICO9918_INST rowNames, addr, rowColors, colorStride, pal, dest, scroll, alwaysOnTop,
1908 TEXT80_NUM_COLS, true, 0, true);
1909}
1910
1911static void (*const text80RowClones[2][4])(TEXT_ROW_PARAMS) = {
1912 {text80RowT1, text80RowT1Ecm1, text80RowT1Ecm2, text80RowT1Ecm3},
1913 {text80RowT2, text80RowT2Ecm1, text80RowT2Ecm2, text80RowT2Ecm3}};
1914#endif
1915
1916/**
1917 * \brief one 80-column text row, six pixels a cell at half a byte each. Four cells are twelve bytes, so a
1918 * group goes out as three words.
1919 *
1920 * `scrolled` splits it in two. Unscrolled, the colour bytes arrive four at a time as one aligned
1921 * word and the row cannot wrap, so the colour memo and the four-cell transparent skip both live in
1922 * that loop. Scrolled, the row starts on an arbitrary cell and wraps to its own first one, so
1923 * neither holds: the colour table is not word-aligned to the group and the group can straddle the
1924 * wrap. That clone reads a colour byte a cell and does without the skip.
1925 *
1926 * The fine offset is only ever 0, 2 or 4 pixels, which at 4bpp is a whole number of
1927 * bytes - so the caller backs the start cell up by that many and moves `pixels` four bytes per
1928 * byte of offset, which keeps the three-word store aligned and puts the cells it skipped over in
1929 * the side border.
1930 */
1931PICO9918_INLINE_HOT void
1932renderText80Row(PICO9918_INST_ARG const uint8_t* __restrict rowNames,
1933 const uint8_t* __restrict patternTable, const uint8_t* __restrict rowColors,
1934 const bool opaq, uint8_t* __restrict pixels, const uint32_t startCol,
1935 const uint32_t numCells, const bool scrolled)
1936{
1937 const uint8_t bgc = tmsMainBgColor(tms9918);
1938 const uint8_t* rowNamesTable = rowNames + startCol;
1939 const uint8_t* colors = rowColors + startCol;
1940 const uint32_t* colorTable32 = (const uint32_t*)PICO9918_ASSUME_ALIGNED(rowColors, 4);
1941 uint32_t* pix32 = (uint32_t*)PICO9918_ASSUME_ALIGNED(pixels, 4);
1942 uint32_t col = startCol;
1943
1944 /* the row wraps to its own first cell, with no page swap */
1945#define TEXT80_NEXT_CELL() \
1946 { \
1947 ++rowNamesTable; \
1948 if (scrolled && ++col == TEXT80_NUM_COLS) \
1949 { \
1950 col = 0; \
1951 rowNamesTable = rowNames; \
1952 colors = rowColors; \
1953 } \
1954 else if (scrolled) \
1955 ++colors; \
1956 }
1957
1958 if (opaq)
1959 {
1960 uint8_t lastColor = 0;
1961 uint32_t bgColorMask = text80ColorWord[bgc];
1962 uint32_t diffColorMask = 0;
1963
1964 for (uint8_t tileX = 0; tileX < numCells; tileX += 4)
1965 {
1966 uint32_t colorWord = scrolled ? 0 : *colorTable32++;
1967 uint32_t word, acc;
1968
1969#define TEXT80_OPAQUE_CELL() \
1970 { \
1971 uint8_t color; \
1972 if (scrolled) \
1973 color = *colors; \
1974 else \
1975 { \
1976 color = (uint8_t)colorWord; \
1977 colorWord >>= 8; \
1978 } \
1979 if (color != lastColor) \
1980 { \
1981 const uint8_t bgColor = color & 0xf; \
1982 const uint8_t fgColor = color >> 4; \
1983 bgColorMask = text80ColorWord[bgColor ? bgColor : bgc]; \
1984 diffColorMask = bgColorMask ^ text80ColorWord[fgColor ? fgColor : bgc]; \
1985 lastColor = color; \
1986 } \
1987 const uint32_t mask = text80MaskWord[patternTable[*rowNamesTable * PATTERN_BYTES]]; \
1988 TEXT80_NEXT_CELL(); \
1989 word = bgColorMask ^ (diffColorMask & mask); \
1990 }
1991
1992 TEXT80_OPAQUE_CELL();
1993 acc = word;
1994 TEXT80_OPAQUE_CELL();
1995 *pix32++ = acc | (word << 24);
1996 acc = word >> 8;
1997 TEXT80_OPAQUE_CELL();
1998 *pix32++ = acc | (word << 16);
1999 acc = word >> 16;
2000 TEXT80_OPAQUE_CELL();
2001 *pix32++ = acc | (word << 8);
2002
2003#undef TEXT80_OPAQUE_CELL
2004 }
2005 }
2006 else
2007 {
2008 for (uint8_t tileX = 0; tileX < numCells; tileX += 4)
2009 {
2010 uint32_t colorWord = scrolled ? 0 : *colorTable32++;
2011 uint32_t val, sel, accVal, accSel;
2012
2013 if (!scrolled && !colorWord)
2014 {
2015 rowNamesTable += 4;
2016 pix32 += 3;
2017 continue;
2018 }
2019
2020#define TEXT80_OVERLAY_CELL() \
2021 { \
2022 const uint32_t mask = text80MaskWord[patternTable[*rowNamesTable * PATTERN_BYTES]]; \
2023 uint8_t colorByte; \
2024 if (scrolled) \
2025 colorByte = *colors; \
2026 else \
2027 { \
2028 colorByte = (uint8_t)colorWord; \
2029 colorWord >>= 8; \
2030 } \
2031 TEXT80_NEXT_CELL(); \
2032 const uint32_t fgWord = text80ColorWord[colorByte >> 4]; \
2033 const uint32_t bgWord = text80ColorWord[colorByte & 0xf]; \
2034 const uint32_t fgSel = fgWord ? mask : 0; \
2035 const uint32_t bgSel = bgWord ? (~mask & 0xffffffu) : 0; \
2036 sel = fgSel | bgSel; \
2037 val = (fgWord & fgSel) | (bgWord & bgSel); \
2038 }
2039
2040#define TEXT80_OVERLAY_STORE(shift) \
2041 { \
2042 const uint32_t m = accSel | (sel << (shift)); \
2043 const uint32_t v = accVal | (val << (shift)); \
2044 const uint32_t old = *pix32; \
2045 *pix32++ = old ^ ((old ^ v) & m); \
2046 }
2047
2048 TEXT80_OVERLAY_CELL();
2049 accVal = val;
2050 accSel = sel;
2051 TEXT80_OVERLAY_CELL();
2052 TEXT80_OVERLAY_STORE(24);
2053 accVal = val >> 8;
2054 accSel = sel >> 8;
2055 TEXT80_OVERLAY_CELL();
2056 TEXT80_OVERLAY_STORE(16);
2057 accVal = val >> 16;
2058 accSel = sel >> 16;
2059 TEXT80_OVERLAY_CELL();
2060 TEXT80_OVERLAY_STORE(8);
2061
2062#undef TEXT80_OVERLAY_CELL
2063#undef TEXT80_OVERLAY_STORE
2064 }
2065 }
2066#undef TEXT80_NEXT_CELL
2067}
2068
2069/* One body per (can this row scroll). The unscrolled one keeps its cell count as a constant. */
2070#define TEXT80_ROW_CLONE(name, cells, scroll) \
2071 static EMITTER_NOINLINE void __time_critical_func(name)( \
2072 PICO9918_INST_ARG const uint8_t* rowNames, const uint8_t* patternTable, const uint8_t* rowColors, \
2073 const bool opaq, uint8_t* pixels, const uint32_t startCol, const uint32_t numCells) \
2074 { \
2075 renderText80Row(PICO9918_INST rowNames, patternTable, rowColors, opaq, pixels, startCol, cells, \
2076 scroll); \
2077 }
2078
2079TEXT80_ROW_CLONE(text80Row, TEXT80_NUM_COLS, false)
2080TEXT80_ROW_CLONE(text80RowScrolled, numCells, true)
2081
2082static inline void renderText80Layer(PICO9918_INST_ARG const uint8_t* rowNames,
2083 const uint8_t* patternTable, const uint8_t* rowColors,
2084 const bool opaq, uint8_t* pixels, const uint32_t startCol,
2085 const uint32_t numCells, const bool scrolled)
2086{
2087 if (scrolled)
2088 text80RowScrolled(PICO9918_INST rowNames, patternTable, rowColors, opaq, pixels, startCol, numCells);
2089 else
2090 text80Row(PICO9918_INST rowNames, patternTable, rowColors, opaq, pixels, 0, TEXT80_NUM_COLS);
2091}
2092
2093
2094/* one run of 80-column cells in the two colours R7 holds. Both are constant down the whole row, so
2095 this is the one text path that needs no colour memo: one lookup turns a pattern byte into all six
2096 pixels and the pair of colours is applied to it whole, the same expansion renderText80Row uses.
2097 Three byte stores a cell, so a run can start anywhere - which is what lets the row's wrap be a
2098 second call rather than a test on every cell. */
2099static inline uint8_t* text80TwoTone(const uint8_t* __restrict names, const uint8_t* __restrict patternTable,
2100 const uint32_t bgWord, const uint32_t diffWord,
2101 uint8_t* __restrict pixels, uint32_t cells)
2102{
2103 while (cells--)
2104 {
2105 const uint32_t pixelWord = bgWord ^ (diffWord & text80MaskWord[patternTable[*names++ * PATTERN_BYTES]]);
2106
2107 *pixels++ = pixelWord;
2108 *pixels++ = pixelWord >> 8;
2109 *pixels++ = pixelWord >> 16;
2110 }
2111 return pixels;
2112}
2113
2114/**
2115 * \brief generate a 40- or 80-column text mode scanline.
2116 *
2117 * Text has a path of its own, and hardware says so: it selects a different name address, a
2118 * different attribute address, a different colour source, a different flip, a different horizontal
2119 * scroll counter and a different expansion width - six pixels against eight. What it shares with
2120 * the graphics modes is the fetch schedule, which for us is the address generator above, the sprite
2121 * pass and the backdrop.
2122 *
2123 * The emitters write the finished line rather than a layer buffer, so the composite runs only where
2124 * something has to be arbitrated - a bitmap layer, or a priority second layer.
2125 */
2126static EMITTER_NOINLINE void __time_critical_func(text_scan_line)(PICO9918_INST_ARG uint16_t y,
2127 uint8_t pixels[TMS9918_PIXELS_X])
2128{
2129 const bool wide = pico9918_cached_mode == TMS_MODE_TEXT80;
2130 const uint8_t numCols = wide ? TEXT80_NUM_COLS : TEXT_NUM_COLS;
2131 const uint8_t nameTableMask = (wide && !PICO9918_UNLOCKED(tms9918)) ? 0x0c : 0x0f;
2132
2133 const bool attrPerPos =
2134 PICO9918_UNLOCKED(tms9918) && (TMS_REGISTER(tms9918, PICO9918_REG_ENHANCED2) & PICO9918_R50_POS_ATTR);
2135
2136 TileRowAddr addr;
2137 uint16_t rowNamesAddr, colorTableAddr;
2138 tileLayerAddr(PICO9918_INST y, &T1_CONFIG, numCols, nameTableMask, true, false, false, attrPerPos, &addr,
2139 &rowNamesAddr, &colorTableAddr);
2140
2141 const uint8_t* patternTable = addr.pattern;
2142 const pico9918_color_t bgColor = tmsMainBgColor(tms9918);
2143 uint32_t* border = (uint32_t*)pixels;
2144
2145 pixels += TEXT_PADDING_PX;
2146
2147 const TextScroll t1 = textScroll80(PICO9918_INST & T1_CONFIG);
2148
2149 PICO9918_FILL32_WAIT(PICO9918_FILL_LINE);
2150
2151 if (attrPerPos)
2152 {
2153 const bool tilesDisabled = TMS_REGISTER(tms9918, PICO9918_REG_ENHANCED2) & PICO9918_R50_TILE1_OFF;
2154 if (!tilesDisabled)
2155 renderText80Layer(PICO9918_INST tms9918->vram.bytes + rowNamesAddr, patternTable,
2156 tms9918->vram.bytes + colorTableAddr, true, pixels - 4 * t1.bytes, t1.startCol,
2157 t1.cells, t1.scrolled);
2158
2159 if (TMS_REGISTER(tms9918, PICO9918_REG_ENHANCED1) & PICO9918_R49_TILE2_ENABLE)
2160 {
2161 const TextScroll t2 = textScroll80(PICO9918_INST & T2_CONFIG);
2162 tileLayerAddr(PICO9918_INST y, &T2_CONFIG, numCols, nameTableMask, true, false, false, attrPerPos, &addr,
2163 &rowNamesAddr, &colorTableAddr);
2164
2165 renderText80Layer(PICO9918_INST tms9918->vram.bytes + rowNamesAddr, addr.pattern,
2166 tms9918->vram.bytes + colorTableAddr, false, pixels - 4 * t2.bytes, t2.startCol,
2167 t2.cells, t2.scrolled);
2168 }
2169 }
2170 else // just plain old two-tone
2171 {
2172 const pico9918_color_t fgColor = tmsMainFgColor(tms9918);
2173 const uint8_t* rowNamesTable = tms9918->vram.bytes + rowNamesAddr;
2174
2175 if (wide)
2176 {
2177 const uint32_t bgWord = text80ColorWord[bgColor];
2178 const uint32_t diffWord = bgWord ^ text80ColorWord[fgColor];
2179
2180 const uint32_t cells = t1.bytes ? TEXT80_NUM_COLS + 1 : TEXT80_NUM_COLS;
2181 uint32_t run = cells;
2182 if (t1.startCell < TEXT80_NUM_COLS && t1.startCell + run > TEXT80_NUM_COLS)
2183 run = TEXT80_NUM_COLS - t1.startCell;
2184
2185 pixels =
2186 text80TwoTone(rowNamesTable + t1.startCell, patternTable, bgWord, diffWord, pixels - t1.bytes, run);
2187 if (run < cells) text80TwoTone(rowNamesTable, patternTable, bgWord, diffWord, pixels, cells - run);
2188 }
2189 else
2190 {
2191 const uint32_t bgWord = repeatedPalette(bgColor);
2192 const uint32_t diffWord = bgWord ^ repeatedPalette(fgColor);
2193 uint32_t* pix32 = (uint32_t*)PICO9918_ASSUME_ALIGNED(pixels, 4);
2194
2195 for (uint8_t tileX = 0; tileX < TEXT_NUM_COLS; tileX += 2)
2196 {
2197 uint16_t* pix16 = (uint16_t*)pix32;
2198 uint32_t patt = patternTable[*rowNamesTable++ * PATTERN_BYTES];
2199
2200 pix32[0] = bgWord ^ (diffWord & maskExpandNibbleToWordRev[patt >> 4]);
2201 pix16[2] = bgWord ^ (diffWord & maskExpandNibbleToWordRev[patt & 0x0f]);
2202
2203 patt = patternTable[*rowNamesTable++ * PATTERN_BYTES];
2204 const uint32_t lo = bgWord ^ (diffWord & maskExpandNibbleToWordRev[patt >> 4]);
2205 const uint32_t hi = bgWord ^ (diffWord & maskExpandNibbleToWordRev[patt & 0x0f]);
2206
2207 pix16[3] = lo;
2208 pix32[2] = (lo >> 16) | (hi << 16);
2209 pix32 += 3;
2210 }
2211 }
2212 }
2213
2214 border[0] = border[1] = bg;
2215 border[62] = border[63] = bg;
2216}
2217
2218/** \brief Write full tile to aligned buffer - 8 pixels at once */
2219static inline void writeToAlignedBuffer(uint8_t* buffer, uint32_t xPos, const uint32_t left,
2220 const uint32_t right)
2221{
2222 uint32_t* buffer_words = (uint32_t*)(buffer + xPos);
2223 buffer_words[0] = left;
2224 buffer_words[1] = right;
2225}
2226
2227/**
2228 * \brief render an ECM0 (enhanced color mode) graphics I tile. basically the same as original, but can scroll
2229 *
2230 * INLINE: so will be different versions generated, depending on hard-coded (or known at compile-time) arguments
2231 */
2232static inline void renderEcm0Tile(PICO9918_INST_ARG uint8_t* buffer, const uint32_t xPos,
2233 const uint8_t pattIdx, const uint8_t patternTable[],
2234 const uint32_t colorTableAddr, const uint32_t pal, const bool isTile2,
2235 const bool gm2Color, const bool mcm)
2236{
2237 /* is the pixel mask already full here? then nothing of this tile can show */
2238 if (!isTile2 && !tmsTestRowBitsMaskAligned(xPos, 0xffu << 24, tms9918->finalMask))
2239 {
2240 writeToAlignedBuffer(buffer, xPos, transparentPixels[0], transparentPixels[1]);
2241 return;
2242 }
2243
2244 const uint32_t pattByte = patternTable[pattIdx * PATTERN_BYTES];
2245 const uint32_t colorByte =
2246 mcm ? pattByte
2247 : tms9918->vram.bytes[colorTableAddr + (gm2Color ? pattIdx * PATTERN_BYTES : (pattIdx >> 3))];
2248 const uint32_t patt = mcm ? 0xf0 : pattByte;
2249
2250 const uint32_t bgColor = colorByte & 0x0f;
2251 const uint32_t fgColor = colorByte >> 4;
2252
2253 const uint32_t bgPalette = ecm0Palette[pal | bgColor];
2254 const uint32_t fgPalette = ecm0Palette[pal | fgColor];
2255
2256 uint32_t pattMask = 0xff;
2257 if (!bgColor) pattMask &= patt;
2258 if (!fgColor) pattMask ^= patt;
2259
2260 pattMask <<= 24;
2261 if (isTile2) tmsUpdateRowBitsMaskAligned(xPos, pattMask, tms9918->layerSelectionMask);
2262
2263 if (!pattMask)
2264 {
2265 if (!isTile2)
2266 {
2267 writeToAlignedBuffer(buffer, xPos, transparentPixels[0], transparentPixels[1]);
2268 }
2269 return;
2270 }
2271
2272 const uint32_t rightMask = maskExpandNibbleToWordRev[patt & 0xf];
2273 const uint32_t leftMask = maskExpandNibbleToWordRev[patt >> 4];
2274
2275 writeToAlignedBuffer(buffer, xPos, (fgPalette & leftMask) | (bgPalette & ~leftMask),
2276 (fgPalette & rightMask) | (bgPalette & ~rightMask));
2277}
2278
2279
2280/** \brief render one ECM tile into the layer buffer */
2281static inline void
2282renderEcmTileToAlignedBuffer(PICO9918_INST_ARG uint8_t* buffer, const uint32_t xPos, const uint32_t pixelOffset,
2283 const uint8_t pattIdx, const uint8_t patternTable[],
2284 const uint32_t colorTableAddr, const uint32_t ecm, const uint32_t ecmOffset,
2285 const uint32_t ecmColorMask, const uint32_t ecmColorOffset, const uint32_t pal,
2286 const bool attrPerPos, const int32_t flipY, const uint32_t tileIndex,
2287 uint32_t* lastEmpty, const bool isTile2, const bool alwaysOnTop)
2288{
2289 if ((*lastEmpty == pattIdx) ||
2290 (!isTile2 && !tmsTestRowBitsMaskAligned(xPos, 0xffu << 24, tms9918->finalMask)))
2291 {
2292 // T1 must still write: transparentPixels is the backdrop, or 0 under a bitmap layer
2293 if (!isTile2)
2294 {
2295 writeToAlignedBuffer(buffer, xPos, transparentPixels[0], transparentPixels[1]);
2296 }
2297 return;
2298 }
2299
2300 /* grab the attributes for this tile */
2301 uint32_t colorTableOffset = attrPerPos ? tileIndex : pattIdx;
2302 uint32_t pattOffset = pattIdx * PATTERN_BYTES;
2303
2304 const uint32_t colorByte = tms9918->vram.bytes[colorTableAddr + colorTableOffset];
2305
2306 const uint8_t* pattData = patternTable + pattOffset;
2307
2308 /* the pattern pointer already carries this row, so a Y flip only has to step to its mirror */
2309 if (colorByte & 0x20) pattData += flipY;
2310
2311 uint32_t pattMask = (colorByte & 0x10) ? 0 : 0xff; // handle transparency flag
2312 uint32_t index = 0;
2313
2314 uint8_t patt[3] = {0}; // indexes into this are reversed. ecm3 is in index 0
2315
2316 switch (ecm)
2317 {
2318 case 3: patt[0] = pattData[ecmOffset * 2]; pattMask |= patt[0];
2319 case 2: patt[1] = pattData[ecmOffset]; pattMask |= patt[1];
2320 default: patt[2] = *pattData; pattMask |= patt[2];
2321 }
2322
2323 /* have we any pixels to draw? */
2324 if (pattMask)
2325 {
2326 if (colorByte & 0x40) // flipX
2327 {
2328 patt[0] = reversedBits[patt[0]];
2329 patt[1] = reversedBits[patt[1]];
2330 patt[2] = reversedBits[patt[2]];
2331 pattMask = reversedBits[pattMask];
2332 }
2333
2334 const uint32_t priority = alwaysOnTop || (colorByte & 0x80);
2335 pattMask <<= 24;
2336
2337 if (isTile2) tmsUpdateRowBitsMaskAligned(xPos, pattMask, tms9918->layerSelectionMask);
2338 if (priority && tms9918->scanlineHasSprites)
2339 {
2340 const uint32_t offScreen = xPos ? 0 : pixelOffset;
2341 tmsClearRowBitsMask(xPos - pixelOffset + offScreen, pattMask << offScreen, 8, rowMasks.rowSpriteBits);
2342 }
2343
2344 switch (ecm)
2345 {
2346 case 3: index = ecmSplitQuads(patt[0]) << 8;
2347 case 2: index |= ecmSplitQuads(patt[1]) << 4;
2348 default: index |= ecmSplitQuads(patt[2]);
2349 }
2350
2351 const uint32_t palette = repeatedPalette(pal | ((colorByte & ecmColorMask) << ecmColorOffset));
2352 const uint32_t left = ecmLookup[index >> 16] | palette;
2353 const uint32_t right = ecmLookup[(uint16_t)index] | palette;
2354
2355 if (!isTile2 && (colorByte & 0x10))
2356 {
2357 const uint32_t clear = transparentPixels[0];
2358 writeToAlignedBuffer(buffer, xPos, clear ^ ((left ^ clear) & maskExpandNibbleToWordRev[pattMask >> 28]),
2359 clear ^ ((right ^ clear) & maskExpandNibbleToWordRev[(pattMask >> 24) & 0x0f]));
2360 }
2361 else
2362 {
2363 // Write to aligned buffer instead of doing expensive bit shifting
2364 writeToAlignedBuffer(buffer, xPos, left, right);
2365 }
2366 }
2367 else
2368 {
2369 // T1 must still write even when the tile is empty, for the same reason
2370 if (!isTile2)
2371 {
2372 writeToAlignedBuffer(buffer, xPos, transparentPixels[0], transparentPixels[1]);
2373 }
2374 *lastEmpty = pattIdx;
2375 }
2376}
2377
2378/* A tile's colour pair as the two words the nibble expansion wants: the background repeated, and
2379 what to flip in it where a pattern bit is set. Four pixels come out of one lookup and one xor,
2380 and it is the same expansion the F18A tile and text paths use rather than a second way of drawing
2381 the same thing. */
2382static inline void lockedFgBg(PICO9918_INST_ARG uint32_t fgbg[2], const uint8_t pal, const uint32_t colorByte)
2383{
2384 fgbg[0] = repeatedPalette(pal | tmsBgColor(tms9918, colorByte));
2385 fgbg[1] = fgbg[0] ^ repeatedPalette(pal | tmsFgColor(tms9918, colorByte));
2386}
2387
2388/** \brief generate a locked (plain TMS9918) tile row, straight into the scanline */
2389PICO9918_INLINE_HOT void
2390renderTileRowLocked(PICO9918_INST_ARG uint16_t rowNamesAddr, uint16_t colorTableAddr, uint8_t tileIndex,
2391 uint8_t pal, uint8_t pixels[TMS9918_PIXELS_X], const TileRowAddr* addr, const bool gm2,
2392 const bool mcm)
2393{
2394 const uint8_t* pattTableRow = addr->pattern;
2395 const uint8_t nameMask = addr->nameMask;
2396
2397 /* locked mode has no horizontal scroll, so 32 tiles cover the screen exactly */
2398 uint32_t numTiles = GRAPHICS_NUM_COLS;
2399
2400 uint32_t* pix32 = (uint32_t*)PICO9918_ASSUME_ALIGNED(pixels, 4);
2401 uint8_t* pattPtr = tms9918->vram.bytes + rowNamesAddr + tileIndex;
2402 uint32_t pattOffset = 0;
2403
2404 uint8_t lastPattIdx = 0;
2405 uint32_t pattByte = mcm ? 0xf0 : (uint8_t)pattTableRow[0];
2406 uint32_t lastColorByte = mcm ? (uint8_t)pattTableRow[0] : tms9918->vram.bytes[colorTableAddr];
2407 if (gm2)
2408 {
2409 lastColorByte = PICO9918_LAYER_SUB(tms9918, PICO9918_SUPPRESS_GM2_COLOUR, 0xf1, lastColorByte);
2410 pattByte = PICO9918_LAYER_SUB(tms9918, PICO9918_SUPPRESS_GM2_PATTERN, 0xff, pattByte);
2411 }
2412 uint32_t fgbg[2];
2413 lockedFgBg(PICO9918_INST fgbg, pal, lastColorByte);
2414
2415 while (numTiles--)
2416 {
2417 uint8_t pattIdx = *pattPtr++;
2418 if (gm2) pattIdx &= nameMask;
2419
2420 if (lastPattIdx != pattIdx)
2421 {
2422 lastPattIdx = pattIdx;
2423 pattOffset = lastPattIdx * PATTERN_BYTES;
2424 uint32_t colorByte =
2425 mcm ? pattTableRow[pattOffset]
2426 : tms9918->vram.bytes[colorTableAddr + (gm2 ? pattOffset : (pattIdx >> 3))];
2427 if (!mcm) pattByte = (uint8_t)pattTableRow[pattOffset];
2428 if (gm2)
2429 {
2430 colorByte = PICO9918_LAYER_SUB(tms9918, PICO9918_SUPPRESS_GM2_COLOUR, 0xf1, colorByte);
2431 pattByte = PICO9918_LAYER_SUB(tms9918, PICO9918_SUPPRESS_GM2_PATTERN, 0xff, pattByte);
2432 }
2433 if (lastColorByte != colorByte)
2434 {
2435 lastColorByte = colorByte;
2436 lockedFgBg(PICO9918_INST fgbg, pal, colorByte);
2437 }
2438 }
2439
2440 pix32[0] = fgbg[0] ^ (fgbg[1] & maskExpandNibbleToWordRev[pattByte >> 4]);
2441 pix32[1] = fgbg[0] ^ (fgbg[1] & maskExpandNibbleToWordRev[pattByte & 0x0f]);
2442 pix32 += 2;
2443 }
2444}
2445
2446#define LOCKED_ROW_CLONE(name, g, m) \
2447 static void __time_critical_func(name)(PICO9918_INST_ARG uint16_t rowNamesAddr, uint16_t colorTableAddr, \
2448 uint8_t tileIndex, uint8_t pal, uint8_t pixels[TMS9918_PIXELS_X], \
2449 const TileRowAddr* addr) \
2450 { \
2451 renderTileRowLocked(PICO9918_INST rowNamesAddr, colorTableAddr, tileIndex, pal, pixels, addr, g, m); \
2452 }
2453
2454LOCKED_ROW_CLONE(rowLockedGm1, false, false)
2455LOCKED_ROW_CLONE(rowLockedGm2, true, false)
2456LOCKED_ROW_CLONE(rowLockedMcm, false, true)
2457
2458/* one F18A tile row into the layer buffer.
2459 *
2460 * isTile2, ecm and gm2 arrive as literals from the wrappers below, so both per-tile switch chains
2461 * and every layer test inside the tile path fold away at compile time, and each wrapper gets
2462 * its own body. The attribute is belt and braces: Priv.h already redefines inline as
2463 * __force_inline for Pico builds.
2464 *
2465 * Everything mode-specific reaches this loop through `addr`, which the caller fills once per layer
2466 * per scanline. Only two things do not fold into it and so ride the clone instead: the name mask,
2467 * which Graphics II applies and Graphics I does not, and Graphics II's colour address, which
2468 * indexes by tile row where Graphics I indexes by a group of eight names.
2469 */
2470#define TILE_ROW_PARAMS \
2471 PICO9918_INST_ARG const bool hpSize, uint16_t rowNamesAddr, uint16_t colorTableAddr, uint8_t tileIndex, \
2472 uint8_t startPattBit, const bool attrPerPos, uint8_t pal, const bool alwaysOnTop, \
2473 const TileRowAddr *addr
2474#define TILE_ROW_ARGS \
2475 PICO9918_INST hpSize, rowNamesAddr, colorTableAddr, tileIndex, startPattBit, attrPerPos, pal, alwaysOnTop, \
2476 addr
2477
2478PICO9918_INLINE_HOT void renderTileRow(TILE_ROW_PARAMS, const bool isTile2,
2479 const uint32_t ecm, const bool gm2, const bool mcm)
2480{
2481 uint32_t xPos = 0;
2482 uint32_t lastEmpty = -1;
2483
2484 const int32_t flipY = addr->flipY;
2485 const uint8_t* patternTable = addr->pattern;
2486 const uint8_t nameMask = addr->nameMask;
2487 uint8_t* targetBuffer = isTile2 ? tms9918->tileLayer2Buffer : tms9918->tileLayer1Buffer;
2488
2489 uint32_t numTiles = GRAPHICS_NUM_COLS + (startPattBit != 0);
2490
2491 uint32_t ecmColorOffset = 0, ecmColorMask = 0, ecmOffset = 0;
2492 if (ecm)
2493 {
2494 ecmColorOffset = (ecm == 3) ? 2 : ecm;
2495 ecmColorMask = (ecm == 3) ? 0x0e : 0x0f;
2496 ecmOffset = 0x800 >> ((TMS_REGISTER(tms9918, PICO9918_REG_PAGE_SIZE) & PICO9918_R29_TILE_STRIDE) >> 2);
2497 pal = (ecm == 1) ? (pal & 0x20) : 0;
2498 }
2499
2500 while (numTiles)
2501 {
2502 uint32_t run = GRAPHICS_NUM_COLS - tileIndex;
2503 if (run > numTiles) run = numTiles;
2504 numTiles -= run;
2505
2506 const uint8_t* names = tms9918->vram.bytes + rowNamesAddr + tileIndex;
2507
2508 while (run--)
2509 {
2510 uint8_t pattIdx = *names++;
2511 if (gm2 || ecm) pattIdx &= nameMask;
2512 if (ecm)
2513 {
2514 renderEcmTileToAlignedBuffer(PICO9918_INST targetBuffer, xPos, startPattBit, pattIdx, patternTable,
2515 colorTableAddr, ecm, ecmOffset, ecmColorMask, ecmColorOffset, pal,
2516 attrPerPos, flipY, tileIndex, &lastEmpty, isTile2, alwaysOnTop);
2517 }
2518 else
2519 {
2520 renderEcm0Tile(PICO9918_INST targetBuffer, xPos, pattIdx, patternTable, colorTableAddr, pal,
2521 isTile2, gm2, mcm);
2522 }
2523 ++tileIndex;
2524 xPos += 8;
2525 }
2526
2527 if (hpSize)
2528 {
2529 /* the attribute address gained the page bit by addition, so it has to lose it the same way */
2530 rowNamesAddr ^= 0x400;
2531 if (attrPerPos) colorTableAddr = (colorTableAddr + ((rowNamesAddr & 0x400) << 1) - 0x400) & VRAM_MASK;
2532 }
2533 tileIndex = 0;
2534 }
2535}
2536
2537#define TILE_ROW_CLONE(name, t2, e, g, m) \
2538 static void __time_critical_func(name)(TILE_ROW_PARAMS) \
2539 { \
2540 renderTileRow(TILE_ROW_ARGS, t2, e, g, m); \
2541 }
2542
2543TILE_ROW_CLONE(rowT1Ecm0, false, 0, false, false)
2544TILE_ROW_CLONE(rowT1Ecm1, false, 1, false, false)
2545TILE_ROW_CLONE(rowT1Ecm2, false, 2, false, false)
2546TILE_ROW_CLONE(rowT1Ecm3, false, 3, false, false)
2547TILE_ROW_CLONE(rowT1Gm2, false, 0, true, false)
2548TILE_ROW_CLONE(rowT1Mcm, false, 0, false, true)
2549TILE_ROW_CLONE(rowT2Ecm0, true, 0, false, false)
2550TILE_ROW_CLONE(rowT2Ecm1, true, 1, false, false)
2551TILE_ROW_CLONE(rowT2Ecm2, true, 2, false, false)
2552TILE_ROW_CLONE(rowT2Ecm3, true, 3, false, false)
2553TILE_ROW_CLONE(rowT2Gm2, true, 0, true, false)
2554TILE_ROW_CLONE(rowT2Mcm, true, 0, false, true)
2555
2556/* [layer][ecm], with slot 4 the Graphics II ECM0 body: ECM composes with mode through the plane 1
2557 address alone, so ECM1-3 needs no mode of its own */
2558#define TILE_ROW_GM2 4
2559#define TILE_ROW_MCM 5
2560static void (*const tileRowClones[2][6])(TILE_ROW_PARAMS) = {
2561 {rowT1Ecm0, rowT1Ecm1, rowT1Ecm2, rowT1Ecm3, rowT1Gm2, rowT1Mcm},
2562 {rowT2Ecm0, rowT2Ecm1, rowT2Ecm2, rowT2Ecm3, rowT2Gm2, rowT2Mcm}};
2563
2564/**
2565 * \brief generate a tile mode scanline for either T1 or T2 layer
2566 *
2567 * Inlined into both callers on purpose, not by the compiler's judgement: only there does `config`
2568 * fold to a constant, and every field it reads is otherwise a load on a hot line. It sits near
2569 * the size gcc stops at, so a statement added here has twice silently cost the board a layer.
2570 */
2571PICO9918_INLINE_HOT void f18a_tile_layer_scan_line(PICO9918_INST_ARG uint16_t y,
2572 const TileLayerConfig* config, const bool blend)
2573{
2574 const uint32_t ecm = (TMS_REGISTER(tms9918, PICO9918_REG_ENHANCED1) & PICO9918_R49_ECM_TILE) >> 4;
2575 const bool gm2 = pico9918_cached_mode == TMS_MODE_GRAPHICS_II;
2576 const bool mcm = pico9918_cached_mode == TMS_MODE_MULTICOLOR;
2577 const bool wide = TEXT80_WIDE_ROW;
2578 const bool text = wide || pico9918_cached_mode == TMS_MODE_TEXT;
2579 const uint8_t textCols = wide ? TEXT80_NUM_COLS : TEXT_NUM_COLS;
2580
2581 /* text takes its colour per cell at ECM0 too; a graphics mode there does not (D6) */
2582 const bool attrPerPos =
2583 (text || ecm) && (TMS_REGISTER(tms9918, PICO9918_REG_ENHANCED2) & PICO9918_R50_POS_ATTR);
2584
2585 TileRowAddr addr;
2586 uint16_t rowNamesAddr, colorTableAddr;
2587 tileLayerAddr(PICO9918_INST y, config, text ? textCols : GRAPHICS_NUM_COLS, 0x0f, text, gm2, mcm, attrPerPos,
2588 &addr, &rowNamesAddr, &colorTableAddr);
2589
2590 const uint8_t pal = (TMS_REGISTER(tms9918, PICO9918_REG_PALETTE_SELECT) & config->paletteMask)
2591 << config->paletteShift;
2592
2593 const bool alwaysOnTop =
2594 config->priorityReg ? !(TMS_REGISTER(tms9918, config->priorityReg) & config->priorityMask) : false;
2595
2596 if (text)
2597 {
2598 const uint8_t fixed = (tmsMainFgColor(tms9918) << 4) | tmsMainBgColor(tms9918);
2599
2600 /* ECM1-3 takes the attribute table by name; only ECM0 has an fg/bg pair to fall back on */
2601 const uint8_t* colors = (attrPerPos || ecm) ? tms9918->vram.bytes + colorTableAddr : &fixed;
2602 uint8_t* dest = (config->isTile2 ? tms9918->tileLayer2Buffer : tms9918->tileLayer1Buffer) +
2603 (wide ? TEXT80_PADDING_PX : TEXT_PADDING_PX);
2604 const uint32_t scroll = textScrollSplit(TMS_REGISTER(tms9918, config->startPattReg), wide);
2605
2606#if PICO9918_TEXT80_8BPP
2607 if (blend)
2608 {
2609 /* into layer 1's line, at the offset layer 2's own scroll puts it there */
2610 dest = tms9918->tileLayer1Buffer + TEXT80_PADDING_PX +
2611 textPixelOffset(TMS_REGISTER(tms9918, PICO9918_REG_T1_HSCROLL), true) -
2612 TEXT_SCROLL_OFFSET(scroll);
2613 text80RowT2Blend(PICO9918_INST tms9918->vram.bytes + rowNamesAddr, &addr, colors, attrPerPos, pal,
2614 dest, scroll, alwaysOnTop);
2615 return;
2616 }
2617#endif
2618
2619#if PICO9918_TEXT80_8BPP
2620 if (wide)
2621 {
2622 text80RowClones[config->isTile2][ecm](PICO9918_INST tms9918->vram.bytes + rowNamesAddr, &addr,
2623 colors, attrPerPos, pal, dest, scroll, alwaysOnTop);
2624 return;
2625 }
2626#endif
2627 textRowClones[config->isTile2][ecm](PICO9918_INST tms9918->vram.bytes + rowNamesAddr, &addr, colors,
2628 attrPerPos, pal, dest, scroll, alwaysOnTop);
2629 return;
2630 }
2631
2632 const uint8_t startPattBit = TMS_REGISTER(tms9918, config->startPattReg) & 0x07;
2633 const uint8_t tileIndex = (TMS_REGISTER(tms9918, config->startPattReg) >> 3);
2634 const bool hpSize = TMS_REGISTER(tms9918, PICO9918_REG_PAGE_SIZE) & config->hpSizeMask;
2635
2636 uint32_t slot = ecm;
2637 if (!ecm) slot = gm2 ? TILE_ROW_GM2 : (mcm ? TILE_ROW_MCM : 0);
2638
2639 /* two indirect calls per scanline buys a body per (isTile2, ecm, mode) with no per-tile dispatch */
2640 tileRowClones[config->isTile2][slot](PICO9918_INST hpSize, rowNamesAddr, colorTableAddr, tileIndex,
2641 startPattBit, attrPerPos, pal, alwaysOnTop, &addr);
2642}
2643
2644/** \brief generate a Graphics I mode scanline for the T1 layer */
2645static void __time_critical_func(f18a_tile1_scan_line)(PICO9918_INST_ARG uint16_t y)
2646{
2647 f18a_tile_layer_scan_line(PICO9918_INST y, &T1_CONFIG, false);
2648}
2649
2650/** \brief generate a Graphics I mode scanline for the T2 layer */
2651static void __time_critical_func(f18a_tile2_scan_line)(PICO9918_INST_ARG uint16_t y, const bool blend)
2652{
2653 f18a_tile_layer_scan_line(PICO9918_INST y, &T2_CONFIG, blend);
2654}
2655
2656static bool underLayer = false;
2657
2658/* Which buffer holds the finished line. Normally the one the caller passed, which the composite
2659 merges into. Where there is nothing to arbitrate it is tile layer 1's own buffer, and the merged
2660 line is then neither written nor read back - so the caller must ask rather than assume, which
2661 `pico9918_line_source` is for. */
2662const uint8_t* pico9918_cached_line_source = 0;
2663/**
2664 * \brief generate an F18A bitmap layer scanline
2665 *
2666 * INLINE: so will be different versions generated, depending on hard-coded (or known at compile-time) arguments
2667 */
2668PICO9918_INLINE_HOT bool renderBitmapLayerBody(PICO9918_INST_ARG uint16_t y, bool opaque, const uint8_t width,
2669 const uint16_t addr, const uint8_t bmlCtl,
2670 uint8_t pixels[TMS9918_PIXELS_X], const bool wide,
2671 const bool intoTile1)
2672{
2673 // written over T1's own buffer the layer is already above it, so the row mask arbitrates nothing
2674 bool writeMask = (bmlCtl & 0x40) && !intoTile1;
2675 underLayer = !(bmlCtl & 0x40);
2676
2677 bool returnVal = true;
2678
2679 if (writeMask && opaque && (width == 64))
2680 {
2681 for (int i = 0; i < TMS9918_PIXELS_X / 32; ++i) rowMasks.rowBits[i] = -1;
2682 writeMask = false;
2683 returnVal = false;
2684 }
2685
2686 uint32_t currentMask = 0;
2687 uint8_t xPos = TMS_REGISTER(tms9918, PICO9918_REG_BML_X);
2688
2689 if (bmlCtl & 0x10) // fat 4bpp pixels?
2690 {
2691 const uint8_t colorMask = 0xf0;
2692 const uint8_t colorOffset = 4;
2693 const uint8_t colorCount = 2;
2694 const uint8_t colorSize = 4;
2695 uint32_t maskPixelMask = 0x3u << 30;
2696 uint32_t maskX = xPos;
2697
2698 uint8_t pal = (bmlCtl & 0xc) << 2;
2699
2700 for (int xOff = 0; xOff < width; ++xOff)
2701 {
2702 uint8_t data = tms9918->vram.bytes[addr + xOff];
2703 for (int sp = 0; sp < colorCount; ++sp)
2704 {
2705 uint8_t color = (data & colorMask);
2706 if (opaque || color)
2707 {
2708 uint8_t finalColour = pal | (color >> colorOffset);
2709 if (wide)
2710 {
2711 const uint16_t pair = (uint16_t)(finalColour | (finalColour << 8));
2712 *(uint16_t*)(pixels + xPos * 2) = pair;
2713 *(uint16_t*)(pixels + xPos * 2 + 2) = pair;
2714 }
2715 else
2716 {
2717 pixels[xPos] = finalColour;
2718 pixels[xPos + 1] = finalColour;
2719 }
2720 currentMask |= maskPixelMask;
2721 }
2722 xPos += 2;
2723 data <<= colorSize;
2724 maskPixelMask >>= 2;
2725 }
2726 if (writeMask && !maskPixelMask && currentMask)
2727 {
2728 tmsTestRowBitsMask(maskX, currentMask, 32, true, false, false);
2729 maskX = xPos;
2730 maskPixelMask = 0x3u << 30;
2731 currentMask = 0;
2732 }
2733 }
2734 if (writeMask && currentMask)
2735 {
2736 tmsTestRowBitsMask(maskX, currentMask, xPos - maskX, true, false, false);
2737 }
2738 }
2739 else // regular 2bpp pixels
2740 {
2741 const uint8_t colorMask = 0xc0;
2742 const uint8_t colorOffset = 6;
2743 const uint8_t colorCount = 4;
2744 const uint8_t colorSize = 2;
2745 uint32_t maskPixelMask = 0x1u << 31;
2746 uint32_t maskX = xPos;
2747
2748 uint8_t pal = (bmlCtl & 0xf) << 2;
2749
2750 if (opaque && !wide && ((xPos & 3) == 0))
2751 {
2752 /* LOAD-BEARING: an aligned start stays aligned because the layer advances four pixels a
2753 * byte and the row is 256 wide, so no word store ever straddles xPos wrapping. That is
2754 * what removes the head/tail an unaligned block expansion would need on Cortex-M0+. */
2755 const uint32_t palQuad = repeatedPalette(pal);
2756 uint32_t* const quadPixels = (uint32_t*)pixels;
2757 uint32_t quadOffset = xPos >> 2;
2758
2759 for (int xOff = 0; xOff < width; ++xOff)
2760 {
2761 const uint8_t data = tms9918->vram.bytes[addr + xOff];
2762 quadPixels[quadOffset & (TMS9918_PIXELS_X / 4 - 1)] =
2763 (bmlExpand2bpp[data >> 4] | ((uint32_t)bmlExpand2bpp[data & 0x0f] << 16)) | palQuad;
2764 ++quadOffset;
2765 }
2766
2767 if (writeMask)
2768 {
2769 /* every pixel is opaque, so a whole group is a full mask and only the tail is partial */
2770 uint32_t left = (uint32_t)width * colorCount;
2771 while (left >= 32)
2772 {
2773 tmsTestRowBitsMask(maskX, 0xffffffffu, 32, true, false, false);
2774 maskX = (maskX + 32) & 0xff;
2775 left -= 32;
2776 }
2777 if (left) tmsTestRowBitsMask(maskX, ~0u << (32 - left), left, true, false, false);
2778 }
2779 return returnVal;
2780 }
2781
2782 for (int xOff = 0; xOff < width; ++xOff)
2783 {
2784 uint8_t data = tms9918->vram.bytes[addr + xOff];
2785 for (int sp = 0; sp < colorCount; ++sp)
2786 {
2787 uint8_t color = (data & colorMask);
2788 if (opaque || color)
2789 {
2790 const uint8_t v = pal | (color >> colorOffset);
2791 if (wide)
2792 *(uint16_t*)(pixels + xPos * 2) = (uint16_t)(v | (v << 8));
2793 else
2794 pixels[xPos] = v;
2795 currentMask |= maskPixelMask;
2796 }
2797 ++xPos;
2798 data <<= colorSize;
2799 maskPixelMask >>= 1;
2800 }
2801
2802 if (writeMask && !maskPixelMask && currentMask)
2803 {
2804 tmsTestRowBitsMask(maskX, currentMask, 32, true, false, false);
2805 maskX = xPos;
2806 maskPixelMask = 0x1u << 31;
2807 currentMask = 0;
2808 }
2809 }
2810 if (writeMask && currentMask)
2811 {
2812 tmsTestRowBitsMask(maskX, currentMask, xPos - maskX, true, false, false);
2813 }
2814 }
2815 return returnVal;
2816}
2817
2818/* One body per line width, for the reason spriteGridWord gives: the layer is on the 256-pixel grid
2819 whatever the mode, so a wide row draws each of its pixels twice. The F18A does the same - see the
2820 text2 case in f18a_tiles.vhd, which halves the layer's pixel clock rather than repeating it. */
2821#if PICO9918_TEXT80_8BPP
2822static EMITTER_NOINLINE bool __time_critical_func(renderBitmapLayer80)(
2823 PICO9918_INST_ARG uint16_t y, bool opaque, const uint8_t width, const uint16_t addr, const uint8_t bmlCtl,
2824 uint8_t pixels[TMS9918_PIXELS_X], const bool intoTile1)
2825{
2826 return renderBitmapLayerBody(PICO9918_INST y, opaque, width, addr, bmlCtl, pixels, true, intoTile1);
2827}
2828#endif
2829
2830static inline bool __time_critical_func(renderBitmapLayer40)(PICO9918_INST_ARG uint16_t y, bool opaque,
2831 const uint8_t width, const uint16_t addr,
2832 const uint8_t bmlCtl,
2833 uint8_t pixels[TMS9918_PIXELS_X])
2834{
2835 return renderBitmapLayerBody(PICO9918_INST y, opaque, width, addr, bmlCtl, pixels, false, false);
2836}
2837
2838/** \brief generate an F18A bitmap layer scanline */
2839static bool __time_critical_func(bitmap_layer_scan_line)(PICO9918_INST_ARG uint16_t y,
2840 uint8_t pixels[TMS9918_PIXELS_X],
2841 const bool intoTile1)
2842{
2843 /* bml enabled? */
2844 const uint8_t bmlCtl = TMS_REGISTER(tms9918, PICO9918_REG_BML_CONTROL);
2845 if (!(bmlCtl & 0x80) || !PICO9918_DRAWS(tms9918, PICO9918_SUPPRESS_BITMAP)) return true;
2846
2847 /* bml on this scanline? */
2848 const uint8_t top = TMS_REGISTER(tms9918, PICO9918_REG_BML_TOP_ROW);
2849 if (top > y) return true;
2850
2851 y -= top;
2852 if (y >= TMS_REGISTER(tms9918, PICO9918_REG_BML_HEIGHT)) return true;
2853
2854 /* row stride in bytes, four pixels each, rounded up so every row starts on a byte */
2855 const uint8_t bmlWidth = TMS_REGISTER(tms9918, PICO9918_REG_BML_WIDTH);
2856 const uint8_t width = bmlWidth ? ((bmlWidth + 3) >> 2) : 64;
2857 const uint16_t addr = (TMS_REGISTER(tms9918, PICO9918_REG_BML_BASE) << 6) + (y * width);
2858
2859 const bool opaque = !(bmlCtl & 0x20);
2860
2861#if PICO9918_TEXT80_8BPP
2862 if (TEXT80_WIDE_ROW)
2863 return renderBitmapLayer80(PICO9918_INST y, opaque, width, addr, bmlCtl, pixels, intoTile1);
2864#endif
2865 return renderBitmapLayer40(PICO9918_INST y, opaque, width, addr, bmlCtl, pixels);
2866}
2867
2868/* One 32-pixel chunk of the composite, four pixels at a time. Selecting a layer per pixel is a byte
2869 * mask over two words, and the nibble-to-word lookup the tile emitters use is exactly that mask - so
2870 * a chunk is eight selects rather than thirty-two.
2871 *
2872 * `hasSprites` arrives as a literal, so a chunk no sprite touches compiles to a plain store and one
2873 * that a sprite crosses to a merge, with no test in either. The caller can only use this where both
2874 * layer pointers are word-aligned, which a scroll that is not a multiple of four denies.
2875 */
2876static inline void compositeChunkWide(uint32_t* __restrict pix32, const uint32_t* __restrict l1,
2877 const uint32_t* __restrict l2, uint32_t sel, uint32_t open,
2878 const bool hasSprites)
2879{
2880 for (int i = 0; i < 8; ++i)
2881 {
2882 const uint32_t layers = maskExpandNibbleToWordRev[sel >> 28];
2883 uint32_t merged = l1[i] ^ ((l1[i] ^ l2[i]) & layers);
2884
2885 if (hasSprites)
2886 {
2887 const uint32_t old = pix32[i];
2888 merged = old ^ ((old ^ merged) & maskExpandNibbleToWordRev[open >> 28]);
2889 open <<= 4;
2890 }
2891
2892 pix32[i] = merged;
2893 sel <<= 4;
2894 }
2895}
2896
2897/* the same chunk with a bitmap layer under the tiles, which is the one case that has to stay per
2898 * pixel: zero means transparent here, so every byte needs its own test and there is no word to
2899 * select whole. `hasSprites` is the same literal the other two chunks take, and it is worth more
2900 * here than anywhere - without it every pixel of every chunk tests and shifts a mask that is zero.
2901 */
2902static inline void compositeChunkUnder(uint8_t* __restrict pixels, const uint8_t* __restrict layer1,
2903 const uint8_t* __restrict layer2, uint32_t mask, uint32_t spriteMask,
2904 const bool hasSprites)
2905{
2906 for (int i = 0; i < 8; ++i)
2907 {
2908#define UNDER_PIXEL(n) \
2909 if (!hasSprites || !(spriteMask & MASK_NEXT_PIXEL)) \
2910 { \
2911 const uint8_t pixel = (mask & MASK_NEXT_PIXEL) ? layer2[n] : layer1[n]; \
2912 if (pixel) pixels[n] = pixel; \
2913 } \
2914 mask <<= 1; \
2915 if (hasSprites) spriteMask <<= 1;
2916
2917 UNDER_PIXEL(0)
2918 UNDER_PIXEL(1)
2919 UNDER_PIXEL(2)
2920 UNDER_PIXEL(3)
2921#undef UNDER_PIXEL
2922
2923 pixels += 4;
2924 layer1 += 4;
2925 layer2 += 4;
2926 }
2927}
2928
2929/* the same chunk a byte at a time, for a scroll that leaves a layer unaligned */
2930static inline void compositeChunkBytes(uint8_t* __restrict pixels, const uint8_t* __restrict layer1,
2931 const uint8_t* __restrict layer2, uint32_t mask, uint32_t spriteMask,
2932 const bool hasSprites)
2933{
2934 for (int i = 0; i < 8; ++i)
2935 {
2936#define MIXED_PIXEL(n) \
2937 if (!hasSprites || !(spriteMask & MASK_NEXT_PIXEL)) \
2938 { \
2939 pixels[n] = (mask & MASK_NEXT_PIXEL) ? layer2[n] : layer1[n]; \
2940 } \
2941 mask <<= 1; \
2942 if (hasSprites) spriteMask <<= 1;
2943
2944 MIXED_PIXEL(0)
2945 MIXED_PIXEL(1)
2946 MIXED_PIXEL(2)
2947 MIXED_PIXEL(3)
2948#undef MIXED_PIXEL
2949
2950 pixels += 4;
2951 layer1 += 4;
2952 layer2 += 4;
2953 }
2954}
2955
2956/* Sprites and the bitmap layer are on the 256-pixel grid whatever the mode, so a wide
2957 row's chunk of 32 tile pixels is only 16 of theirs: half a mask word, doubled bit by bit through
2958 the table the magnified sprite emitter already uses. `wide` is a clone parameter rather than a
2959 count, because that read is inside the chunk loop. */
2960static inline uint32_t spriteGridWord(const uint32_t maskWord, const BitMask mask, const bool wide)
2961{
2962 if (!wide) return mask[maskWord];
2963
2964 const uint32_t half = (maskWord & 1) ? (mask[maskWord >> 1] & 0xffff) : (mask[maskWord >> 1] >> 16);
2965 return ((uint32_t)doubledBits[half >> 8] << 16) | doubledBits[half & 0xff];
2966}
2967
2968/* A line the zero-copy path would have handed out whole, but for the sprites drawn into pixels[].
2969 * Rather than copy the tile line over them and merge the sprites back, put the sprites into the
2970 * tile line and hand that out: a word no sprite touches then costs nothing at all, and there is
2971 * no second layer or selection mask to read for the ones that do. Every other zero-copy condition
2972 * already holds here, so nothing else in pixels[] has to survive.
2973 */
2974static inline void overlaySpritesOnTile1(uint32_t* __restrict dst, const uint32_t* __restrict src,
2975 const bool wide)
2976{
2977 const uint32_t maskWords = (wide ? SCANLINE_BYTES_MAX : TMS9918_PIXELS_X) / 32;
2978
2979 for (uint32_t maskWord = 0; maskWord < maskWords; ++maskWord, dst += 8, src += 8)
2980 {
2981 uint32_t open = spriteGridWord(maskWord, rowMasks.rowSpriteBits, wide);
2982
2983 for (int i = 0; open; ++i, open <<= 4)
2984 {
2985 const uint32_t nibble = open >> 28;
2986 if (nibble)
2987 {
2988 const uint32_t take = maskExpandNibbleToWordRev[nibble];
2989 dst[i] = dst[i] ^ ((dst[i] ^ src[i]) & take);
2990 }
2991 }
2992 }
2993}
2994
2995PICO9918_INLINE_HOT void
2996compositeAlignedBody(PICO9918_INST_ARG uint8_t pixels[TMS9918_PIXELS_X], const int t1Scroll, const int t2Scroll,
2997 const bool wide)
2998{
2999 uint8_t* layer1 = tms9918->tileLayer1Buffer + t1Scroll;
3000 uint8_t* layer2 = tms9918->tileLayer2Buffer + t2Scroll;
3001 uint32_t* selectionMask = tms9918->layerSelectionMask;
3002 const uint32_t maskWords = (wide ? SCANLINE_BYTES_MAX : TMS9918_PIXELS_X) / 32;
3003
3004
3005 const bool wordAligned = ((t1Scroll | t2Scroll) & 3) == 0;
3006 const uint32_t chunkCount = wordAligned ? 32 / sizeof(uint32_t) : 32;
3007 PICO9918_COPY_SET_WIDTH(PICO9918_COPY, wordAligned);
3008
3009 // Process in 32-pixel chunks (1 mask word at a time)
3010 for (uint32_t maskWord = 0; maskWord < maskWords; maskWord++)
3011 {
3012 uint32_t mask = selectionMask[maskWord];
3013
3014 /* a priority bitmap layer wins over T1 only - T2 still draws over it */
3015 uint32_t spriteMask = spriteGridWord(maskWord, rowMasks.rowSpriteBits, wide) |
3016 (spriteGridWord(maskWord, rowMasks.rowBits, wide) & ~mask);
3017
3018 if (spriteMask == 0xffffffffu)
3019 {
3020 layer1 += 32;
3021 layer2 += 32;
3022 pixels += 32;
3023 continue;
3024 }
3025
3026 if (!underLayer && !spriteMask)
3027 {
3028 if (mask == 0)
3029 {
3030 // All T1 pixels - use DMA copy for speed
3031 PICO9918_COPY_WAIT(PICO9918_COPY);
3032 PICO9918_COPY_SET_SRC(PICO9918_COPY, layer1);
3033 PICO9918_COPY_SET_DST(PICO9918_COPY, pixels);
3034 PICO9918_COPY_TRIGGER(PICO9918_COPY, chunkCount);
3035 pixels += 32;
3036 layer1 += 32;
3037 layer2 += 32;
3038 continue;
3039 }
3040 else if (mask == 0xffffffffu)
3041 {
3042 // All T2 pixels - use DMA copy for speed
3043 PICO9918_COPY_WAIT(PICO9918_COPY);
3044 PICO9918_COPY_SET_SRC(PICO9918_COPY, layer2);
3045 PICO9918_COPY_SET_DST(PICO9918_COPY, pixels);
3046 PICO9918_COPY_TRIGGER(PICO9918_COPY, chunkCount);
3047 pixels += 32;
3048 layer1 += 32;
3049 layer2 += 32;
3050 continue;
3051 }
3052 }
3053
3054
3055 // mixed - process 4 pixels at a time with individual byte access
3056 if (underLayer)
3057 {
3058 if (spriteMask)
3059 compositeChunkUnder(pixels, layer1, layer2, mask, spriteMask, true);
3060 else
3061 compositeChunkUnder(pixels, layer1, layer2, mask, 0, false);
3062
3063 pixels += 32;
3064 layer1 += 32;
3065 layer2 += 32;
3066 }
3067 else if (wordAligned)
3068 {
3069 uint32_t* pix32 = (uint32_t*)PICO9918_ASSUME_ALIGNED(pixels, 4);
3070 const uint32_t* l1 = (const uint32_t*)PICO9918_ASSUME_ALIGNED(layer1, 4);
3071 const uint32_t* l2 = (const uint32_t*)PICO9918_ASSUME_ALIGNED(layer2, 4);
3072
3073 if (spriteMask)
3074 compositeChunkWide(pix32, l1, l2, mask, ~(uint32_t)spriteMask, true);
3075 else
3076 compositeChunkWide(pix32, l1, l2, mask, 0, false);
3077
3078 pixels += 32;
3079 layer1 += 32;
3080 layer2 += 32;
3081 }
3082 else
3083 {
3084 if (spriteMask)
3085 compositeChunkBytes(pixels, layer1, layer2, mask, spriteMask, true);
3086 else
3087 compositeChunkBytes(pixels, layer1, layer2, mask, 0, false);
3088
3089 pixels += 32;
3090 layer1 += 32;
3091 layer2 += 32;
3092 }
3093 }
3094}
3095
3096
3097/* One body per line width. The chunk loop reads the sprite grid on every pass, so the rate it reads
3098 it at has to be a literal there rather than a value carried in - which is also what keeps the
3099 256-pixel modes paying nothing for the tier. */
3100#if PICO9918_TEXT80_8BPP
3101static EMITTER_NOINLINE
3102#else
3103/* one width, one caller, so it inlines: standing it out of line costs RP2040 two-layer lines dearly */
3104static inline
3105#endif
3106 void __time_critical_func(compositeAligned40)(PICO9918_INST_ARG uint8_t pixels[TMS9918_PIXELS_X],
3107 const int t1Scroll, const int t2Scroll)
3108{
3109 compositeAlignedBody(PICO9918_INST pixels, t1Scroll, t2Scroll, false);
3110}
3111
3112#if PICO9918_TEXT80_8BPP
3113static EMITTER_NOINLINE void
3114__time_critical_func(compositeAligned80)(PICO9918_INST_ARG uint8_t pixels[TMS9918_PIXELS_X], const int t1Scroll,
3115 const int t2Scroll)
3116{
3117 compositeAlignedBody(PICO9918_INST pixels, t1Scroll, t2Scroll, true);
3118}
3119#endif
3120
3121static inline void compositeAlignedTileBuffers(PICO9918_INST_ARG uint8_t pixels[TMS9918_PIXELS_X],
3122 const int t1Scroll, const int t2Scroll, const bool wide)
3123{
3124#if PICO9918_TEXT80_8BPP
3125 if (wide)
3126 {
3127 compositeAligned80(PICO9918_INST pixels, t1Scroll, t2Scroll);
3128 return;
3129 }
3130#endif
3131 compositeAligned40(PICO9918_INST pixels, t1Scroll, t2Scroll);
3132}
3133
3134/**
3135 * \brief Composite with tile layer 1 disabled (reg 0x32 bit 4). Layer 1 contributes no pixel at all, so
3136 * every position the selection mask leaves to it must keep whatever the backdrop, bitmap layer and
3137 * sprites already put there - the colour stage falls through to the backdrop for a transparent
3138 * merged pixel. Writing layer 1's buffer, or a zero buffer, would paint palette entry 0 instead.
3139 */
3140static PICO9918_NOINLINE void
3142 const int t2Scroll, const bool wide)
3143{
3144 const uint8_t* layer2 = tms9918->tileLayer2Buffer + t2Scroll;
3145 const uint32_t maskWords = (wide ? SCANLINE_BYTES_MAX : TMS9918_PIXELS_X) / 32;
3146
3147 for (uint32_t maskWord = 0; maskWord < maskWords; ++maskWord)
3148 {
3149 uint32_t mask =
3150 tms9918->layerSelectionMask[maskWord] & ~spriteGridWord(maskWord, rowMasks.rowSpriteBits, wide);
3151
3152 if (!mask) /* nothing of layer 2 reaches the screen here */
3153 {
3154 layer2 += 32;
3155 pixels += 32;
3156 continue;
3157 }
3158
3159 for (int i = 0; i < 8; ++i)
3160 {
3161 if (mask & MASK_NEXT_PIXEL) pixels[0] = layer2[0];
3162 mask <<= 1;
3163
3164 if (mask & MASK_NEXT_PIXEL) pixels[1] = layer2[1];
3165 mask <<= 1;
3166
3167 if (mask & MASK_NEXT_PIXEL) pixels[2] = layer2[2];
3168 mask <<= 1;
3169
3170 if (mask & MASK_NEXT_PIXEL) pixels[3] = layer2[3];
3171 mask <<= 1;
3172
3173 pixels += 4;
3174 layer2 += 4;
3175 }
3176 }
3177}
3178
3179/**
3180 * \brief A text row draws 240 of the 256 pixels and the composite copies all of them, so layer 1's buffer
3181 * has to carry the side borders itself. The row is emitted on cell boundaries and read back
3182 * `t1Scroll` pixels along, so the border moves with the scroll - and the cells that fall outside it
3183 * either side are overwritten here rather than never written.
3184 */
3185static void __time_critical_func(textRowBorder)(PICO9918_INST_ARG const int t1Scroll, const uint32_t padding,
3186 const uint32_t width)
3187{
3188 uint8_t* left = tms9918->tileLayer1Buffer + t1Scroll;
3189 uint8_t* right = left + width - padding;
3190
3191 for (uint32_t i = 0; i < padding; ++i)
3192 {
3193 left[i] = bg;
3194 right[i] = bg;
3195 }
3196}
3197
3198/** \brief generate a Graphics I or Graphics II mode scanline */
3199static uint8_t __time_critical_func(graphics_i_scan_line)(PICO9918_INST_ARG uint16_t y,
3200 uint8_t pixels[TMS9918_PIXELS_X])
3201{
3202 uint8_t tempStatus = 0;
3203
3204 /* locked or unlocked is decided once here, not re-tested per row */
3205 if (PICO9918_UNLOCKED(tms9918))
3206 {
3207 /* the background fill owns pixels[] until it completes */
3208 PICO9918_FILL32_WAIT(PICO9918_FILL_LINE);
3209
3210 const uint8_t bmlCtlReg = TMS_REGISTER(tms9918, PICO9918_REG_BML_CONTROL);
3211
3212 /* LOAD-BEARING: drawn over tile layer 1's buffer a priority layer is above T1 by construction,
3213 which is what lets the blend and the zero-copy line survive it. Each condition breaks that:
3214 an UNDER layer needs per-pixel arbitration, tile 1 off means that buffer is never read, and
3215 only a wide row doubles. Relax any of them and the layer is lost or lands under T1. */
3216 const bool bmlInTile1 = TEXT80_WIDE_ROW && (bmlCtlReg & PICO9918_R31_BML_ENABLE) &&
3217 (bmlCtlReg & 0x40) &&
3218 !(TMS_REGISTER(tms9918, PICO9918_REG_ENHANCED2) & PICO9918_R50_TILE1_OFF) &&
3219 PICO9918_DRAWS(tms9918, PICO9918_SUPPRESS_TILE1) &&
3220 PICO9918_DRAWS(tms9918, PICO9918_SUPPRESS_BITMAP);
3221
3222 bool writeMask = true;
3223 if (bmlInTile1)
3224 underLayer = false;
3225 else
3226 writeMask = bitmap_layer_scan_line(PICO9918_INST y, pixels, false);
3227
3228 const uint32_t transparent = underLayer ? 0 : bg;
3229 transparentPixels[0] = transparentPixels[1] = transparent;
3230 ecm0Palette[0x00] = ecm0Palette[0x10] = ecm0Palette[0x20] = ecm0Palette[0x30] = transparent;
3231
3232 tempStatus = pico9918_output_sprites(PICO9918_INST y, pixels);
3233
3234 if (writeMask) // bitmap layer completely masked it?
3235 {
3236 const bool wide = TEXT80_WIDE_ROW;
3237 const bool textRow = wide || pico9918_cached_mode == TMS_MODE_TEXT;
3238 const int t1Scroll = scrollOffset(TMS_REGISTER(tms9918, PICO9918_REG_T1_HSCROLL), textRow, wide);
3239 const int t2Scroll = scrollOffset(TMS_REGISTER(tms9918, PICO9918_REG_T2_HSCROLL), textRow, wide);
3240 const bool tile2Enabled = (TMS_REGISTER(tms9918, PICO9918_REG_ENHANCED1) & PICO9918_R49_TILE2_ENABLE) &&
3241 PICO9918_DRAWS(tms9918, PICO9918_SUPPRESS_TILE2);
3242 const bool tile1Enabled = !(TMS_REGISTER(tms9918, PICO9918_REG_ENHANCED2) & PICO9918_R50_TILE1_OFF) &&
3243 PICO9918_DRAWS(tms9918, PICO9918_SUPPRESS_TILE1);
3244
3245 const bool blend = wide && tile1Enabled && tile2Enabled &&
3246 !((TMS_REGISTER(tms9918, PICO9918_REG_ENHANCED1) & PICO9918_R49_ECM_TILE) >> 4) &&
3247 (bmlInTile1 || !(bmlCtlReg & PICO9918_R31_BML_ENABLE));
3248
3249 if (tile2Enabled && !blend)
3250 {
3252 tmsCopyAlignMask(tms9918->finalMask, tms9918->layerSelectionMask, t1Scroll - t2Scroll);
3253 }
3254
3255 if (tile1Enabled)
3256 {
3258 if (bmlInTile1)
3259 bitmap_layer_scan_line(PICO9918_INST y, tms9918->tileLayer1Buffer + t1Scroll, true);
3260 if (blend) f18a_tile2_scan_line(PICO9918_INST y, true);
3261 if (textRow)
3262 textRowBorder(PICO9918_INST t1Scroll, wide ? TEXT80_PADDING_PX : TEXT_PADDING_PX,
3263 wide ? SCANLINE_BYTES_MAX : TMS9918_PIXELS_X);
3264 }
3265
3266 if (tile2Enabled && !blend)
3267 tmsRestoreAlignMask(tms9918->layerSelectionMask, tms9918->finalMask, -t1Scroll, t2Scroll);
3268
3269 if (textRow && !blend)
3270 {
3271 const uint32_t pad = wide ? TEXT80_PADDING_PX : TEXT_PADDING_PX;
3272 const uint32_t end = pad + (wide ? TEXT80_NUM_COLS : TEXT_NUM_COLS) * TEXT_CHAR_WIDTH;
3273 tms9918->layerSelectionMask[0] &= 0xffffffffu >> pad;
3274 tms9918->layerSelectionMask[(end - 1) >> 5] &= ~(0xffffffffu >> (end & 0x1f));
3275 }
3276
3277 /* WARNING: the scroll test keeps the handed-out line word-aligned. A caller may read it a
3278 word at a time, and on Cortex-M0+ an unaligned word load HardFaults rather than running slow */
3279 if (tile1Enabled && (!tile2Enabled || blend) && !underLayer && !(t1Scroll & 3) &&
3280 (bmlInTile1 || !(bmlCtlReg & PICO9918_R31_BML_ENABLE)))
3281 {
3282 uint8_t* line = tms9918->tileLayer1Buffer + t1Scroll;
3283
3284 if (tms9918->scanlineHasSprites)
3285 overlaySpritesOnTile1((uint32_t*)PICO9918_ASSUME_ALIGNED(line, 4),
3286 (const uint32_t*)PICO9918_ASSUME_ALIGNED(pixels, 4), wide);
3287
3288 pico9918_cached_line_source = line;
3289 }
3290 else if (tile1Enabled)
3291 {
3292 compositeAlignedTileBuffers(PICO9918_INST pixels, t1Scroll, t2Scroll, wide);
3293 }
3294 else if (tile2Enabled)
3295 {
3296 compositeTile2OnlyBuffer(PICO9918_INST pixels, t2Scroll, wide);
3297 }
3298 /* both layers off: the backdrop, bitmap layer and sprites are already in pixels[] */
3299 }
3300 }
3301 else
3302 {
3303 const uint8_t tileY = y >> 3; /* which name table row (0 - 23)... or 29 */
3304
3305 /* address in name table at the start of this row */
3306 const uint16_t rowOffset = tileY * GRAPHICS_NUM_COLS;
3307 uint16_t rowNamesAddr = tmsNameTableAddr(tms9918) + rowOffset;
3308 uint16_t colorTableAddr = tmsColorTableAddr(tms9918);
3309
3310 const bool gm2 = pico9918_cached_mode == TMS_MODE_GRAPHICS_II;
3311 const bool mcm = pico9918_cached_mode == TMS_MODE_MULTICOLOR;
3312 TileRowAddr addr;
3313 tileRowAddr(PICO9918_INST y, y, TMS_REGISTER(tms9918, TMS_REG_COLOR_TABLE), gm2, mcm, &addr,
3314 &colorTableAddr);
3315
3316 PICO9918_FILL32_WAIT(PICO9918_FILL_LINE);
3317
3318 if (PICO9918_DRAWS(tms9918, PICO9918_SUPPRESS_TILE1))
3319 {
3320 if (gm2)
3321 rowLockedGm2(PICO9918_INST rowNamesAddr, colorTableAddr, 0, 0, pixels, &addr);
3322 else if (mcm)
3323 rowLockedMcm(PICO9918_INST rowNamesAddr, colorTableAddr, 0, 0, pixels, &addr);
3324 else
3325 rowLockedGm1(PICO9918_INST rowNamesAddr, colorTableAddr, 0, 0, pixels, &addr);
3326 }
3327
3328 tempStatus = pico9918_output_sprites(PICO9918_INST y, pixels);
3329 }
3330
3331 return tempStatus;
3332}
3333
3334/** \brief generate a scanline */
3335PICO9918_DLLEXPORT uint8_t __time_critical_func(pico9918_scan_line)(PICO9918_INST_ARG uint16_t y)
3336{
3337 uint8_t* const pixels = scanlineBuffer;
3338 uint8_t tempStatus = 0;
3339
3340 if (!lookupsReady) initLookups();
3341
3342 pico9918_mode_t currentCachedMode = tmsMode(tms9918);
3343 if (currentCachedMode != pico9918_cached_mode)
3344 {
3345 pico9918_cached_mode = currentCachedMode;
3346 tms9918->palDirty = 1;
3347 }
3348
3349 const uint8_t bgc = tmsMainBgColor(tms9918);
3350 const bool packedNibbles = pico9918_cached_mode == TMS_MODE_TEXT80 && !TEXT80_WIDE_ROW;
3351 bg = repeatedPalette(
3352 bgc |
3353 (packedNibbles ? bgc << 4
3354 : (TMS_REGISTER(tms9918, PICO9918_REG_PALETTE_SELECT) & PICO9918_R24_TILE1_PS) << 4));
3355#if PICO9918_TEXT80_8BPP
3356 /* a wide row is twice the line to fill, and the count is per mode rather than per build */
3357 PICO9918_FILL32_SET_COUNT(PICO9918_FILL_LINE, pico9918_line_bytes(PICO9918_INST_ONLY) / 4);
3358#endif
3359 PICO9918_FILL32_TRIGGER(PICO9918_FILL_LINE, pixels);
3360 pico9918_cached_line_source = pixels;
3361 underLayer = false;
3362
3363 bool dispActive = (TMS_REGISTER(tms9918, TMS_REG_1) & TMS_R1_DISP_ACTIVE) ||
3364 PICO9918_SUPPRESSED(tms9918, PICO9918_SUPPRESS_BLANKING);
3365
3366 if (dispActive)
3367 {
3368 /* the three row masks go out as one transfer; the instance masks below cover its latency */
3369 PICO9918_FILL32_TRIGGER(PICO9918_FILL_MASKS, &rowMasks);
3370
3371 for (int i = 0; i < SCANLINE_MASK_WORDS; ++i)
3372 {
3373 tms9918->layerSelectionMask[i] = 0; // Default to all T1 pixels
3374 tms9918->finalMask[i] = 0;
3375 }
3376 tms9918->scanlineHasSprites = false;
3377
3378 PICO9918_FILL32_WAIT(PICO9918_FILL_MASKS);
3379
3380 /* WARNING: the unreachable modes are deliberately not named, and there is no default.
3381 Either one makes the compiler stop assuming the value is in range, and it pays for a
3382 bounds check on every scanline. Suppressed rather than silenced so a consumer
3383 building these sources is not the one who sees it. */
3384#if defined(__GNUC__) || defined(__clang__)
3385#pragma GCC diagnostic push
3386#pragma GCC diagnostic ignored "-Wswitch"
3387#endif
3388 switch (pico9918_cached_mode)
3389 {
3390 case TMS_MODE_GRAPHICS_I:
3391 case TMS_MODE_GRAPHICS_II:
3392 case TMS_MODE_MULTICOLOR: tempStatus = graphics_i_scan_line(PICO9918_INST y, pixels); break;
3393
3394 case TMS_MODE_TEXT:
3395 case TMS_MODE_TEXT80:
3396 if (PICO9918_UNLOCKED(tms9918) && (pico9918_cached_mode == TMS_MODE_TEXT || TEXT80_WIDE_ROW))
3397 {
3398 tempStatus = graphics_i_scan_line(PICO9918_INST y, pixels);
3399 break;
3400 }
3401
3402 if (PICO9918_DRAWS(tms9918, PICO9918_SUPPRESS_TILE1)) text_scan_line(PICO9918_INST y, pixels);
3403 if (PICO9918_UNLOCKED(tms9918)) tempStatus = pico9918_output_sprites(PICO9918_INST y, pixels);
3404 break;
3405 }
3406#if defined(__GNUC__) || defined(__clang__)
3407#pragma GCC diagnostic pop
3408#endif
3409 }
3410
3411 /* pixels[] must be complete, and owned by nobody, when we return */
3412 PICO9918_FILL32_WAIT(PICO9918_FILL_LINE);
3413 PICO9918_COPY_WAIT(PICO9918_COPY);
3414
3415 return tempStatus;
3416}
3417
3418/** \brief return a register value - see the header for the locked-device aliasing */
3421{
3422 return TMS_REGISTER(tms9918, reg & tms9918->lockedMask); // was 0x07
3423}
3424
3425/** \brief return a status register value without the side effects of reading it */
3428{
3429 return TMS_STATUS(tms9918, reg & PICO9918_R15_STATUS_NUM);
3430}
3431
3433void __time_critical_func(pico9918_write_reg_value_impl)(PICO9918_INST_ARG uint8_t reg, uint8_t value)
3434{
3435 if (PICO9918_HAS(tms9918, PICO9918_FEAT_UNLOCK) && PICO9918_UNLOCK_REG(reg))
3436 {
3437 /* Recomputed on every write, so a redundant unlock is a no-op and anything else locks */
3438 const bool unlockValue = PICO9918_UNLOCK_VALUE(value);
3439 const bool unlocked = unlockValue && tms9918->unlockCount;
3440
3441 TMS_REGISTER(tms9918, PICO9918_REG_UNLOCK) = value; // through even when locked
3442 tms9918->unlockCount = unlockValue;
3443
3444 if (unlocked != tms9918->isUnlocked)
3445 {
3446 tms9918->isUnlocked = unlocked;
3447 tms9918->lockedMask = unlocked ? 0x3f : 0x07;
3448 tms9918->palDirty = 1;
3449 if (unlocked) TMS_REGISTER(tms9918, PICO9918_REG_MAX_SCAN_SPRITES) = MAX_SPRITES - 1;
3450 }
3451 }
3452 else
3453 {
3454 /* A write the personality never sees does not disturb the unlock counter either */
3455 if (((reg & ~tms9918->lockedMask) != 0x80) && PICO9918_M4(tms9918)) return;
3456
3457 tms9918->unlockCount = 0;
3458
3459 const int regIndex = reg & tms9918->lockedMask; // was 0x07
3460
3461 TMS_REGISTER(tms9918, regIndex) = value;
3462 if (regIndex < PICO9918_REG_STATUS_SELECT)
3463 {
3464 /* LOAD-BEARING: R0 and R1 hold the only register bits pico9918_interrupt_status_impl
3465 * reads, and regIndex is post-mask - a locked write to R25 lands on R1 and must
3466 * reconcile, so testing the byte the host sent would miss it. */
3467 if (regIndex <= TMS_REG_1) pico9918_write_reconcile_int_impl(PICO9918_INST_ONLY);
3468 return;
3469 }
3470
3471 if ((regIndex == PICO9918_REG_GPU_PC_LSB) ||
3472 ((regIndex == PICO9918_REG_GPU_CONTROL) && ((value & PICO9918_R56_GPU_RUN) == 0)))
3473 {
3474 tms9918->gpuAddress = ((TMS_REGISTER(tms9918, PICO9918_REG_GPU_PC_MSB) << 8) |
3475 TMS_REGISTER(tms9918, PICO9918_REG_GPU_PC_LSB)) &
3476 0xFFFE;
3477 if (regIndex == PICO9918_REG_GPU_PC_LSB)
3478 {
3479 TMS_REGISTER(tms9918, PICO9918_REG_GPU_CONTROL) = 0;
3480 tms9918->gpuStatus = 0; /* a new program, not a resumed one */
3481 tms9918->restart = 1;
3482 pico9918_gpu_service(PICO9918_INST_ONLY);
3483 }
3484 }
3485 else if ((regIndex == PICO9918_REG_GPU_CONTROL) && (value & PICO9918_R56_GPU_RUN))
3486 {
3487 tms9918->restart = 1;
3488 pico9918_gpu_service(PICO9918_INST_ONLY);
3489 }
3490 else if (regIndex == PICO9918_REG_FLASH_CONTROL &&
3491 PICO9918_HAS(tms9918, PICO9918_FEAT_CONFIG)) // firmware update
3492 {
3493 // b7 : 0 = idle: 1 = execute
3494 // b6 : 0 = verify: 1 = write
3495 // b5 - b0 : address to read firmware data (256 byte boundaries)
3496 // reads one UF2 frame (512 bytes)
3497 if (TMS_REGISTER(tms9918, PICO9918_REG_GPU_CONTROL) == 0)
3498 {
3499 TMS_STATUS(tms9918, PICO9918_SR_GPU) = 0x80; // set gpu processing flag
3500 tms9918->flash = 1;
3501 }
3502 else
3503 {
3504 TMS_STATUS(tms9918, PICO9918_SR_GPU) = 0x14; // error - busy
3505 }
3506 }
3507 else if (regIndex == PICO9918_REG_MAX_SCAN_SPRITES && value == 0)
3508 {
3509 TMS_REGISTER(tms9918, PICO9918_REG_MAX_SCAN_SPRITES) = MAX_SPRITES - 1;
3510 }
3511 else if ((regIndex == PICO9918_REG_ENHANCED2) && (value & PICO9918_R50_RESET))
3512 { // reset all registers?
3513 vdpRegisterReset(tms9918);
3514
3515 // reset palette, etc as well?
3516 if (value & 0x40)
3517 {
3518 tms9918->configDirty = true;
3519 tms9918->configVdpDirty = true;
3520 }
3521 }
3522 else if (regIndex == PICO9918_REG_STATUS_SELECT)
3523 {
3524 uint8_t statReg = (value & 0x0f);
3525 TMS_STATUS(tms9918, 0x0F) = statReg; // is this right? or should this be the read-ahead value?
3526 if (value & 0x40) tms9918->startTime = PICO9918_HOST_TIME_US(); // reset
3527 if (value & 0x20)
3528 tms9918->currentTime = PICO9918_HOST_TIME_US(); // snap
3529 else if (value & 0x10)
3530 tms9918->startTime += (tms9918->stopTime - tms9918->startTime);
3531 else
3532 tms9918->currentTime = tms9918->stopTime = PICO9918_HOST_TIME_US();
3533
3534 if (statReg > 3 && statReg < 12)
3535 {
3536 uint32_t elapsed = tms9918->currentTime - tms9918->startTime;
3537 uint32_t microQ, microR;
3538 PICO9918_DIVMOD_U32(elapsed, 1000, microQ, microR);
3539 uint32_t milliQ, milliR;
3540 PICO9918_DIVMOD_U32(microQ, 1000, milliQ, milliR);
3541
3542 TMS_STATUS(tms9918, PICO9918_SR_MICROS_LSB) = microR & 0x0ff;
3543 TMS_STATUS(tms9918, PICO9918_SR_MICROS_MSB) = microR >> 8;
3544 TMS_STATUS(tms9918, PICO9918_SR_MILLIS_LSB) = milliR & 0x0ff;
3545 TMS_STATUS(tms9918, PICO9918_SR_MILLIS_MSB) = milliR >> 8;
3546 TMS_STATUS(tms9918, PICO9918_SR_SECONDS_LSB) = milliQ & 0x00ff;
3547 TMS_STATUS(tms9918, PICO9918_SR_SECONDS_MSB) = milliQ >> 8;
3548 }
3549 }
3550 // SR12 holds the value of the option in VR58 (options)
3551 else if (regIndex == PICO9918_REG_CONFIG_INDEX && PICO9918_HAS(tms9918, PICO9918_FEAT_CONFIG))
3552 {
3553 const uint8_t option = TMS_REGISTER(tms9918, PICO9918_REG_CONFIG_INDEX);
3554
3555 TMS_STATUS(tms9918, PICO9918_SR_CONFIG_VALUE) = tms9918->config[option];
3556 }
3557 // option number in reg 58, value in 59 (options)
3558 else if (regIndex == PICO9918_REG_CONFIG_VALUE && PICO9918_HAS(tms9918, PICO9918_FEAT_CONFIG) &&
3560 {
3561 const uint8_t option = TMS_REGISTER(tms9918, PICO9918_REG_CONFIG_INDEX);
3562
3563 tms9918->config[option] = value;
3564 TMS_STATUS(tms9918, PICO9918_SR_CONFIG_VALUE) = value;
3565 tms9918->configDirty = true;
3566 }
3567 }
3568}
3569
3570
3571/** \brief return a value from vram */
3573uint8_t __time_critical_func(pico9918_vram_value)(PICO9918_INST_ARG uint16_t addr)
3574{
3575 return tms9918->vram.bytes[addr & VRAM_MASK];
3576}
3577
3578/** \brief check BLANK flag */
3581{
3582 return (TMS_REGISTER(tms9918, TMS_REG_1) & TMS_R1_DISP_ACTIVE);
3583}
3584
3585/** \brief current display mode */
3588{
3589 return pico9918_cached_mode;
3590}
3591
3592#if PICO9918_BUILD_DEBUG_API
3593/** \brief see impl/pico9918_priv.h. What the scanline entry does, for a caller between two. */
3595{
3596 pico9918_cached_mode = tmsMode(tms9918);
3597}
3598#endif
3599
3600/**
3601 * \brief how many bytes of pixels[] this mode fills. Every mode is 256 but unlocked 80-column text on a
3602 * board with the 8bpp tier, which is 512 - so the palette expansion, the backdrop fill and anything
3603 * reading the line ask here rather than each deciding it again.
3604 */
3606uint32_t __time_critical_func(pico9918_line_bytes)(PICO9918_INST_ONLY_ARG)
3607{
3608 return TEXT80_WIDE_ROW ? SCANLINE_BYTES_MAX : TMS9918_PIXELS_X;
3609}
3610
3611/**
3612 * \brief where the scanline just generated actually is. Usually the buffer that was passed in, but on a
3613 * line with nothing to arbitrate it is a tile layer's own buffer and the passed one holds only the
3614 * backdrop fill - so read the line from here rather than from what was handed over.
3615 */
3617const uint8_t* __time_critical_func(pico9918_line_source)(PICO9918_INST_ONLY_ARG)
3618{
3619 return pico9918_cached_line_source;
3620}
3621
3622/** \brief a default palette entry, 0xargb */
3624uint16_t pico9918_default_palette(int index)
3625{
3626 return defaultPalette[index & 0x3f];
3627}
uint8_t pico9918_peek_status(pico9918_t *tms9918)
read from the status register without resetting it
Definition pico9918.c:445
static bool tmsSpriteMag(pico9918_t *tms9918)
sprite size (0 = 1x, 1 = 2x)
Definition pico9918.c:194
bool pico9918_unlocked(pico9918_t *tms9918)
see the header.
Definition pico9918.c:124
uint8_t pico9918_read_data(pico9918_t *tms9918)
read data (mode = 0) from the tms9918
Definition pico9918.c:462
PICO9918_INTERNAL void pico9918_write_reg_value_impl(pico9918_t *tms9918, uint8_t reg, uint8_t value)
set a register from the second byte of a host register write
Definition pico9918.c:3433
static uint16_t tmsSpritePatternTableAddr(pico9918_t *tms9918)
sprite pattern table base address
Definition pico9918.c:242
static uint16_t tmsPatternTableAddr(pico9918_t *tms9918)
pattern table base address
Definition pico9918.c:228
static PICO9918_NOINLINE void compositeTile2OnlyBuffer(pico9918_t *tms9918, uint8_t pixels[TMS9918_PIXELS_X], const int t2Scroll, const bool wide)
Composite with tile layer 1 disabled (reg 0x32 bit 4).
Definition pico9918.c:3141
static uint16_t tmsNameTable2Addr(pico9918_t *tms9918)
name table base address
Definition pico9918.c:206
PICO9918_INLINE_HOT void renderText80Row(pico9918_t *tms9918, const uint8_t *__restrict rowNames, const uint8_t *__restrict patternTable, const uint8_t *__restrict rowColors, const bool opaq, uint8_t *__restrict pixels, const uint32_t startCol, const uint32_t numCells, const bool scrolled)
one 80-column text row, six pixels a cell at half a byte each.
Definition pico9918.c:1932
pico9918_mode_t pico9918_display_mode(pico9918_t *tms9918)
current display mode
Definition pico9918.c:3587
bool pico9918_display_enabled(pico9918_t *tms9918)
check BLANK flag
Definition pico9918.c:3580
static void tileRowAddr(pico9918_t *tms9918, const uint16_t y, const uint16_t rawY, const uint8_t colorReg, const bool gm2, const bool mcm, TileRowAddr *addr, uint16_t *colorTableAddr)
the per-mode half of a tile row's addressing, once per layer per scanline.
Definition pico9918.c:1368
void pico9918_destroy(pico9918_t *tms9918)
destroy a TMS9918
Definition pico9918.c:420
static bool bitmap_layer_scan_line(pico9918_t *tms9918, uint16_t y, uint8_t pixels[TMS9918_PIXELS_X], const bool intoTile1)
generate an F18A bitmap layer scanline
Definition pico9918.c:2839
void pico9918_set_interrupt_callback(pico9918_t *tms9918, pico9918_interrupt_fn cb, void *userdata)
register the host's /INT hook, so a host need not poll
Definition pico9918.c:141
void pico9918_write_addr(pico9918_t *tms9918, uint8_t data)
write an address (mode = 1) to the tms9918
Definition pico9918.c:433
PICO9918_INLINE_HOT void renderTileRowLocked(pico9918_t *tms9918, uint16_t rowNamesAddr, uint16_t colorTableAddr, uint8_t tileIndex, uint8_t pal, uint8_t pixels[TMS9918_PIXELS_X], const TileRowAddr *addr, const bool gm2, const bool mcm)
generate a locked (plain TMS9918) tile row, straight into the scanline
Definition pico9918.c:2390
static void f18a_tile1_scan_line(pico9918_t *tms9918, uint16_t y)
generate a Graphics I mode scanline for the T1 layer
Definition pico9918.c:2645
static void tmsClearRowBitsMask(const uint32_t xPos, const uint32_t tilePixels, const uint32_t tileWidth, BitMask rowBitsMask)
Clear the row pixels bit mask.
Definition pico9918.c:556
PICO9918_INLINE_HOT void f18a_tile_layer_scan_line(pico9918_t *tms9918, uint16_t y, const TileLayerConfig *config, const bool blend)
generate a tile mode scanline for either T1 or T2 layer
Definition pico9918.c:2571
void pico9918_set_chip(pico9918_t *tms9918, pico9918_chip_t chip)
select which chip this instance answers as
Definition pico9918.c:330
const uint8_t * pico9918_line_source(pico9918_t *tms9918)
where the scanline just generated actually is.
Definition pico9918.c:3617
static uint8_t renderSprites(pico9918_t *tms9918, const uint32_t spriteCount, const bool spriteMag, const bool wide, const bool ecm0, uint8_t pixels[TMS9918_PIXELS_X])
Output Sprites to a scanline.
Definition pico9918.c:988
static uint16_t tmsColorTable2Addr(pico9918_t *tms9918)
color table base address
Definition pico9918.c:220
uint8_t pico9918_read_data_no_inc(pico9918_t *tms9918)
read data (mode = 0) from the tms9918
Definition pico9918.c:468
static uint16_t tmsColorTableAddr(pico9918_t *tms9918)
color table base address
Definition pico9918.c:212
static pico9918_color_t tmsMainFgColor(pico9918_t *tms9918)
foreground color
Definition pico9918.c:254
uint8_t pico9918_status_value(pico9918_t *tms9918, pico9918_status_register_t reg)
return a status register value without the side effects of reading it
Definition pico9918.c:3427
PICO9918_INLINE_HOT bool renderBitmapLayerBody(pico9918_t *tms9918, uint16_t y, bool opaque, const uint8_t width, const uint16_t addr, const uint8_t bmlCtl, uint8_t pixels[TMS9918_PIXELS_X], const bool wide, const bool intoTile1)
generate an F18A bitmap layer scanline
Definition pico9918.c:2668
static uint8_t graphics_i_scan_line(pico9918_t *tms9918, uint16_t y, uint8_t pixels[TMS9918_PIXELS_X])
generate a Graphics I or Graphics II mode scanline
Definition pico9918.c:3199
static uint32_t tmsTestRowBitsMaskAligned(const uint32_t xPos, const uint32_t tilePixels, const BitMask rowBitsMask)
Test against the row pixels bit mask (aligned - no word boundary crossing).
Definition pico9918.c:582
static void tmsSetTransparentSpriteMask(const uint32_t xPos, const uint32_t spritePixels, const uint32_t spriteWidth)
set the transparent sprite mask.
Definition pico9918.c:539
PICO9918_INLINE_HOT void renderTextRow(pico9918_t *tms9918, const uint8_t *__restrict rowNames, const TileRowAddr *__restrict addr, const uint8_t *__restrict rowColors, const uint32_t colorStride, uint32_t pal, uint8_t *__restrict dest, const uint32_t scroll, const bool alwaysOnTop, const uint32_t numCols, const bool isTile2, const uint32_t ecm, const bool blend)
one 40- or 80-column text row, six pixels a cell at one byte each
Definition pico9918.c:1601
static void textRowBorder(pico9918_t *tms9918, const int t1Scroll, const uint32_t padding, const uint32_t width)
A text row draws 240 of the 256 pixels and the composite copies all of them, so layer 1's buffer has ...
Definition pico9918.c:3185
void pico9918_debug_sync_mode_impl(pico9918_t *tms9918)
see impl/pico9918_priv.h.
Definition pico9918.c:3594
static void tileLayerAddr(pico9918_t *tms9918, const uint16_t rawY, const TileLayerConfig *config, const uint8_t numCols, const uint8_t nameAddrMask, const bool textMode, const bool gm2, const bool mcm, const bool attrPerPos, TileRowAddr *addr, uint16_t *namesAddr, uint16_t *colorAddr)
where one layer reads this scanline from: the vertical scroll and its page swap, the name and colour ...
Definition pico9918.c:1517
pico9918_chip_t pico9918_chip(pico9918_t *tms9918)
which chip this instance answers as
Definition pico9918.c:370
void pico9918_interrupt_dispatch(pico9918_t *tms9918, bool active)
see impl.
Definition pico9918.c:149
static void renderEcm0Tile(pico9918_t *tms9918, uint8_t *buffer, const uint32_t xPos, const uint8_t pattIdx, const uint8_t patternTable[], const uint32_t colorTableAddr, const uint32_t pal, const bool isTile2, const bool gm2Color, const bool mcm)
render an ECM0 (enhanced color mode) graphics I tile.
Definition pico9918.c:2232
static uint32_t tmsTestCollisionMask(const uint32_t xPos, const uint32_t spritePixels, const uint32_t spriteWidth)
Test and update the sprite collision mask.
Definition pico9918.c:515
uint8_t pico9918_vram_value(pico9918_t *tms9918, uint16_t addr)
return a value from vram
Definition pico9918.c:3573
static void text_scan_line(pico9918_t *tms9918, uint16_t y, uint8_t pixels[TMS9918_PIXELS_X])
generate a 40- or 80-column text mode scanline.
Definition pico9918.c:2126
size_t pico9918_instance_size(void)
see the header.
Definition pico9918.c:118
static uint16_t tmsSpriteAttrTableAddr(pico9918_t *tms9918)
sprite attribute table base address
Definition pico9918.c:236
static void tmsUpdateRowBitsMaskAligned(const uint32_t xPos, const uint32_t tilePixels, BitMask rowBitsMask)
Update the row pixels bit mask (aligned - no word boundary crossing).
Definition pico9918.c:575
void pico9918_interrupt_set(pico9918_t *tms9918)
raise the INT status flag
Definition pico9918.c:480
uint32_t pico9918_line_bytes(pico9918_t *tms9918)
how many bytes of pixels[] this mode fills.
Definition pico9918.c:3606
void pico9918_set_status(pico9918_t *tms9918, uint8_t status)
set status flag
Definition pico9918.c:487
uint8_t pico9918_scan_line(pico9918_t *tms9918, uint16_t y)
generate a scanline
Definition pico9918.c:3335
static pico9918_color_t tmsFgColor(pico9918_t *tms9918, uint8_t colorByte)
foreground color
Definition pico9918.c:261
static void writeToAlignedBuffer(uint8_t *buffer, uint32_t xPos, const uint32_t left, const uint32_t right)
Write full tile to aligned buffer - 8 pixels at once.
Definition pico9918.c:2219
void pico9918_reset(pico9918_t *tms9918)
reset the new TMS9918
Definition pico9918.c:378
static uint8_t chipFeatures(pico9918_chip_t chip)
the feature bits a personality answers to - the ladder, in one place
Definition pico9918.c:313
static uint32_t spriteEcm(pico9918_t *tms9918)
the sprite ECM level, zero on a locked device - the one term a clone can pin
Definition pico9918.c:979
uint8_t pico9918_read_status(pico9918_t *tms9918)
read from the status register
Definition pico9918.c:439
static pico9918_color_t tmsBgColor(pico9918_t *tms9918, uint8_t colorByte)
background color
Definition pico9918.c:268
static uint16_t tmsNameTableAddr(pico9918_t *tms9918)
name table base address
Definition pico9918.c:200
static void f18a_tile2_scan_line(pico9918_t *tms9918, uint16_t y, const bool blend)
generate a Graphics I mode scanline for the T2 layer
Definition pico9918.c:2651
uint16_t pico9918_default_palette(int index)
a default palette entry, 0xargb
Definition pico9918.c:3624
static uint32_t tmsTestRowBitsMask(const uint32_t xPos, const uint32_t tilePixels, const uint32_t tileWidth, const bool update, const bool test, const bool testColl)
Test and update the row pixels bit mask.
Definition pico9918.c:657
bool pico9918_interrupt_status(pico9918_t *tms9918)
return true if both INT status and INT control set
Definition pico9918.c:474
static void renderEcmTileToAlignedBuffer(pico9918_t *tms9918, uint8_t *buffer, const uint32_t xPos, const uint32_t pixelOffset, const uint8_t pattIdx, const uint8_t patternTable[], const uint32_t colorTableAddr, const uint32_t ecm, const uint32_t ecmOffset, const uint32_t ecmColorMask, const uint32_t ecmColorOffset, const uint32_t pal, const bool attrPerPos, const int32_t flipY, const uint32_t tileIndex, uint32_t *lastEmpty, const bool isTile2, const bool alwaysOnTop)
render one ECM tile into the layer buffer
Definition pico9918.c:2282
uint8_t pico9918_reg_value(pico9918_t *tms9918, pico9918_register_t reg)
return a register value - see the header for the locked-device aliasing
Definition pico9918.c:3420
static pico9918_color_t tmsMainBgColor(pico9918_t *tms9918)
background color
Definition pico9918.c:248
static uint8_t tmsSpriteSize(pico9918_t *tms9918)
sprite size (8 or 16)
Definition pico9918.c:188
void pico9918_write_data(pico9918_t *tms9918, uint8_t data)
write data (mode = 0) to the tms9918
Definition pico9918.c:455
#define TMS_R1_SPRITE_MAG2
sprites drawn at twice their pattern size
Definition pico9918.h:324
pico9918_color_t
the sixteen TMS9918 colours, in palette-index order
Definition pico9918.h:191
#define PICO9918_R29_SPRITE_STRIDE
register 29 fields: scroll page sizes, and the stride between ECM pattern planes
Definition pico9918.h:336
#define PICO9918_MAP_REGISTERS
the register file, VR0-VR63
Definition pico9918.h:452
pico9918_t * pico9918_new(void)
create a new TMS9918
#define PICO9918_MAP_PRAM
palette RAM, 64 entries of RGB444
Definition pico9918.h:451
#define PICO9918_R49_TILE2_ENABLE
register 49 bits: tile layer 2, row count, and the enhanced colour modes
Definition pico9918.h:356
#define PICO9918_SR0_5S
more sprites on a line than the limit allows
Definition pico9918.h:290
#define PICO9918_R29_TILE_STRIDE
tile pattern plane stride, 0x800 >> n
Definition pico9918.h:339
#define PICO9918_R24_SPRITE_PS
register 24 bits: the sub-palette each layer takes
Definition pico9918.h:330
#define TMS_R1_MODE_TEXT
40-column text
Definition pico9918.h:320
#define PICO9918_R56_GPU_RUN
register 56 bit: the GPU trigger
Definition pico9918.h:379
#define PICO9918_INST_ARG
declare the instance ahead of other parameters
Definition pico9918.h:83
#define PICO9918_INST_ONLY
pass the instance as the only argument
Definition pico9918.h:86
void(* pico9918_interrupt_fn)(pico9918_t *instance, bool active, void *userdata)
the host's /INT hook: called whenever the library drives the line
Definition pico9918.h:563
#define TMS_R1_MODE_MULTICOLOR
Multicolor.
Definition pico9918.h:319
#define PICO9918_R50_TILE1_OFF
stop drawing tile layer 1
Definition pico9918.h:372
#define PICO9918_R50_RESET
register 50 bits: GPU triggers and the remaining layer controls
Definition pico9918.h:369
#define PICO9918_R24_TILE1_PS
tile layer 1 palette select
Definition pico9918.h:333
#define PICO9918_R49_ECM_TILE
tile ECM level field
Definition pico9918.h:358
#define PICO9918_R49_Y_REAL
sprite Y is the real row, not row minus one
Definition pico9918.h:362
#define TMS_R1_DISP_ACTIVE
render the active display
Definition pico9918.h:314
#define TMS_R0_MODE_GRAPHICS_II
Graphics II - the only mode R0 selects.
Definition pico9918.h:301
#define TMS_R1_SPRITE_16
16x16 sprite patterns
Definition pico9918.h:322
#define PICO9918_MAP_STATUS
the status registers, SR0-SR15
Definition pico9918.h:454
#define PICO9918_MAP_SCANLINE
the current scanline, then the blanking flag
Definition pico9918.h:453
#define PICO9918_R15_STATUS_NUM
which status register S1 reads back
Definition pico9918.h:388
#define PICO9918_R50_REPORT_MAX
S0's sprite number reports the highest seen.
Definition pico9918.h:373
pico9918_register_t
the eight TMS9918 registers, by number and by what each one holds
Definition pico9918.h:212
@ PICO9918_REG_COLOR_TABLE2
tile layer 2 colour table base
Definition pico9918.h:232
@ PICO9918_REG_CONFIG_INDEX
PICO9918 only: which configuration byte R59 addresses.
Definition pico9918.h:257
@ PICO9918_REG_BML_X
bitmap layer left edge
Definition pico9918.h:244
@ PICO9918_REG_BML_BASE
bitmap layer base address, in 64-byte units
Definition pico9918.h:243
@ PICO9918_REG_VRAM_INC
signed VRAM address increment per access
Definition pico9918.h:249
@ PICO9918_REG_CONFIG_VALUE
PICO9918 only: the configuration byte R58 selected.
Definition pico9918.h:258
@ PICO9918_REG_NAME_TABLE2
tile layer 2 name table base
Definition pico9918.h:231
@ PICO9918_REG_ENHANCED1
tile layer 2, 30-row mode, ECM levels, real Y
Definition pico9918.h:250
@ PICO9918_REG_GPU_PC_MSB
GPU program counter, high byte.
Definition pico9918.h:253
@ PICO9918_REG_STATUS_SELECT
which status register S1 reads back, and the counter controls
Definition pico9918.h:233
@ PICO9918_REG_ENHANCED2
GPU triggers, per-position attributes, layer priority.
Definition pico9918.h:251
@ PICO9918_REG_T2_HSCROLL
tile layer 2 horizontal scroll
Definition pico9918.h:236
@ PICO9918_REG_PAGE_SIZE
scroll page sizes, and the ECM pattern plane stride
Definition pico9918.h:240
@ PICO9918_REG_UNLOCK
0x1c twice unlocks the F18A personality; any other value locks
Definition pico9918.h:256
@ PICO9918_REG_BML_WIDTH
bitmap layer width in pixels
Definition pico9918.h:246
@ PICO9918_REG_MAX_SCAN_SPRITES
sprites drawn per scanline before the limit bites
Definition pico9918.h:241
@ PICO9918_REG_GPU_PC_LSB
GPU program counter, low byte - writing it also starts the GPU.
Definition pico9918.h:254
@ PICO9918_REG_BML_TOP_ROW
bitmap layer top row
Definition pico9918.h:245
@ PICO9918_REG_MAX_SPRITES
sprites processed per frame before the scan stops
Definition pico9918.h:252
@ PICO9918_REG_PALETTE_SELECT
sub-palette for sprites and each tile layer
Definition pico9918.h:235
@ PICO9918_REG_BML_CONTROL
bitmap layer enable, priority, transparency, fat pixels
Definition pico9918.h:242
@ PICO9918_REG_T1_HSCROLL
tile layer 1 horizontal scroll
Definition pico9918.h:238
@ PICO9918_REG_GPU_CONTROL
GPU load and trigger.
Definition pico9918.h:255
@ PICO9918_REG_FLASH_CONTROL
PICO9918 only: flash operation control.
Definition pico9918.h:259
@ PICO9918_REG_BML_HEIGHT
bitmap layer height in rows
Definition pico9918.h:247
pico9918_status_register_t
the status registers, by number and by what each one reports
Definition pico9918.h:269
@ PICO9918_SR_STATUS
the TMS9918A status: interrupt, 5th sprite, collision, sprite number
Definition pico9918.h:270
@ PICO9918_SR_MICROS_LSB
microsecond counter, low byte
Definition pico9918.h:276
@ PICO9918_SR_CONFIG_VALUE
PICO9918 only: the configuration byte R58 selected.
Definition pico9918.h:282
@ PICO9918_SR_MILLIS_LSB
millisecond counter, low byte
Definition pico9918.h:278
@ PICO9918_SR_SECONDS_LSB
second counter, low byte
Definition pico9918.h:280
@ PICO9918_SR_MICROS_MSB
microsecond counter, high bits
Definition pico9918.h:277
@ PICO9918_SR_IDENT
chip identity, blanking, and the scanline interrupt flag
Definition pico9918.h:271
@ PICO9918_SR_MILLIS_MSB
millisecond counter, high bits
Definition pico9918.h:279
@ PICO9918_SR_VERSION
the F18A feature level, as major and minor nibbles
Definition pico9918.h:284
@ PICO9918_SR_SECONDS_MSB
second counter, high byte
Definition pico9918.h:281
@ PICO9918_SR_GPU
GPU running and its status byte.
Definition pico9918.h:272
#define PICO9918_R49_ECM_SPRITE
sprite ECM level field
Definition pico9918.h:363
#define PICO9918_INST_ONLY_ARG
declare the instance as the only parameter
Definition pico9918.h:84
#define PICO9918_CHIP_MAX
the highest personality this build can be, and what a new instance is
Definition pico9918.h:182
#define TMS9918_PIXELS_X
active display width, every mode
Definition pico9918.h:390
#define TMS_R0_DOUBLE_ROWS
PICO9918 only: twice the rows, drawn interlaced.
Definition pico9918.h:307
#define PICO9918_R31_BML_ENABLE
register 31 bits: the bitmap layer
Definition pico9918.h:344
#define PICO9918_SR0_COLLISION
two sprites overlapped on an opaque pixel
Definition pico9918.h:291
#define PICO9918_R50_POS_ATTR
tile attributes come per position, not per tile
Definition pico9918.h:375
#define PICO9918_R49_ROW30
30 rows of tiles rather than 24
Definition pico9918.h:357
pico9918_chip_t
which chip an instance answers as
Definition pico9918.h:164
@ PICO9918_CHIP_PICO9918
an F18A plus the PICO9918's own extensions
Definition pico9918.h:168
@ PICO9918_CHIP_PICO9918_PRO
a PICO9918 PRO: 8bpp 80-column text, its own splash
Definition pico9918.h:169
@ PICO9918_CHIP_TMS9918A
a TMS9918A: locked, no GPU, no extensions
Definition pico9918.h:166
@ PICO9918_CHIP_F18A
an F18A: unlock, enhanced renderer, GPU
Definition pico9918.h:167
#define PICO9918_INST
pass the instance ahead of other arguments
Definition pico9918.h:85
#define PICO9918_DLLEXPORT
the linkage every public entry point carries - see LINKAGE MODES above
Definition pico9918.h:50
pico9918_mode_t
the display modes the VDP can be in, TMS9918A modes and F18A alike
Definition pico9918.h:118
#define PICO9918_INTERNAL
cross-TU linkage for what is not public API, so a DLL or wasm build exports none of it
Definition pico9918.h:54
#define PICO9918_BASE_TMS9918
vdpBase values - the render base selected by PICO9918_CONF_VDP_BASE
#define PICO9918_CONFIG_FIRST_SETTABLE
the first config byte a guest may write through VR58/59
pico9918-core - GPU privileged surface
pico9918-core - the private instance layout
PICO9918_INLINE_HOT void pico9918_write_addr_impl(pico9918_t *tms9918, uint8_t data)
write an address (mode = 1) to the tms9918
PICO9918_INLINE uint8_t pico9918_read_status_impl(pico9918_t *tms9918)
read from the status register
PICO9918_INLINE void pico9918_write_reconcile_int_impl(pico9918_t *tms9918)
bring /INT into agreement after a write that can change the predicate.
PICO9918_INLINE_HOT void pico9918_set_status_impl(pico9918_t *tms9918, uint8_t status)
set status flag
PICO9918_INLINE void pico9918_frame_reset_int_impl(pico9918_t *tms9918)
console-reset entry for the interrupt/status state.
PICO9918_INLINE_HOT void pico9918_interrupt_set_impl(pico9918_t *tms9918)
raise the interrupt flag in SR0 and the frame shadow
PICO9918_INLINE_HOT uint8_t pico9918_peek_status_impl(pico9918_t *tms9918)
read from the status register without resetting it
PICO9918_INLINE_HOT uint8_t pico9918_read_data_no_inc_impl(pico9918_t *tms9918)
return the buffered value without reading VRAM or advancing the address
PICO9918_INLINE_HOT bool pico9918_interrupt_status_impl(pico9918_t *tms9918)
whether /INT should be asserted
PICO9918_INLINE_HOT uint8_t pico9918_read_data_impl(pico9918_t *tms9918)
read data (mode = 0) from the tms9918
PICO9918_INLINE_HOT void pico9918_write_data_impl(pico9918_t *tms9918, uint8_t data)
write data (mode = 0) to the tms9918
void pico9918_splash_select_pro(bool pro)
render the F18A's power-on badge into the scanline buffer
Definition splash.c:36
void pico9918_splash_reset(void)
restart the splash animation (after... reset)
Definition splash.c:57
pico9918-core - Splash overlay