pico9918-core 1.3.0
TMS9918A / F18A video display processor emulation in C99
Loading...
Searching...
No Matches
gpu_test.c
Go to the documentation of this file.
1/**
2 * \file
3 * \brief pico9918-core - the library-paced GPU
4 *
5 * Copyright (c) 2026 Troy Schrapel
6 *
7 * This code is licensed under the MIT license
8 *
9 * https://github.com/visrealm/pico9918-core
10 *
11 * pico9918_gpu_set_clock's contract, which nothing else covers: a rate makes the library
12 * run an armed program from inside the arming write, and zero leaves the GPU to the host.
13 *
14 * The first half is what software detecting an F18A depends on. The probe arms a tiny
15 * self-modifying program and reads its result back within a handful of host cycles, so a
16 * host servicing the GPU once a scanline sees the chip only when a scanline boundary
17 * happens to fall in between - a real flaky-detection bug in two emulators.
18 *
19 * The last section is the DMA engine, which is here because a program triggering one is
20 * the only way to reach it: the ports it answers to are above the address space a host
21 * can write.
22 */
23
24#include "impl/pico9918_priv.h"
25#include "gpu/gpu.h"
26#include "pico9918_debug.h"
27
28#include <stdio.h>
29
30#define PROGRAM_AT 0x2000u
31#define RESULT_AT 0x2100u
32#define MARKER 0xbeefu
33
34static int failures;
35
36static void fail(const char* what, unsigned wanted, unsigned got)
37{
38 ++failures;
39 printf(" FAIL %s: want %04x got %04x\n", what, wanted, got);
40}
41
42static void regWrite(uint8_t reg, uint8_t value)
43{
45}
46
47/* Two writes of 0x1c to R57, which is what an F18A answers to. */
48static void unlock(void)
49{
50 regWrite(0x39, 0x1c);
51 regWrite(0x39, 0x1c);
52}
53
54/*
55 * LI R0, >BEEF 0200 BEEF
56 * MOV R0, @>2100 C800 2100
57 * STST R3 02C3 so the status accessor has something to agree with
58 * IDLE 0340
59 *
60 * Written straight into VRAM rather than through the host bus, which masks to 16K.
61 */
62static void loadProgram(void)
63{
64 static const uint8_t program[] = {0x02, 0x00, 0xbe, 0xef, 0xc8, 0x00,
65 0x21, 0x00, 0x02, 0xc3, 0x03, 0x40};
66
67 for (unsigned i = 0; i < sizeof(program); ++i) tms9918->vram.bytes[PROGRAM_AT + i] = program[i];
68
69 tms9918->vram.bytes[RESULT_AT] = 0;
70 tms9918->vram.bytes[RESULT_AT + 1] = 0;
71}
72
73static uint16_t result(void)
74{
75 return (uint16_t)((tms9918->vram.bytes[RESULT_AT] << 8) | tms9918->vram.bytes[RESULT_AT + 1]);
76}
77
78/* Arm at PROGRAM_AT. Writing R55 is what arms it, so R54 goes first. */
79static void arm(void)
80{
81 regWrite(0x36, (uint8_t)(PROGRAM_AT >> 8));
82 regWrite(0x37, (uint8_t)(PROGRAM_AT & 0xff));
83}
84
85/*
86 * LI R1, >8008 0201 8008 the DMA trigger port
87 * LI R2, >0100 0202 0100
88 * MOV R2, *R1 C442 any write to >8008 starts the transfer
89 * IDLE 0340
90 *
91 * The registers below it are set by the case rather than by the program: what is under
92 * test is the engine, not a program's ability to load eight bytes.
93 */
94static void loadDmaProgram(void)
95{
96 static const uint8_t program[] = {0x02, 0x01, 0x80, 0x08, 0x02, 0x02,
97 0x01, 0x00, 0xc4, 0x42, 0x03, 0x40};
98
99 for (unsigned i = 0; i < sizeof(program); ++i) tms9918->vram.bytes[PROGRAM_AT + i] = program[i];
100}
101
102#define NEW_WP 0x2800u
103
104/*
105 * LWPI >2800 02E0 2800
106 * LI R0, >BEEF 0200 BEEF lands at >2800, which is R0 of the new workspace
107 * IDLE 0340
108 */
109static void loadWorkspaceProgram(void)
110{
111 static const uint8_t program[] = {0x02, 0xe0, 0x28, 0x00, 0x02, 0x00, 0xbe, 0xef, 0x03, 0x40};
112
113 for (unsigned i = 0; i < sizeof(program); ++i) tms9918->vram.bytes[PROGRAM_AT + i] = program[i];
114
115 tms9918->vram.bytes[NEW_WP] = 0;
116 tms9918->vram.bytes[NEW_WP + 1] = 0;
117 tms9918->vram.map.wrksp[0] = 0;
118 tms9918->vram.bytes[PICO9918_GPU_WORKSPACE] = 0;
119}
120
121/*
122 * MOV R0, @>2100 C800 2100 R0 is the caller's to set, not the program's
123 * IDLE 0340
124 */
125static void loadEchoProgram(void)
126{
127 static const uint8_t program[] = {0xc8, 0x00, 0x21, 0x00, 0x03, 0x40};
128
129 for (unsigned i = 0; i < sizeof(program); ++i) tms9918->vram.bytes[PROGRAM_AT + i] = program[i];
130
131 tms9918->vram.bytes[RESULT_AT] = 0;
132 tms9918->vram.bytes[RESULT_AT + 1] = 0;
133 tms9918->vram.bytes[PICO9918_GPU_WORKSPACE] = 0;
134 tms9918->vram.bytes[PICO9918_GPU_WORKSPACE + 1] = 0;
135}
136
137/*
138 * STST R3 02C3 nothing above it, so R3 is what the caller wrote
139 * IDLE 0340
140 */
141static void loadStatusProgram(void)
142{
143 static const uint8_t program[] = {0x02, 0xc3, 0x03, 0x40};
144
145 for (unsigned i = 0; i < sizeof(program); ++i) tms9918->vram.bytes[PROGRAM_AT + i] = program[i];
146}
147
148#define DMA_SRC 0x1000u
149#define DMA_DST 0x1800u
150
151/* One transfer, from a source that is 0x40 counting up. Returns with the destination
152 wherever the engine left it. */
153static void dma(uint32_t src, uint32_t dst, uint8_t width, uint8_t height, uint8_t stride,
154 uint8_t params)
155{
156 for (unsigned i = 0; i < 0x400; ++i)
157 {
158 tms9918->vram.bytes[DMA_SRC + i] = (uint8_t)(0x40 + i);
159 tms9918->vram.bytes[DMA_DST + i] = 0;
160 }
161
162 tms9918->vram.bytes[0x8000] = (uint8_t)(src >> 8);
163 tms9918->vram.bytes[0x8001] = (uint8_t)src;
164 tms9918->vram.bytes[0x8002] = (uint8_t)(dst >> 8);
165 tms9918->vram.bytes[0x8003] = (uint8_t)dst;
166 tms9918->vram.bytes[0x8004] = width;
167 tms9918->vram.bytes[0x8005] = height;
168 tms9918->vram.bytes[0x8006] = stride;
169 tms9918->vram.bytes[0x8007] = params;
170
171 loadDmaProgram();
172 arm();
173}
174
175static void expect(const char* what, uint32_t at, uint8_t wanted)
176{
177 if (tms9918->vram.bytes[at] != wanted) fail(what, wanted, tms9918->vram.bytes[at]);
178}
179
180static uint16_t stepPc[8];
181static unsigned stepCalls;
182static uint16_t stepStopAt;
183static unsigned stepWrongArgs;
184
185static bool stepWatch(pico9918_t* vdp, uint16_t pc, void* userdata)
186{
187 if (vdp != tms9918 || userdata != &stepStopAt) ++stepWrongArgs;
188 if (stepCalls < 8) stepPc[stepCalls] = pc;
189 ++stepCalls;
190
191 return pc != stepStopAt;
192}
193
194#if PICO9918_BUILD_STEP_CALLBACK
195static uint16_t armedPc[8];
196static unsigned armedCalls;
197static uint16_t armedStopAt;
198static unsigned armedWrongArgs;
199
200static bool armedWatch(pico9918_t* vdp, uint16_t pc, void* userdata)
201{
202 if (vdp != tms9918 || userdata != &armedStopAt) ++armedWrongArgs;
203 if (armedCalls < 8) armedPc[armedCalls] = pc;
204 ++armedCalls;
205
206 return pc != armedStopAt;
207}
208#endif
209
210int main(void)
211{
212 pico9918_init();
214
215 /* 1. no rate is the default, and leaves the GPU to whoever else drives it. The
216 program is armed, so a host's own step_n would still run it - but the arming
217 write must not. */
218 unlock();
219 loadProgram();
220 arm();
221 if (result() != 0) fail("no-rate-ran", 0, result());
222
223 /* and armed is where pico9918_gpu_pc says it is, before a single instruction runs */
224 if (pico9918_gpu_pc(PICO9918_INST_ONLY) != PROGRAM_AT)
225 fail("armed-pc", PROGRAM_AT, pico9918_gpu_pc(PICO9918_INST_ONLY));
226
227 /* and a disassembler can reach the program: the guest's own view stops at 0x3fff, so
228 PROGRAM_AT is in range for both, but only the GPU's reaches past it. */
229 if (pico9918_gpu_mem_value(PICO9918_INST PROGRAM_AT) != 0x02)
230 fail("gpu-mem-program", 0x02, pico9918_gpu_mem_value(PICO9918_INST PROGRAM_AT));
231
232 tms9918->vram.bytes[0x4100] = 0x5a;
233 if (pico9918_gpu_mem_value(PICO9918_INST 0x4100) != 0x5a)
234 fail("gpu-mem-gram", 0x5a, pico9918_gpu_mem_value(PICO9918_INST 0x4100));
235 if (pico9918_vram_value(PICO9918_INST 0x4100) == 0x5a) fail("vram-value-reached-gram", 0, 0x5a);
236
237 /* and it stops where the map does rather than walking into the instance */
239 fail("gpu-mem-past-end", 0, pico9918_gpu_mem_value(PICO9918_INST pico9918_gpu_mem_size()));
240
241 /* 2. and it is only armed, not lost: the host-driven path still works. */
243 if (result() != MARKER) fail("host-driven", MARKER, result());
244
245 /* the program left >BEEF in R0, which is the last word of the map, and R1 is above it
246 in the workspace overflow - so reading either through the 16-bit map cannot work */
247 if (pico9918_gpu_reg_value(PICO9918_INST 0) != MARKER)
248 fail("gpu-r0", MARKER, pico9918_gpu_reg_value(PICO9918_INST 0));
249
250 tms9918->vram.map.wrksp[0] = 0x12;
251 tms9918->vram.map.wrksp[1] = 0x34;
252 if (pico9918_gpu_reg_value(PICO9918_INST 1) != 0x1234)
253 fail("gpu-r1", 0x1234, pico9918_gpu_reg_value(PICO9918_INST 1));
254
255 /* the accessor publishes what STST stored, bit for bit - a non-zero check cannot see
256 the two disagreeing by the eight places the low-byte storage sits at */
258 fail("gpu-status-stst", pico9918_gpu_reg_value(PICO9918_INST 3),
260
261 /* and it is what >BEEF earns from a cleared status: logical greater than alone, since
262 arming zeroed it, MOV keeps only C/OV/P across, and >BEEF is negative and non-zero */
265
266 /* and it moved: a run leaves the point it reached, not the address it was armed at */
267 if (pico9918_gpu_pc(PICO9918_INST_ONLY) == PROGRAM_AT)
268 fail("pc-did-not-move", 0, pico9918_gpu_pc(PICO9918_INST_ONLY));
269
270 /* 3. a rate runs it from inside the arming write, which is the whole point */
272 loadProgram();
273 arm();
274 if (result() != MARKER) fail("armed-not-run", MARKER, result());
275
276 /* 4. the other arming route: R56 bit 0. It resumes from gpuAddress rather than
277 restarting, so R55 sets the address with no rate on - arming without running -
278 and the R56 write is what starts it. */
280 loadProgram();
281 arm();
282 if (result() != 0) fail("r56-early-run", 0, result());
283
285 regWrite(0x38, 1);
286 if (result() != MARKER) fail("r56-not-run", MARKER, result());
287
288 /* 5. a locked device has no GPU to arm - a reset re-locks it. The registers that
289 would arm one are above the eight it admits, so the write is ignored outright,
290 and the rate is still set so this is the lock doing it. */
292
293 /* the reset parks an odd address, which is what the header calls "nothing armed" */
294 if ((pico9918_gpu_pc(PICO9918_INST_ONLY) & 1) == 0)
295 fail("reset-pc-even", 1, pico9918_gpu_pc(PICO9918_INST_ONLY));
296
299 loadProgram();
300 arm();
301 if (result() != 0) fail("locked-ran", 0, result());
302
303 /* and the locked arm did not move it: the registers that would are above the eight */
304 if ((pico9918_gpu_pc(PICO9918_INST_ONLY) & 1) == 0)
305 fail("locked-armed-pc", 1, pico9918_gpu_pc(PICO9918_INST_ONLY));
306
307 /* 6. and unlocking again brings it back, so nothing above latched */
308 unlock();
309 loadProgram();
310 arm();
311 if (result() != MARKER) fail("relocked", MARKER, result());
312
313 /* 7. back to zero, back to the host */
315 loadProgram();
316 arm();
317 if (result() != 0) fail("cleared-rate-ran", 0, result());
318
319 /* 8. a workspace the program moved has to survive the slice that stopped after it.
320 Both legs run the same three instructions; the stepped one stops between every
321 pair, which is the only place the workspace can be dropped. */
322 for (int stepped = 0; stepped < 2; ++stepped)
323 {
324 const char* leg = stepped ? "wp-stepped" : "wp-one-slice";
325
326 loadWorkspaceProgram();
327 arm();
328
329 if (pico9918_gpu_wp(PICO9918_INST_ONLY) != PICO9918_GPU_WORKSPACE)
330 fail("wp-armed", PICO9918_GPU_WORKSPACE, pico9918_gpu_wp(PICO9918_INST_ONLY));
331
332 if (stepped)
334 else
336
337 if (pico9918_gpu_wp(PICO9918_INST_ONLY) != NEW_WP)
338 fail(leg, NEW_WP, pico9918_gpu_wp(PICO9918_INST_ONLY));
339
340 expect(stepped ? "wp-stepped-r0-hi" : "wp-one-slice-r0-hi", NEW_WP, MARKER >> 8);
341 expect(stepped ? "wp-stepped-r0-lo" : "wp-one-slice-r0-lo", NEW_WP + 1, MARKER & 0xff);
342 expect(stepped ? "wp-stepped-not-start" : "wp-one-slice-not-start", PICO9918_GPU_WORKSPACE, 0);
343
344 if (pico9918_gpu_reg_value(PICO9918_INST 0) != MARKER)
345 fail(stepped ? "wp-stepped-reg" : "wp-one-slice-reg", MARKER,
347 }
348
349 loadProgram();
350 arm();
351 if (pico9918_gpu_wp(PICO9918_INST_ONLY) != PICO9918_GPU_WORKSPACE)
352 fail("wp-new-program", PICO9918_GPU_WORKSPACE, pico9918_gpu_wp(PICO9918_INST_ONLY));
353
354 /* 9. the writers, which are what make a GPU pane editable. Each is checked through the
355 reader it mirrors and then through a program, because a value that is stored and
356 read back and then dropped on the way into the slice looks exactly like success. */
357 loadWorkspaceProgram();
358 arm();
360
361 pico9918_debug_gpu_set_reg_value(PICO9918_INST 3, 0x1234);
362 if (pico9918_gpu_reg_value(PICO9918_INST 3) != 0x1234)
363 fail("set-reg", 0x1234, pico9918_gpu_reg_value(PICO9918_INST 3));
364 expect("set-reg-hi", NEW_WP + 6, 0x12);
365 expect("set-reg-lo", NEW_WP + 7, 0x34);
366
367 pico9918_debug_gpu_set_reg_value(PICO9918_INST 0x13, 0x5678);
368 if (pico9918_gpu_reg_value(PICO9918_INST 3) != 0x5678)
369 fail("set-reg-masked", 0x5678, pico9918_gpu_reg_value(PICO9918_INST 3));
370
371 if (!pico9918_debug_gpu_set_wp(PICO9918_INST PICO9918_GPU_WORKSPACE)) fail("set-wp-kept", 1, 0);
372
373 pico9918_debug_gpu_set_reg_value(PICO9918_INST 15, MARKER);
374 if (pico9918_gpu_reg_value(PICO9918_INST 15) != MARKER)
375 fail("set-reg-overflow", MARKER, pico9918_gpu_reg_value(PICO9918_INST 15));
376 expect("set-reg-overflow-hi", PICO9918_GPU_WORKSPACE + 30, MARKER >> 8);
377
378 pico9918_debug_gpu_set_status(PICO9918_INST PICO9918_GPU_ST_EQ | PICO9918_GPU_ST_C);
381
382 pico9918_debug_gpu_set_status(PICO9918_INST PICO9918_GPU_ST_OV | 0x00ff);
384 fail("set-status-low-dropped", PICO9918_GPU_ST_OV, pico9918_gpu_status(PICO9918_INST_ONLY));
385
386 loadStatusProgram();
387 arm();
388 pico9918_debug_gpu_set_status(PICO9918_INST PICO9918_GPU_ST_LGT | PICO9918_GPU_ST_P);
391 fail("set-status-ran", PICO9918_GPU_ST_LGT | PICO9918_GPU_ST_P,
393
394 loadEchoProgram();
395 arm();
396 if (!pico9918_debug_gpu_set_wp(PICO9918_INST NEW_WP)) fail("set-wp-armed", 1, 0);
397 if (pico9918_gpu_wp(PICO9918_INST_ONLY) != NEW_WP)
398 fail("set-wp-readback", NEW_WP, pico9918_gpu_wp(PICO9918_INST_ONLY));
399
400 pico9918_debug_gpu_set_reg_value(PICO9918_INST 0, MARKER);
402 if (result() != MARKER) fail("set-wp-ran", MARKER, result());
403 expect("set-wp-not-start", PICO9918_GPU_WORKSPACE, 0);
404
405 arm();
406 if (pico9918_gpu_wp(PICO9918_INST_ONLY) != PICO9918_GPU_WORKSPACE)
407 fail("set-wp-rearmed", PICO9918_GPU_WORKSPACE, pico9918_gpu_wp(PICO9918_INST_ONLY));
408
409 /* 10. the step callback. Four instructions with known addresses, so the count, the
410 order and the address a break stops on are all one program. */
411 stepCalls = 0;
412 stepWrongArgs = 0;
413 stepStopAt = 0xffff;
414 loadProgram();
415 arm();
416
417 if (pico9918_debug_gpu_step_n(PICO9918_INST 0, stepWatch, &stepStopAt))
418 fail("step-cb-still-running", 0, 1);
419 if (stepWrongArgs) fail("step-cb-args", 0, stepWrongArgs);
420 if (stepCalls != 4) fail("step-cb-count", 4, stepCalls);
421 if (stepPc[0] != PROGRAM_AT) fail("step-cb-pc0", PROGRAM_AT, stepPc[0]);
422 if (stepPc[1] != PROGRAM_AT + 4) fail("step-cb-pc1", PROGRAM_AT + 4, stepPc[1]);
423 if (stepPc[2] != PROGRAM_AT + 8) fail("step-cb-pc2", PROGRAM_AT + 8, stepPc[2]);
424 if (stepPc[3] != PROGRAM_AT + 10) fail("step-cb-pc3", PROGRAM_AT + 10, stepPc[3]);
425 if (result() != MARKER) fail("step-cb-ran", MARKER, result());
426
427 /* stopping on the MOV: the PC left behind is the address broken on rather than the one
428 after it, and the store that instruction would have made has not happened */
429 stepCalls = 0;
430 stepStopAt = PROGRAM_AT + 4;
431 loadProgram();
432 arm();
433
434 if (!pico9918_debug_gpu_step_n(PICO9918_INST 0, stepWatch, &stepStopAt))
435 fail("step-cb-break-stopped", 1, 0);
436 if (stepCalls != 2) fail("step-cb-break-count", 2, stepCalls);
437 if (pico9918_gpu_pc(PICO9918_INST_ONLY) != PROGRAM_AT + 4)
438 fail("step-cb-break-pc", PROGRAM_AT + 4, pico9918_gpu_pc(PICO9918_INST_ONLY));
439 if (result() != 0) fail("step-cb-break-early", 0, result());
440
441 /* and the resume starts at the instruction that did not run, not the one after it */
442 stepCalls = 0;
443 stepStopAt = 0xffff;
444
445 if (pico9918_debug_gpu_step_n(PICO9918_INST 0, stepWatch, &stepStopAt))
446 fail("step-cb-resume-running", 0, 1);
447 if (stepCalls != 3) fail("step-cb-resume-count", 3, stepCalls);
448 if (result() != MARKER) fail("step-cb-resume-ran", MARKER, result());
449
450 /* the callback is the call's, not the instance's: neither the plain entry nor a null
451 one may reach the last callback a debugger happened to pass */
452 stepCalls = 0;
453 loadProgram();
454 arm();
456 if (stepCalls) fail("step-cb-persisted", 0, stepCalls);
457 if (result() != MARKER) fail("step-cb-plain-ran", MARKER, result());
458
459 loadProgram();
460 arm();
461 if (pico9918_debug_gpu_step_n(PICO9918_INST 0, NULL, NULL)) fail("step-cb-null-running", 0, 1);
462 if (stepCalls) fail("step-cb-null-called", 0, stepCalls);
463 if (result() != MARKER) fail("step-cb-null-ran", MARKER, result());
464
465#if PICO9918_BUILD_STEP_CALLBACK
466 /* 10b. the same watch armed on the instance, which is what a host that leaves the
467 pacing to the library has instead of a call of its own to pass one to. */
468 armedCalls = 0;
469 armedStopAt = 0xffff;
470 armedWrongArgs = 0;
471 void* armedData = NULL;
472 pico9918_debug_set_step_callback(PICO9918_INST armedWatch, &armedStopAt);
473 if (pico9918_debug_step_callback(PICO9918_INST &armedData) != armedWatch)
474 fail("armed-cb-readback", 1, 0);
475 if (armedData != &armedStopAt) fail("armed-cb-readback-data", 1, 0);
476
477 loadProgram();
478 arm();
480 if (armedWrongArgs) fail("armed-cb-args", 0, armedWrongArgs);
481 if (armedCalls != 4) fail("armed-cb-count", 4, armedCalls);
482 if (result() != MARKER) fail("armed-cb-ran", MARKER, result());
483
484 /* breaking from it stops the slice the same way, with the PC on the instruction that
485 did not run and its store not made */
486 armedCalls = 0;
487 armedStopAt = PROGRAM_AT + 4;
488 loadProgram();
489 arm();
490 if (!pico9918_gpu_step_n(PICO9918_INST 0)) fail("armed-cb-break-stopped", 1, 0);
491 if (armedCalls != 2) fail("armed-cb-break-count", 2, armedCalls);
492 if (pico9918_gpu_pc(PICO9918_INST_ONLY) != PROGRAM_AT + 4)
493 fail("armed-cb-break-pc", PROGRAM_AT + 4, pico9918_gpu_pc(PICO9918_INST_ONLY));
494 if (result() != 0) fail("armed-cb-break-early", 0, result());
495
496 /* RESUMING RE-OFFERS THE ADDRESS THAT BROKE. The slice stopped on that instruction
497 rather than after it, so a host that does not skip its own breakpoint once never
498 gets past the first one - which is the contract EMULATOR-INTEGRATION.md states. */
499 armedCalls = 0;
500 if (!pico9918_gpu_step_n(PICO9918_INST 0)) fail("armed-cb-rebreak-stopped", 1, 0);
501 if (armedCalls != 1) fail("armed-cb-rebreak-count", 1, armedCalls);
502 if (armedPc[0] != PROGRAM_AT + 4) fail("armed-cb-rebreak-pc", PROGRAM_AT + 4, armedPc[0]);
503 if (result() != 0) fail("armed-cb-rebreak-early", 0, result());
504
505 /* and moving it on is all the host owes: the same resume then finishes the program */
506 armedCalls = 0;
507 armedStopAt = 0xffff;
508 if (pico9918_gpu_step_n(PICO9918_INST 0)) fail("armed-cb-rebreak-resume-running", 0, 1);
509 if (armedCalls != 3) fail("armed-cb-rebreak-resume-count", 3, armedCalls);
510 if (result() != MARKER) fail("armed-cb-rebreak-resume-ran", MARKER, result());
511
512 /* and the case the per-call callback cannot reach at all: the run the library starts
513 from inside the arming write */
514 armedCalls = 0;
515 armedStopAt = 0xffff;
517 loadProgram();
518 arm();
519 if (armedCalls != 4) fail("armed-cb-clocked-count", 4, armedCalls);
520 if (result() != MARKER) fail("armed-cb-clocked-ran", MARKER, result());
522
523 /* a reset is the guest's to make and the callback is the host's, so it survives one */
525 if (pico9918_debug_step_callback(PICO9918_INST NULL) != armedWatch) fail("armed-cb-reset", 1, 0);
526
527 /* a call that brings its own wins, so one pane's stepping is not the other's */
528 unlock();
529 armedCalls = 0;
530 stepCalls = 0;
531 stepStopAt = 0xffff;
532 loadProgram();
533 arm();
534 if (pico9918_debug_gpu_step_n(PICO9918_INST 0, stepWatch, &stepStopAt))
535 fail("armed-cb-override-running", 0, 1);
536 if (stepCalls != 4) fail("armed-cb-override-count", 4, stepCalls);
537 if (armedCalls) fail("armed-cb-override-leaked", 0, armedCalls);
538
539 pico9918_debug_set_step_callback(PICO9918_INST NULL, NULL);
540 if (pico9918_debug_step_callback(PICO9918_INST NULL)) fail("armed-cb-disarm", 0, 1);
541
542 armedCalls = 0;
543 loadProgram();
544 arm();
546 if (armedCalls) fail("armed-cb-disarmed-called", 0, armedCalls);
547 if (result() != MARKER) fail("armed-cb-disarmed-ran", MARKER, result());
548#endif
549
550 /* 11. the DMA engine's geometry. The source is 0x40 counting up, so where a byte landed
551 says which one it was and therefore which row and column the engine thought it
552 was on. Zero width and height mean 256; stride is a different animal entirely. */
553 unlock();
555
556 dma(DMA_SRC, DMA_DST, 4, 3, 4, 0x00);
557 expect("dma-run-first", DMA_DST, 0x40);
558 expect("dma-run-last", DMA_DST + 11, 0x4b);
559 expect("dma-run-past", DMA_DST + 12, 0x00);
560
561 dma(DMA_SRC, DMA_DST, 4, 3, 16, 0x00);
562 expect("dma-stride-row0", DMA_DST, 0x40);
563 expect("dma-stride-gap", DMA_DST + 4, 0x00);
564 expect("dma-stride-row1", DMA_DST + 16, 0x50);
565 expect("dma-stride-row2", DMA_DST + 32, 0x60);
566
567 /* stride zero is a pitch of zero, not of 256: every row lands on the one before it */
568 dma(DMA_SRC, DMA_DST, 4, 3, 0, 0x00);
569 expect("dma-stride0-row0", DMA_DST, 0x40);
570 expect("dma-stride0-end", DMA_DST + 4, 0x00);
571 expect("dma-stride0-not256", DMA_DST + 256, 0x00);
572
573 /* and a stride the eight-bit difference overflows walks backwards: 200 with a width of
574 8 is a pitch of -56, not +200 */
575 dma(0x1100, 0x1900, 8, 2, 200, 0x00);
576 expect("dma-back-row0", 0x1900, 0x40);
577 expect("dma-back-row1", 0x18c8, 0x08);
578 expect("dma-back-not-forward", 0x19c8, 0x00);
579
580 /* either side of where it overflows, which for a width of 8 is a stride of 135 */
581 dma(0x1100, 0x1900, 8, 2, 134, 0x00);
582 expect("dma-edge-fwd-row1", 0x1986, 0xc6);
583 expect("dma-edge-fwd-gap", 0x1985, 0x00);
584
585 dma(0x1100, 0x1900, 8, 2, 135, 0x00);
586 expect("dma-edge-back-row1", 0x1887, 0xc7);
587 expect("dma-edge-back-not-forward", 0x1987, 0x00);
588
589 /* a width of zero is 256, and with stride zero the difference wraps to a pitch of 256 */
590 dma(DMA_SRC, DMA_DST, 0, 1, 0, 0x00);
591 expect("dma-width256-first", DMA_DST, 0x40);
592 expect("dma-width256-last", DMA_DST + 255, 0x3f);
593 expect("dma-width256-past", DMA_DST + 256, 0x00);
594
595 /* and a height of zero is 256 rows of it */
596 dma(DMA_SRC, DMA_DST, 1, 0, 1, 0x00);
597 expect("dma-height256-first", DMA_DST, 0x40);
598 expect("dma-height256-last", DMA_DST + 255, 0x3f);
599 expect("dma-height256-past", DMA_DST + 256, 0x00);
600
601 /* the top of each register: 255 wide by one, then one wide by 255 */
602 dma(DMA_SRC, DMA_DST, 255, 1, 255, 0x00);
603 expect("dma-width255-first", DMA_DST, 0x40);
604 expect("dma-width255-last", DMA_DST + 254, 0x3e);
605 expect("dma-width255-past", DMA_DST + 255, 0x00);
606
607 dma(DMA_SRC, DMA_DST, 1, 255, 1, 0x00);
608 expect("dma-height255-first", DMA_DST, 0x40);
609 expect("dma-height255-last", DMA_DST + 254, 0x3e);
610 expect("dma-height255-past", DMA_DST + 255, 0x00);
611
612 /* a stride under the width steps back into the row just written */
613 dma(DMA_SRC, DMA_DST, 8, 2, 4, 0x00);
614 expect("dma-narrow-kept", DMA_DST + 3, 0x43);
615 expect("dma-narrow-rewritten", DMA_DST + 4, 0x44);
616 expect("dma-narrow-last", DMA_DST + 11, 0x4b);
617 expect("dma-narrow-past", DMA_DST + 12, 0x00);
618
619 /* only two bits of the parameter byte are decoded, so the other six say nothing */
620 dma(DMA_SRC, DMA_DST, 4, 3, 4, 0xfc);
621 expect("dma-params-spare-first", DMA_DST, 0x40);
622 expect("dma-params-spare-last", DMA_DST + 11, 0x4b);
623 expect("dma-params-spare-past", DMA_DST + 12, 0x00);
624
625 /* both parameter bits at once: a fill that decrements */
626 dma(0x1100, 0x1900, 4, 2, 4, 0x03);
627 expect("dma-fill-dec-first", 0x1900, 0x40);
628 expect("dma-fill-dec-row0", 0x18fd, 0x40);
629 expect("dma-fill-dec-row1", 0x18f9, 0x40);
630 expect("dma-fill-dec-past", 0x18f8, 0x00);
631
632 /* LOAD-BEARING: overlapping forwards, so each byte read has already been written. A
633 block copy would answer 40..47 here, which is what makes this the guard on the fast
634 path being taken only where source and destination are disjoint. */
635 dma(DMA_SRC, DMA_SRC + 2, 8, 1, 8, 0x00);
636 expect("dma-overlap-0", DMA_SRC + 2, 0x40);
637 expect("dma-overlap-2", DMA_SRC + 4, 0x40);
638 expect("dma-overlap-7", DMA_SRC + 9, 0x41);
639
640 /* sixteen bits of address, either end */
641 dma(DMA_SRC, 0xfffe, 4, 1, 4, 0x00);
642 expect("dma-dst-wrap-before", 0xffff, 0x41);
643 expect("dma-dst-wrap-after", 0x0000, 0x42);
644 expect("dma-dst-wrap-last", 0x0001, 0x43);
645
646 tms9918->vram.bytes[0xfffe] = 0x11;
647 tms9918->vram.bytes[0xffff] = 0x22;
648 tms9918->vram.bytes[0x0000] = 0x33;
649 tms9918->vram.bytes[0x0001] = 0x44;
650 dma(0xfffe, DMA_DST, 4, 1, 4, 0x00);
651 expect("dma-src-wrap-before", DMA_DST + 1, 0x22);
652 expect("dma-src-wrap-after", DMA_DST + 2, 0x33);
653 expect("dma-src-wrap-last", DMA_DST + 3, 0x44);
654
655 /* a fill reads its byte once and strides like a copy */
656 dma(DMA_SRC, DMA_DST, 4, 3, 16, 0x01);
657 expect("dma-fill-row0", DMA_DST + 3, 0x40);
658 expect("dma-fill-gap", DMA_DST + 4, 0x00);
659 expect("dma-fill-row2", DMA_DST + 32, 0x40);
660
661 /* decrementing runs both ends backwards, so a row ends below the address it started at */
662 dma(0x1100, 0x1900, 4, 2, 4, 0x02);
663 expect("dma-dec-row0-first", 0x1900, 0x40);
664 expect("dma-dec-row0-last", 0x18fd, 0x3d);
665 expect("dma-dec-row1-first", 0x18fc, 0x3c);
666 expect("dma-dec-row1-last", 0x18f9, 0x39);
667 expect("dma-dec-past", 0x1901, 0x00);
668
669 /* decrementing with a stride of zero cancels the row's own backwards run exactly */
670 dma(0x1100, 0x1900, 4, 3, 0, 0x02);
671 expect("dma-dec-stride0-first", 0x1900, 0x40);
672 expect("dma-dec-stride0-last", 0x18fd, 0x3d);
673 expect("dma-dec-stride0-end", 0x18fc, 0x00);
674 expect("dma-dec-stride0-past", 0x1901, 0x00);
675
676 /* TRAP: the difference is (width - 1) - stride here, so an ordinary stride always makes
677 the pitch negative and only one that underflows walks the rows FORWARDS while each row
678 is still written backwards. 200 with a width of 8 gives (7 - 200) & 0xff = 63, so +56 -
679 the mirror of dma-back-row1 above, which is the same numbers incrementing. */
680 dma(0x1100, 0x1900, 8, 2, 200, 0x02);
681 expect("dma-dec-fwd-row0-first", 0x1900, 0x40);
682 expect("dma-dec-fwd-row0-last", 0x18f9, 0x39);
683 expect("dma-dec-fwd-row1-first", 0x1938, 0x78);
684 expect("dma-dec-fwd-row1-last", 0x1931, 0x71);
685 expect("dma-dec-fwd-gap", 0x1901, 0x00);
686 expect("dma-dec-fwd-not-back", 0x18c8, 0x00);
687
688 /* and its own edge, which is NOT where incrementing turns over: two's complement holds
689 one more negative than positive, so dec flips at 136 where inc flipped at 135 */
690 dma(0x1100, 0x1900, 8, 2, 135, 0x02);
691 expect("dma-dec-edge-back-row1", 0x1879, 0xb9);
692 expect("dma-dec-edge-back-not-forward", 0x1979, 0x00);
693
694 dma(0x1100, 0x1900, 8, 2, 136, 0x02);
695 expect("dma-dec-edge-fwd-row1", 0x1978, 0xb8);
696 expect("dma-dec-edge-fwd-not-back", 0x1878, 0x00);
697
698 printf("%s: library-paced GPU, %d failure(s)\n", failures ? "FAIL" : "PASS", failures);
699 return failures != 0;
700}
uint16_t pico9918_gpu_status(pico9918_t *tms9918)
see the header.
Definition gpu.c:587
uint16_t pico9918_gpu_pc(pico9918_t *tms9918)
see the header.
Definition gpu.c:540
bool pico9918_debug_gpu_step_n(pico9918_t *tms9918, uint32_t instructions, pico9918_gpu_step_fn cb, void *userdata)
see pico9918_debug.h.
Definition gpu.c:638
void pico9918_gpu_init(pico9918_t *tms9918)
Initialize the TMS9900 GPU.
Definition gpu.c:466
uint32_t pico9918_gpu_mem_size(void)
see the header.
Definition gpu.c:560
uint16_t pico9918_gpu_wp(pico9918_t *tms9918)
see the header.
Definition gpu.c:547
uint8_t pico9918_gpu_mem_value(pico9918_t *tms9918, uint32_t addr)
see the header.
Definition gpu.c:567
uint16_t pico9918_gpu_reg_value(pico9918_t *tms9918, uint8_t reg)
see the header.
Definition gpu.c:577
bool pico9918_gpu_step_n(pico9918_t *tms9918, uint32_t instructions)
The same pass, capped at instructions, returning true while the program still has work left.
Definition gpu.c:600
void pico9918_gpu_set_clock(pico9918_t *tms9918, uint32_t instructionsPerSecond)
Hand GPU execution to the library, at this many instructions a second.
Definition gpu.c:679
pico9918-core - GPU Interface
#define PICO9918_GPU_ST_P
ST5, odd parity.
Definition gpu.h:212
#define PICO9918_GPU_IPS_PRO
PICO9918 PRO, RP2350 at 352MHz.
Definition gpu.h:83
#define PICO9918_GPU_ST_LGT
ST0, logical greater than.
Definition gpu.h:207
#define PICO9918_GPU_ST_OV
ST4, overflow.
Definition gpu.h:211
#define PICO9918_GPU_ST_EQ
ST2, equal.
Definition gpu.h:209
#define PICO9918_GPU_ST_C
ST3, carry.
Definition gpu.h:210
uint8_t pico9918_vram_value(pico9918_t *tms9918, uint16_t addr)
return a value from vram
Definition pico9918.c:3573
void pico9918_reset(pico9918_t *tms9918)
reset the new TMS9918
Definition pico9918.c:378
#define PICO9918_INST_ONLY
pass the instance as the only argument
Definition pico9918.h:86
#define PICO9918_INST
pass the instance ahead of other arguments
Definition pico9918.h:85
pico9918-core - debugger access
pico9918-core - the private instance layout
PICO9918_INTERNAL void pico9918_write_reg_value_impl(pico9918_t *tms9918, uint8_t regSelect, uint8_t value)
set a register from the second byte of a host register write
Definition pico9918.c:3433