--- uae/src/custom.c 2018/04/24 16:56:49 1.1.1.13 +++ uae/src/custom.c 2018/04/24 17:03:59 1.1.1.17 @@ -3,8 +3,9 @@ * * Custom chip emulation * - * Copyright 1995-1998 Bernd Schmidt + * Copyright 1995-2001 Bernd Schmidt * Copyright 1995 Alessandro Bissacco + * Copyright 2000,2001 Toni Wilen */ #include "sysconfig.h" @@ -15,7 +16,7 @@ #include "config.h" #include "options.h" -#include "threaddep/penguin.h" +#include "threaddep/thread.h" #include "uae.h" #include "gensound.h" #include "sounddep/sound.h" @@ -36,6 +37,7 @@ #include "gui.h" #include "picasso96.h" #include "drawing.h" +#include "savestate.h" static unsigned int n_consecutive_skipped = 0; static unsigned int total_skipped = 0; @@ -51,7 +53,7 @@ unsigned int joy0dir, joy1dir; /* Events */ -unsigned long int cycles, nextevent, is_lastline; +unsigned long int currcycle, nextevent, is_lastline; static int rpt_did_reset; struct ev eventtab[ev_max]; @@ -90,9 +92,8 @@ int vblank_hz = VBLANK_HZ_PAL; unsigned long syncbase; static int fmode; static unsigned int beamcon0, new_beamcon0; -static int ntscmode = 0; -#define MAX_SPRITES 32 +#define MAX_SPRITES 8 /* This is but an educated guess. It seems to be correct, but this stuff * isn't documented well. */ @@ -112,21 +113,24 @@ static struct sprite spr[8]; static int sprite_vblank_endline = VBLANK_ENDLINE_NTSC + 2; -static unsigned int sprdata[MAX_SPRITES], sprdatb[MAX_SPRITES], sprctl[MAX_SPRITES], sprpos[MAX_SPRITES]; +static unsigned int sprctl[MAX_SPRITES], sprpos[MAX_SPRITES]; +static uae_u16 sprdata[MAX_SPRITES][4], sprdatb[MAX_SPRITES][4]; static int sprite_last_drawn_at[MAX_SPRITES]; static int last_sprite_point, nr_armed; +static int sprite_width, sprres, sprite_buffer_res; static uae_u32 bpl1dat, bpl2dat, bpl3dat, bpl4dat, bpl5dat, bpl6dat, bpl7dat, bpl8dat; static uae_s16 bpl1mod, bpl2mod; static uaecptr bplpt[8]; uae_u8 *real_bplpt[8]; +/* Used as a debugging aid, to offset any bitplane temporarily. */ +int bpl_off[8]; /*static int blitcount[256]; blitter debug */ static struct color_entry current_colors; static unsigned int bplcon0, bplcon1, bplcon2, bplcon3, bplcon4; -static int nr_planes_from_bplcon0, corrected_nr_planes_from_bplcon0; static unsigned int diwstrt, diwstop, diwhigh; static int diwhigh_written; static unsigned int ddfstrt, ddfstop; @@ -146,28 +150,62 @@ int diwfirstword, diwlastword; static enum diw_states diwstate, hdiwstate; /* Sprite collisions */ -uae_u16 clxdat, clxcon; -int clx_sprmask; +static unsigned int clxdat, clxcon, clxcon2, clxcon_bpl_enable, clxcon_bpl_match; +static int clx_sprmask; enum copper_states { COP_stop, - COP_read1, COP_read2, + COP_read1_in2, + COP_read1_wr_in4, + COP_read1_wr_in2, + COP_read1, + COP_read2_wr_in2, + COP_read2, COP_bltwait, - COP_wait, - COP_wait1 + COP_wait_in4, + COP_wait_in2, + COP_skip_in4, + COP_skip_in2, + COP_wait1, + COP_wait }; struct copper { /* The current instruction words. */ unsigned int i1, i2; + unsigned int saved_i1, saved_i2; enum copper_states state; /* Instruction pointer. */ - uaecptr ip; + uaecptr ip, saved_ip; int hpos, vpos; unsigned int ignore_next; int vcmp, hcmp; + + /* When we schedule a copper event, knowing a few things about the future + of the copper list can reduce the number of sync_with_cpu calls + dramatically. */ + unsigned int first_sync; + unsigned int regtypes_modified; }; +#define REGTYPE_NONE 0 +#define REGTYPE_COLOR 1 +#define REGTYPE_SPRITE 2 +#define REGTYPE_PLANE 4 +#define REGTYPE_BLITTER 8 +#define REGTYPE_JOYPORT 16 +#define REGTYPE_DISK 32 +#define REGTYPE_POS 64 +#define REGTYPE_AUDIO 128 + +#define REGTYPE_ALL 255 +/* Always set in regtypes_modified, to enable a forced update when things like + DMACON, BPLCON0, COPJMPx get written. */ +#define REGTYPE_FORCE 256 + + +static unsigned int regtypes[512]; + static struct copper cop_state; static int copper_enabled_thisline; static int cop_min_waittime; @@ -182,6 +220,25 @@ static unsigned long int seconds_base; int bogusframe; int n_frames; +#define DEBUG_COPPER 0 +#if DEBUG_COPPER +/* 10000 isn't enough! */ +#define NR_COPPER_RECORDS 40000 +#else +#define NR_COPPER_RECORDS 1 +#endif + +/* Record copper activity for the debugger. */ +struct cop_record +{ + int hpos, vpos; + uaecptr addr; +}; +static struct cop_record cop_record[2][NR_COPPER_RECORDS]; +static int nr_cop_records[2]; +static int curr_cop_set; + +/* Recording of custom chip register changes. */ static int current_change_set; #ifdef OS_WITHOUT_MEMORY_MANAGEMENT @@ -218,10 +275,6 @@ static int next_color_change; static int next_color_entry, remembered_color_entry; static int color_src_match, color_dest_match, color_compare_result; -/* These few are only needed during/at the end of the scanline, and don't - * have to be remembered. */ -static int decided_bpl1mod, decided_bpl2mod, decided_nr_planes; - static uae_u32 thisline_changed; #ifdef SMART_UPDATE @@ -231,7 +284,6 @@ static uae_u32 thisline_changed; #endif static struct decision thisline_decision; -static int modulos_added, plane_decided, color_decided; static int passed_plfstop, fetch_cycle; enum fetchstate { @@ -244,6 +296,43 @@ enum fetchstate { * helper functions */ +uae_u32 get_copper_address (int copno) +{ + switch (copno) { + case 1: return cop1lc; + case 2: return cop2lc; + default: return 0; + } +} + +STATIC_INLINE void record_copper (uaecptr addr, int hpos, int vpos) +{ +#if DEBUG_COPPER + int t = nr_cop_records[curr_cop_set]; + if (t < NR_COPPER_RECORDS) { + cop_record[curr_cop_set][t].addr = addr; + cop_record[curr_cop_set][t].hpos = hpos; + cop_record[curr_cop_set][t].vpos = vpos; + nr_cop_records[curr_cop_set] = t + 1; + } +#endif +} + +int find_copper_record (uaecptr addr, int *phpos, int *pvpos) +{ + int s = curr_cop_set ^ 1; + int t = nr_cop_records[s]; + int i; + for (i = 0; i < t; i++) { + if (cop_record[s][i].addr == addr) { + *phpos = cop_record[s][i].hpos; + *pvpos = cop_record[s][i].vpos; + return 1; + } + } + return 0; +} + int rpt_available = 0; void reset_frame_rate_hack (void) @@ -274,7 +363,8 @@ void check_prefs_changed_custom (void) } currprefs.immediate_blits = changed_prefs.immediate_blits; currprefs.blits_32bit_enabled = changed_prefs.blits_32bit_enabled; - + currprefs.collision_level = changed_prefs.collision_level; + currprefs.fast_copper = changed_prefs.fast_copper; } STATIC_INLINE void setclr (uae_u16 *p, uae_u16 val) @@ -287,7 +377,7 @@ STATIC_INLINE void setclr (uae_u16 *p, u __inline__ int current_hpos (void) { - return cycles - eventtab[ev_hsync].oldcycles; + return (get_cycles () - eventtab[ev_hsync].oldcycles) / CYCLE_UNIT; } STATIC_INLINE uae_u8 *pfield_xlateptr (uaecptr plpt, int bytecount) @@ -308,7 +398,7 @@ STATIC_INLINE void docols (struct color_ if (currprefs.chipset_mask & CSMASK_AGA) { for (i = 0; i < 256; i++) { int v = color_reg_get (colentry, i); - if(v < 0 || v > 16777215) + if (v < 0 || v > 16777215) continue; colentry->acolors[i] = CONVERT_RGB (v); } @@ -402,105 +492,283 @@ static void decide_diw (int hpos) static void finish_playfield_line (void) { + int m1, m2; + /* The latter condition might be able to happen in interlaced frames. */ if (vpos >= minfirstline && (thisframe_first_drawn_line == -1 || vpos < thisframe_first_drawn_line)) thisframe_first_drawn_line = vpos; thisframe_last_drawn_line = vpos; + if ((currprefs.chipset_mask & CSMASK_AGA) && (fmode & 0x4000)) { + if (((diwstrt >> 8) ^ vpos) & 1) + m1 = m2 = bpl2mod; + else + m1 = m2 = bpl1mod; + } else { + m1 = bpl1mod; + m2 = bpl2mod; + } + if (dmaen (DMA_BITPLANE)) - switch (nr_planes_from_bplcon0) { - case 8: bplpt[7] += bpl2mod; - case 7: bplpt[6] += bpl1mod; - case 6: bplpt[5] += bpl2mod; - case 5: bplpt[4] += bpl1mod; - case 4: bplpt[3] += bpl2mod; - case 3: bplpt[2] += bpl1mod; - case 2: bplpt[1] += bpl2mod; - case 1: bplpt[0] += bpl1mod; + switch (GET_PLANES (bplcon0)) { + case 8: bplpt[7] += m2; + case 7: bplpt[6] += m1; + case 6: bplpt[5] += m2; + case 5: bplpt[4] += m1; + case 4: bplpt[3] += m2; + case 3: bplpt[2] += m1; + case 2: bplpt[1] += m2; + case 1: bplpt[0] += m1; } /* These are for comparison. */ thisline_decision.bplcon0 = bplcon0; thisline_decision.bplcon2 = bplcon2; + thisline_decision.bplcon3 = bplcon3; thisline_decision.bplcon4 = bplcon4; - thisline_decision.fmode = fmode; #ifdef SMART_UPDATE if (line_decisions[next_lineno].plflinelen != thisline_decision.plflinelen || line_decisions[next_lineno].plfleft != thisline_decision.plfleft || line_decisions[next_lineno].bplcon0 != thisline_decision.bplcon0 || line_decisions[next_lineno].bplcon2 != thisline_decision.bplcon2 + || line_decisions[next_lineno].bplcon3 != thisline_decision.bplcon3 || line_decisions[next_lineno].bplcon4 != thisline_decision.bplcon4 - || line_decisions[next_lineno].fmode != thisline_decision.fmode ) #endif /* SMART_UPDATE */ thisline_changed = 1; } +static int fetchmode; + +/* The fetch unit mainly controls ddf stop. It's the number of cycles that + are contained in an indivisible block during which ddf is active. E.g. + if DDF starts at 0x30, and fetchunit is 8, then possible DDF stops are + 0x30 + n * 8. */ +static int fetchunit, fetchunit_mask; +/* The delay before fetching the same bitplane again. Can be larger than + the number of bitplanes; in that case there are additional empty cycles + with no data fetch (this happens for high fetchmodes and low + resolutions). */ +static int fetchstart, fetchstart_shift, fetchstart_mask; +/* fm_maxplane holds the maximum number of planes possible with the current + fetch mode. This selects the cycle diagram: + 8 planes: 73516240 + 4 planes: 3120 + 2 planes: 10. */ +static int fm_maxplane, fm_maxplane_shift; + +/* The corresponding values, by fetchmode and display resolution. */ +static int fetchunits[] = { 8,8,8,0, 16,8,8,0, 32,16,8,0 }; +static int fetchstarts[] = { 3,2,1,0, 4,3,2,0, 5,4,3,0 }; +static int fm_maxplanes[] = { 3,2,1,0, 3,3,2,0, 3,3,3,0 }; + +static int cycle_diagram_table[3][3][9][32]; +static int *curr_diagram; +static int cycle_sequences[3*8] = { 2,1,2,1,2,1,2,1, 4,2,3,1,4,2,3,1, 8,4,6,2,7,3,5,1 }; + +static void debug_cycle_diagram(void) +{ + int fm, res, planes, cycle, v; + char aa; + + for (fm = 0; fm < 3; fm++) { + write_log ("FMODE %d\n=======\n", fm); + for (res = 0; res <= 2; res++) { + for (planes = 0; planes <= 8; planes++) { + write_log("%d: ",planes); + for (cycle = 0; cycle < 32; cycle++) { + v=cycle_diagram_table[fm][res][planes][cycle]; + if (v==0) aa='-'; else if(v>0) aa='1'; else aa='X'; + write_log("%c",aa); + } + write_log("\n"); + } + write_log("\n"); + } + } + fm=0; +} + +static void create_cycle_diagram_table(void) +{ + int fm, res, cycle, planes, v; + int fetch_start, max_planes; + int *cycle_sequence; + + for (fm = 0; fm <= 2; fm++) { + for (res = 0; res <= 2; res++) { + max_planes = fm_maxplanes[fm * 4 + res]; + fetch_start = 1 << fetchstarts[fm * 4 + res]; + cycle_sequence = &cycle_sequences[(max_planes - 1) * 8]; + max_planes = 1 << max_planes; + for (planes = 0; planes <= 8; planes++) { + for (cycle = 0; cycle < 32; cycle++) + cycle_diagram_table[fm][res][planes][cycle] = -1; + if (planes <= max_planes) { + for (cycle = 0; cycle < fetch_start; cycle++) { + if (cycle < max_planes && planes >= cycle_sequence[cycle & 7]) { + v = 1; + } else { + v = 0; + } + cycle_diagram_table[fm][res][planes][cycle] = v; + } + } + } + } + } +#if 0 + debug_cycle_diagram (); +#endif +} + + +/* Used by the copper. */ static int estimated_last_fetch_cycle; +static int cycle_diagram_shift; static void estimate_last_fetch_cycle (int hpos) { + int fetchunit = fetchunits[fetchmode * 4 + GET_RES (bplcon0)]; + if (! passed_plfstop) { int stop = plfstop < hpos || plfstop > HARD_DDF_STOP ? HARD_DDF_STOP : plfstop; /* We know that fetching is up-to-date up until hpos, so we can use fetch_cycle. */ int fetch_cycle_at_stop = fetch_cycle + (stop - hpos); - int starting_last_block_at = (fetch_cycle_at_stop + 7) & ~7; + int starting_last_block_at = (fetch_cycle_at_stop + fetchunit - 1) & ~(fetchunit - 1); - estimated_last_fetch_cycle = hpos + (starting_last_block_at - fetch_cycle) + 8; + estimated_last_fetch_cycle = hpos + (starting_last_block_at - fetch_cycle) + fetchunit; } else { - int starting_last_block_at = (fetch_cycle + 7) & ~7; + int starting_last_block_at = (fetch_cycle + fetchunit - 1) & ~(fetchunit - 1); if (passed_plfstop == 2) - starting_last_block_at -= 8; + starting_last_block_at -= fetchunit; - estimated_last_fetch_cycle = hpos + (starting_last_block_at - fetch_cycle) + 8; + estimated_last_fetch_cycle = hpos + (starting_last_block_at - fetch_cycle) + fetchunit; } } -static uae_u32 outword[8]; +static uae_u32 outword[MAX_PLANES]; static int out_nbits, out_offs; -static uae_u32 todisplay[8]; - -static uae_u32 fetched[8]; +static uae_u32 todisplay[MAX_PLANES][4]; +static uae_u32 fetched[MAX_PLANES]; +static uae_u32 fetched_aga0[MAX_PLANES]; +static uae_u32 fetched_aga1[MAX_PLANES]; /* Expansions from bplcon0/bplcon1. */ -static int toscr_res, toscr_delay1, toscr_delay2, toscr_nr_planes; +static int toscr_res, toscr_delay1, toscr_delay2, toscr_nr_planes, fetchwidth; -/* The number of bits left from the last fetched words. */ +/* The number of bits left from the last fetched words. + This is an optimization - conceptually, we have to make sure the result is + the same as if toscr is called in each clock cycle. However, to speed this + up, we accumulate display data; this variable keeps track of how much. + Thus, once we do call toscr_nbits (which happens at least every 16 bits), + we can do more work at once. */ static int toscr_nbits; -STATIC_INLINE void maybe_first_bpl1dat (int hpos) -{ - if (thisline_decision.plfleft == -1) - thisline_decision.plfleft = hpos; -} +static int delayoffset; -STATIC_INLINE void move_fetched (int hpos) +STATIC_INLINE void compute_delay_offset (int hpos) { + /* this fixes most horizontal scrolling jerkyness but can't be correct */ + delayoffset = ((hpos - fm_maxplane - 0x18) & fetchstart_mask) << 1; + delayoffset &= ~7; + if (delayoffset & 8) + delayoffset = 8; + else if (delayoffset & 16) + delayoffset = 16; + else if (delayoffset & 32) + delayoffset = 32; + else + delayoffset = 0; } -STATIC_INLINE void fetch (int nr) +static void expand_fmodes (void) { - if (nr < toscr_nr_planes) { - fetched[nr] = chipmem_wget (bplpt[nr]); - bplpt[nr] += 2; - } + int res = GET_RES(bplcon0); + int fm = fetchmode; + fetchunit = fetchunits[fm * 4 + res]; + fetchunit_mask = fetchunit - 1; + fetchstart_shift = fetchstarts[fm * 4 + res]; + fetchstart = 1 << fetchstart_shift; + fetchstart_mask = fetchstart - 1; + fm_maxplane_shift = fm_maxplanes[fm * 4 + res]; + fm_maxplane = 1 << fm_maxplane_shift; } +static int maxplanes_ocs[]={ 6,4,0,0 }; +static int maxplanes_ecs[]={ 6,4,2,0 }; +static int maxplanes_aga[]={ 8,4,2,0, 8,8,4,0, 8,8,8,0 }; + /* Expand bplcon0/bplcon1 into the toscr_xxx variables. */ -static void compute_toscr_delay (void) +static void compute_toscr_delay_1 (void) { int delay1 = (bplcon1 & 0x0f) | ((bplcon1 & 0x0c00) >> 6); int delay2 = ((bplcon1 >> 4) & 0x0f) | (((bplcon1 >> 4) & 0x0c00) >> 6); int delaymask; + int fetchwidth = 16 << fetchmode; - toscr_res = GET_RES (bplcon0); - delaymask = ((16 << fetchmode) - 1) >> toscr_res; - + delay1 += delayoffset; + delay2 += delayoffset; + delaymask = (fetchwidth - 1) >> toscr_res; toscr_delay1 = (delay1 & delaymask) << toscr_res; toscr_delay2 = (delay2 & delaymask) << toscr_res; +} + +static void compute_toscr_delay (int hpos) +{ + int v = bplcon0; + int *planes; + + if (currprefs.chipset_mask & CSMASK_AGA) + planes = maxplanes_aga; + else if (! (currprefs.chipset_mask & CSMASK_ECS_DENISE)) + planes = maxplanes_ocs; + else + planes = maxplanes_ecs; + /* Disable bitplane DMA if planes > maxplanes. This is needed e.g. by the + Sanity WOC demo (at the "Party Effect"). */ + if (GET_PLANES(v) > planes[fetchmode*4 + GET_RES (v)]) + v &= ~0x7010; + toscr_res = GET_RES (v); + + toscr_nr_planes = GET_PLANES (v); - toscr_nr_planes = GET_PLANES (bplcon0); + compute_toscr_delay_1 (); +} + +STATIC_INLINE void maybe_first_bpl1dat (int hpos) +{ + if (thisline_decision.plfleft == -1) { + thisline_decision.plfleft = hpos; + compute_delay_offset (hpos); + compute_toscr_delay_1 (); + } +} + +STATIC_INLINE void fetch (int nr, int fm) +{ + uaecptr p; + if (nr >= toscr_nr_planes) + return; + p = bplpt[nr] + bpl_off[nr]; + switch (fm) { + case 0: + fetched[nr] = chipmem_wget (p); + bplpt[nr] += 2; + break; + case 1: + fetched_aga0[nr] = chipmem_lget (p); + bplpt[nr] += 4; + break; + case 2: + fetched_aga1[nr] = chipmem_lget (p); + fetched_aga0[nr] = chipmem_lget (p + 4); + bplpt[nr] += 8; + break; + } + if (nr == 0) + fetch_state = fetch_was_plane0; } static void clear_fetchbuffer (uae_u32 *ptr, int nwords) @@ -531,7 +799,7 @@ static void update_toscr_planes (void) } } -static void toscr_1 (int nbits) +STATIC_INLINE void toscr_3_ecs (int nbits) { int delay1 = toscr_delay1; int delay2 = toscr_delay2; @@ -540,16 +808,94 @@ static void toscr_1 (int nbits) for (i = 0; i < toscr_nr_planes; i += 2) { outword[i] <<= nbits; - outword[i] |= (todisplay[i] >> (16 - nbits + delay1)) & mask; - todisplay[i] <<= nbits; + outword[i] |= (todisplay[i][0] >> (16 - nbits + delay1)) & mask; + todisplay[i][0] <<= nbits; } for (i = 1; i < toscr_nr_planes; i += 2) { outword[i] <<= nbits; - outword[i] |= (todisplay[i] >> (16 - nbits + delay2)) & mask; - todisplay[i] <<= nbits; + outword[i] |= (todisplay[i][0] >> (16 - nbits + delay2)) & mask; + todisplay[i][0] <<= nbits; + } +} + +STATIC_INLINE void shift32plus (uae_u32 *p, int n) +{ + uae_u32 t = p[1]; + t <<= n; + t |= p[0] >> (32 - n); + p[1] = t; +} + +STATIC_INLINE void aga_shift (uae_u32 *p, int n, int fm) +{ + if (fm == 2) { + shift32plus (p + 2, n); + shift32plus (p + 1, n); + } + shift32plus (p + 0, n); + p[0] <<= n; +} + +STATIC_INLINE void toscr_3_aga (int nbits, int fm) +{ + int delay1 = toscr_delay1; + int delay2 = toscr_delay2; + int i; + uae_u32 mask = 0xFFFF >> (16 - nbits); + + { + int offs = (16 << fm) - nbits + delay1; + int off1 = offs >> 5; + if (off1 == 3) + off1 = 2; + offs -= off1 * 32; + for (i = 0; i < toscr_nr_planes; i += 2) { + uae_u32 t0 = todisplay[i][off1]; + uae_u32 t1 = todisplay[i][off1 + 1]; + uae_u64 t = (((uae_u64)t1) << 32) | t0; + outword[i] <<= nbits; + outword[i] |= (t >> offs) & mask; + aga_shift (todisplay[i], nbits, fm); + } + } + { + int offs = (16 << fm) - nbits + delay2; + int off1 = offs >> 5; + if (off1 == 3) + off1 = 2; + offs -= off1 * 32; + for (i = 1; i < toscr_nr_planes; i += 2) { + uae_u32 t0 = todisplay[i][off1]; + uae_u32 t1 = todisplay[i][off1 + 1]; + uae_u64 t = (((uae_u64)t1) << 32) | t0; + outword[i] <<= nbits; + outword[i] |= (t >> offs) & mask; + aga_shift (todisplay[i], nbits, fm); + } + } +} + +static void toscr_2_0 (int nbits) { toscr_3_ecs (nbits); } +static void toscr_2_1 (int nbits) { toscr_3_aga (nbits, 1); } +static void toscr_2_2 (int nbits) { toscr_3_aga (nbits, 2); } + +STATIC_INLINE void toscr_1 (int nbits, int fm) +{ + switch (fm) { + case 0: + toscr_2_0 (nbits); + break; + case 1: + toscr_2_1 (nbits); + break; + case 2: + toscr_2_2 (nbits); + break; } + out_nbits += nbits; if (out_nbits == 32) { + int i; uae_u8 *dataptr = line_data[next_lineno] + out_offs * 4; /* Don't use toscr_nr_planes here; if the plane count drops during the line we still want the data to be correct for the full number of planes @@ -566,83 +912,130 @@ static void toscr_1 (int nbits) } } -static void toscr (int nbits) +static void toscr_fm0 (int); +static void toscr_fm1 (int); +static void toscr_fm2 (int); + +STATIC_INLINE void toscr (int nbits, int fm) +{ + switch (fm) { + case 0: toscr_fm0 (nbits); break; + case 1: toscr_fm1 (nbits); break; + case 2: toscr_fm2 (nbits); break; + } +} + +STATIC_INLINE void toscr_0 (int nbits, int fm) { int t; + if (nbits > 16) { - toscr (16); + toscr (16, fm); nbits -= 16; } t = 32 - out_nbits; if (t < nbits) { - toscr_1 (t); + toscr_1 (t, fm); nbits -= t; } - toscr_1 (nbits); + toscr_1 (nbits, fm); } - -static int flush_plane_data (void) + +static void toscr_fm0 (int nbits) { toscr_0 (nbits, 0); } +static void toscr_fm1 (int nbits) { toscr_0 (nbits, 1); } +static void toscr_fm2 (int nbits) { toscr_0 (nbits, 2); } + +static int flush_plane_data (int fm) { int i = 0; + int fetchwidth = 16 << fm; if (out_nbits <= 16) { i += 16; - toscr_1 (16); + toscr_1 (16, fm); } if (out_nbits != 0) { i += 32 - out_nbits; - toscr_1 (32 - out_nbits); + toscr_1 (32 - out_nbits, fm); } i += 32; - toscr_1 (16); - toscr_1 (16); + toscr_1 (16, fm); + toscr_1 (16, fm); return i >> (1 + toscr_res); } -/* The usual inlining tricks - don't touch unless you know what you are doing. */ -STATIC_INLINE void long_fetch (int plane, int nwords, int weird_number_of_bits) +STATIC_INLINE void flush_display (int fm) +{ + if (toscr_nbits > 0 && thisline_decision.plfleft != -1) + toscr (toscr_nbits, fm); + toscr_nbits = 0; +} + +/* Called when all planes have been fetched, i.e. when a new block + of data is available to be displayed. The data in fetched[] is + moved into todisplay[]. */ +STATIC_INLINE void beginning_of_plane_block (int pos, int dma, int fm) { - uae_u16 *real_pt = (uae_u16 *)pfield_xlateptr (bplpt[plane], nwords * 2); - int delay = ((plane & 1) ? toscr_delay2 : toscr_delay1); int i; + + flush_display (fm); + + if (fm == 0) + for (i = 0; i < MAX_PLANES; i++) + todisplay[i][0] |= fetched[i]; + else + for (i = 0; i < MAX_PLANES; i++) { + if (fm == 2) + todisplay[i][1] = fetched_aga1[i]; + todisplay[i][0] = fetched_aga0[i]; + } + + maybe_first_bpl1dat (pos); +} + +#define SPEEDUP + +#ifdef SPEEDUP + +/* The usual inlining tricks - don't touch unless you know what you are doing. */ +STATIC_INLINE void long_fetch_ecs (int plane, int nwords, int weird_number_of_bits, int dma) +{ + uae_u16 *real_pt = (uae_u16 *)pfield_xlateptr (bplpt[plane] + bpl_off[plane], nwords * 2); + int delay = ((plane & 1) ? toscr_delay2 : toscr_delay1); int tmp_nbits = out_nbits; - uae_u32 shiftbuffer = todisplay[plane]; + uae_u32 shiftbuffer = todisplay[plane][0]; uae_u32 outval = outword[plane]; uae_u32 fetchval = fetched[plane]; uae_u32 *dataptr = (uae_u32 *)(line_data[next_lineno] + 2 * plane * MAX_WORDS_PER_LINE + 4 * out_offs); - bplpt[plane] += nwords * 2; + if (dma) + bplpt[plane] += nwords * 2; if (real_pt == 0) /* @@@ Don't do this, fall back on chipmem_wget instead. */ return; - if (thisline_decision.plfleft == -1) { - nwords--; - fetchval = do_get_mem_word (real_pt); - real_pt += 2; - } - while (nwords > 0) { int bits_left = 32 - tmp_nbits; + uae_u32 t; shiftbuffer |= fetchval; + t = (shiftbuffer >> delay) & 0xFFFF; + if (weird_number_of_bits && bits_left < 16) { outval <<= bits_left; - outval |= (shiftbuffer >> (16 + delay - bits_left)) & (0xFFFF >> (16 - bits_left)); - shiftbuffer <<= bits_left; + outval |= t >> (16 - bits_left); thisline_changed |= *dataptr ^ outval; *dataptr++ = outval; + outval = t; tmp_nbits = 16 - bits_left; - outval = (shiftbuffer >> (16 + delay - tmp_nbits)) & (0xFFFF >> (16 - tmp_nbits)); - shiftbuffer <<= tmp_nbits; + shiftbuffer <<= 16; } else { - outval <<= 16; - outval |= (shiftbuffer >> delay) & 0xFFFF; + outval = (outval << 16) | t; shiftbuffer <<= 16; tmp_nbits += 16; if (tmp_nbits == 32) { @@ -651,100 +1044,236 @@ STATIC_INLINE void long_fetch (int plane tmp_nbits = 0; } } - - fetchval = do_get_mem_word (real_pt); - nwords--; - real_pt++; + if (dma) { + fetchval = do_get_mem_word (real_pt); + real_pt++; + } } fetched[plane] = fetchval; - todisplay[plane] = shiftbuffer; + todisplay[plane][0] = shiftbuffer; + outword[plane] = outval; +} + +STATIC_INLINE void long_fetch_aga (int plane, int nwords, int weird_number_of_bits, int fm, int dma) +{ + uae_u32 *real_pt = (uae_u32 *)pfield_xlateptr (bplpt[plane] + bpl_off[plane], nwords * 2); + int delay = ((plane & 1) ? toscr_delay2 : toscr_delay1); + int tmp_nbits = out_nbits; + uae_u32 *shiftbuffer = todisplay[plane]; + uae_u32 outval = outword[plane]; + uae_u32 fetchval0 = fetched_aga0[plane]; + uae_u32 fetchval1 = fetched_aga1[plane]; + uae_u32 *dataptr = (uae_u32 *)(line_data[next_lineno] + 2 * plane * MAX_WORDS_PER_LINE + 4 * out_offs); + int offs = (16 << fm) - 16 + delay; + int off1 = offs >> 5; + if (off1 == 3) + off1 = 2; + offs -= off1 * 32; + + if (dma) + bplpt[plane] += nwords * 2; + + if (real_pt == 0) + /* @@@ Don't do this, fall back on chipmem_wget instead. */ + return; + + while (nwords > 0) { + int i; + + shiftbuffer[0] = fetchval0; + if (fm == 2) + shiftbuffer[1] = fetchval1; + + for (i = 0; i < (1 << fm); i++) { + int bits_left = 32 - tmp_nbits; + + uae_u32 t0 = shiftbuffer[off1]; + uae_u32 t1 = shiftbuffer[off1 + 1]; + uae_u64 t = (((uae_u64)t1) << 32) | t0; + + t0 = (t >> offs) & 0xFFFF; + + if (weird_number_of_bits && bits_left < 16) { + outval <<= bits_left; + outval |= t0 >> (16 - bits_left); + + thisline_changed |= *dataptr ^ outval; + *dataptr++ = outval; + + outval = t0; + tmp_nbits = 16 - bits_left; + aga_shift (shiftbuffer, 16, fm); + } else { + outval = (outval << 16) | t0; + aga_shift (shiftbuffer, 16, fm); + tmp_nbits += 16; + if (tmp_nbits == 32) { + thisline_changed |= *dataptr ^ outval; + *dataptr++ = outval; + tmp_nbits = 0; + } + } + } + + nwords -= 1 << fm; + + if (dma) { + if (fm == 1) + fetchval0 = do_get_mem_long (real_pt); + else { + fetchval1 = do_get_mem_long (real_pt); + fetchval0 = do_get_mem_long (real_pt + 1); + } + real_pt += fm; + } + } + fetched_aga0[plane] = fetchval0; + fetched_aga1[plane] = fetchval1; outword[plane] = outval; } -static void long_fetch_0 (int hpos, int nwords) { long_fetch (hpos, nwords, 0); } -static void long_fetch_1 (int hpos, int nwords) { long_fetch (hpos, nwords, 1); } +static void long_fetch_ecs_0 (int hpos, int nwords, int dma) { long_fetch_ecs (hpos, nwords, 0, dma); } +static void long_fetch_ecs_1 (int hpos, int nwords, int dma) { long_fetch_ecs (hpos, nwords, 1, dma); } +static void long_fetch_aga_1_0 (int hpos, int nwords, int dma) { long_fetch_aga (hpos, nwords, 0, 1, dma); } +static void long_fetch_aga_1_1 (int hpos, int nwords, int dma) { long_fetch_aga (hpos, nwords, 1, 1, dma); } +static void long_fetch_aga_2_0 (int hpos, int nwords, int dma) { long_fetch_aga (hpos, nwords, 0, 2, dma); } +static void long_fetch_aga_2_1 (int hpos, int nwords, int dma) { long_fetch_aga (hpos, nwords, 1, 2, dma); } -static void do_long_fetch (int hpos, int nwords) +static void do_long_fetch (int hpos, int nwords, int dma, int fm) { + int added; int i; - if (out_nbits & 15) { - for (i = 0; i < toscr_nr_planes; i++) - long_fetch_1 (i, nwords); - } else { - for (i = 0; i < toscr_nr_planes; i++) - long_fetch_0 (i, nwords); + flush_display (fm); + switch (fm) { + case 0: + if (out_nbits & 15) { + for (i = 0; i < toscr_nr_planes; i++) + long_fetch_ecs_1 (i, nwords, dma); + } else { + for (i = 0; i < toscr_nr_planes; i++) + long_fetch_ecs_0 (i, nwords, dma); + } + break; + case 1: + if (out_nbits & 15) { + for (i = 0; i < toscr_nr_planes; i++) + long_fetch_aga_1_1 (i, nwords, dma); + } else { + for (i = 0; i < toscr_nr_planes; i++) + long_fetch_aga_1_0 (i, nwords, dma); + } + break; + case 2: + if (out_nbits & 15) { + for (i = 0; i < toscr_nr_planes; i++) + long_fetch_aga_2_1 (i, nwords, dma); + } else { + for (i = 0; i < toscr_nr_planes; i++) + long_fetch_aga_2_0 (i, nwords, dma); + } + break; } - if (thisline_decision.plfleft == -1) - nwords--; - - out_nbits += 16 * nwords; + out_nbits += nwords * 16; out_offs += out_nbits >> 5; out_nbits &= 31; + + if (dma && toscr_nr_planes > 0) + fetch_state = fetch_was_plane0; } -//#define SPEEDUP +#endif + +/* make sure fetch that goes beyond maxhpos is finished */ +static void finish_final_fetch (int i, int fm) +{ + passed_plfstop = 3; + + if (thisline_decision.plfleft != -1) { + i += flush_plane_data (fm); + thisline_decision.plfright = i; + thisline_decision.plflinelen = out_offs; + thisline_decision.bplres = toscr_res; + finish_playfield_line (); + } +} -STATIC_INLINE int one_fetch_cycle (int i, int ddfstop_to_test, int unit, int unit_mask, int dma) +STATIC_INLINE int one_fetch_cycle_0 (int i, int ddfstop_to_test, int dma, int fm) { if (! passed_plfstop && i == ddfstop_to_test) passed_plfstop = 1; - if ((fetch_cycle & 7) == 0) { + if ((fetch_cycle & fetchunit_mask) == 0) { if (passed_plfstop == 2) { - passed_plfstop = 3; - - i += flush_plane_data (); - - thisline_decision.plfright = i; - thisline_decision.plflinelen = out_offs; - thisline_decision.bplres = toscr_res; - - finish_playfield_line (); + finish_final_fetch (i, fm); return 1; } if (passed_plfstop) passed_plfstop++; } if (dma) { - int masked_cycle = fetch_cycle & unit_mask; - - switch (masked_cycle) { - case 0: fetch (7 >> toscr_res); break; - case 1: fetch (3 >> toscr_res); break; - case 2: fetch (5 >> toscr_res); break; - case 3: fetch (1 >> toscr_res); break; - case 4: fetch (6); break; - case 5: fetch (2); break; - case 6: fetch (4); break; - case 7: fetch (0); break; + /* fetchstart_mask can be larger than fm_maxplane if FMODE > 0. This means + that the remaining cycles are idle; we'll fall through the whole switch + without doing anything. */ + int cycle_start = fetch_cycle & fetchstart_mask; + switch (fm_maxplane) { + case 8: + switch (cycle_start) { + case 0: fetch (7, fm); break; + case 1: fetch (3, fm); break; + case 2: fetch (5, fm); break; + case 3: fetch (1, fm); break; + case 4: fetch (6, fm); break; + case 5: fetch (2, fm); break; + case 6: fetch (4, fm); break; + case 7: fetch (0, fm); break; + } + break; + case 4: + switch (cycle_start) { + case 0: fetch (3, fm); break; + case 1: fetch (1, fm); break; + case 2: fetch (2, fm); break; + case 3: fetch (0, fm); break; + } + break; + case 2: + switch (cycle_start) { + case 0: fetch (1, fm); break; + case 1: fetch (0, fm); break; + } + break; } - if (masked_cycle == unit_mask && toscr_nr_planes > 0) - fetch_state = fetch_was_plane0; } fetch_cycle++; toscr_nbits += 2 << toscr_res; + + if (toscr_nbits == 16) + flush_display (fm); + if (toscr_nbits > 16) + abort (); + return 0; } -STATIC_INLINE void beginning_of_unit_block (int pos, int dma) -{ - int i; - if (toscr_nbits > 0 && thisline_decision.plfleft != -1) - toscr (toscr_nbits); - toscr_nbits = 0; - - for (i = 0; i < 8; i++) - todisplay[i] |= fetched[i]; +static int one_fetch_cycle_fm0 (int i, int ddfstop_to_test, int dma) { return one_fetch_cycle_0 (i, ddfstop_to_test, dma, 0); } +static int one_fetch_cycle_fm1 (int i, int ddfstop_to_test, int dma) { return one_fetch_cycle_0 (i, ddfstop_to_test, dma, 1); } +static int one_fetch_cycle_fm2 (int i, int ddfstop_to_test, int dma) { return one_fetch_cycle_0 (i, ddfstop_to_test, dma, 2); } - maybe_first_bpl1dat (pos); +STATIC_INLINE int one_fetch_cycle (int i, int ddfstop_to_test, int dma, int fm) +{ + switch (fm) { + case 0: return one_fetch_cycle_fm0 (i, ddfstop_to_test, dma); + case 1: return one_fetch_cycle_fm1 (i, ddfstop_to_test, dma); + case 2: return one_fetch_cycle_fm2 (i, ddfstop_to_test, dma); + default: abort (); + } } -STATIC_INLINE void update_fetch (int until) +STATIC_INLINE void update_fetch (int until, int fm) { - int unit, unit_mask; int pos; int dma = dmaen (DMA_BITPLANE); @@ -760,92 +1289,114 @@ STATIC_INLINE void update_fetch (int unt if (ddfstop >= last_fetch_hpos && ddfstop < HARD_DDF_STOP) ddfstop_to_test = ddfstop; - compute_toscr_delay (); + compute_toscr_delay (last_fetch_hpos); update_toscr_planes (); - unit = 8 >> toscr_res; - unit_mask = unit - 1; - - /* @@@ This needs updating for different FMODEs in AGA. */ - pos = last_fetch_hpos; + cycle_diagram_shift = (last_fetch_hpos - fetch_cycle) & fetchstart_mask; - /* First, finish one fetch block if we already started one. */ + /* First, a loop that prepares us for the speedup code. We want to enter + the SPEEDUP case with fetch_state == fetch_was_plane0, and then unroll + whole blocks, so that we end on the same fetch_state again. */ for (; ; pos++) { - if (pos == until) - goto out; + if (pos == until) { + if (until >= maxhpos && passed_plfstop == 2) { + finish_final_fetch (pos, fm); + return; + } + flush_display (fm); + return; + } - if ((fetch_cycle & 7) == 0) + if (fetch_state == fetch_was_plane0) break; - if (fetch_state == fetch_was_plane0) - beginning_of_unit_block (pos, dma); fetch_state = fetch_started; - - if (one_fetch_cycle (pos, ddfstop_to_test, unit, unit_mask, dma)) + if (one_fetch_cycle (pos, ddfstop_to_test, dma, fm)) return; } - /* Now, our position is aligned to the fetch block boundaries. */ - #ifdef SPEEDUP /* Unrolled version of the for loop below. */ - if (! passed_plfstop && toscr_nr_planes == thisline_decision.nr_planes) { - int stop = until < ddfstop_to_test + 0 * unit_mask ? until : ddfstop_to_test + 0 * unit_mask; - int count = stop - pos; - - if (count >= unit) { - if (toscr_nbits > 0 && thisline_decision.plfleft != -1) - toscr (toscr_nbits); - toscr_nbits = 0; - - if (fetch_state == fetch_was_plane0) - beginning_of_unit_block (pos, dma); + if (! passed_plfstop + && dma + && (fetch_cycle & fetchstart_mask) == (fm_maxplane & fetchstart_mask) +# if 0 + /* @@@ We handle this case, but the code would be simpler if we + * disallowed it - it may even be possible to guarantee that + * this condition never is false. Later. */ + && (out_nbits & 15) == 0 +# endif + && toscr_nr_planes == thisline_decision.nr_planes) + { + int offs = (pos - fetch_cycle) & fetchunit_mask; + int ddf2 = ((ddfstop_to_test - offs + fetchunit - 1) & ~fetchunit_mask) + offs; + int ddf3 = ddf2 + fetchunit; + int stop = until < ddf2 ? until : until < ddf3 ? ddf2 : ddf3; + int count; + + count = stop - pos; + + if (count >= fetchstart) { + count &= ~fetchstart_mask; + + if (thisline_decision.plfleft == -1) { + compute_delay_offset (pos); + compute_toscr_delay_1 (); + } + do_long_fetch (pos, count >> (3 - toscr_res), dma, fm); - do_long_fetch (pos, count >> (3 - toscr_res)); - - if (toscr_nr_planes > 0) - maybe_first_bpl1dat (pos + unit); + /* This must come _after_ do_long_fetch so as not to confuse flush_display + into thinking the first fetch has produced any output worth emitting to + the screen. But the calculation of delay_offset must happen _before_. */ + maybe_first_bpl1dat (pos); - count &= ~unit_mask; if (pos <= ddfstop_to_test && pos + count > ddfstop_to_test) passed_plfstop = 1; - + if (pos <= ddfstop_to_test && pos + count > ddf2) + passed_plfstop = 2; pos += count; fetch_cycle += count; - fetch_state = fetch_was_plane0; - if (pos == until) - return; } } #endif - /* We enter this with I aligned to unit cycles (i.e. (fetch_cycle & unit_mask) == 0. */ for (; pos < until; pos++) { if (fetch_state == fetch_was_plane0) - beginning_of_unit_block (pos, dma); + beginning_of_plane_block (pos, dma, fm); fetch_state = fetch_started; - if (one_fetch_cycle (pos, ddfstop_to_test, unit, unit_mask, dma)) + if (one_fetch_cycle (pos, ddfstop_to_test, dma, fm)) return; } - - out: - if (thisline_decision.plfleft != -1 && toscr_nbits > 0) { - toscr (toscr_nbits); + if (until >= maxhpos && passed_plfstop == 2) { + finish_final_fetch (pos, fm); + return; } - toscr_nbits = 0; + flush_display (fm); } +static void update_fetch_0 (int hpos) { update_fetch (hpos, 0); } +static void update_fetch_1 (int hpos) { update_fetch (hpos, 1); } +static void update_fetch_2 (int hpos) { update_fetch (hpos, 2); } + STATIC_INLINE void decide_fetch (int hpos) { - if (fetch_state != fetch_not_started && hpos > last_fetch_hpos) - update_fetch (hpos); + if (fetch_state != fetch_not_started && hpos > last_fetch_hpos) { + switch (fetchmode) { + case 0: update_fetch_0 (hpos); break; + case 1: update_fetch_1 (hpos); break; + case 2: update_fetch_2 (hpos); break; + default: abort (); + } + } last_fetch_hpos = hpos; } /* This function is responsible for turning on datafetch if necessary. */ STATIC_INLINE void decide_line (int hpos) { + if (hpos <= last_decide_line_hpos) + return; if (fetch_state != fetch_not_started) return; @@ -860,7 +1411,11 @@ STATIC_INLINE void decide_line (int hpos diwstate = DIW_waiting_start; } - if (diwstate == DIW_waiting_stop) { + /* If DMA isn't on by the time we reach plfstrt, then there's no + bitplane DMA at all for the whole line. */ + if (dmaen (DMA_BITPLANE) + && diwstate == DIW_waiting_stop) + { fetch_state = fetch_started; fetch_cycle = 0; last_fetch_hpos = plfstrt; @@ -868,17 +1423,18 @@ STATIC_INLINE void decide_line (int hpos out_offs = 0; toscr_nbits = 0; - compute_toscr_delay (); + compute_toscr_delay (last_fetch_hpos); /* If someone already wrote BPL1DAT, clear the area between that point and the real fetch start. */ - if (thisline_decision.plfleft != -1) { - out_nbits = (plfstrt - thisline_decision.plfleft) << (1 + toscr_res); - out_offs = out_nbits >> 5; - out_nbits &= 31; + if (framecnt == 0) { + if (thisline_decision.plfleft != -1) { + out_nbits = (plfstrt - thisline_decision.plfleft) << (1 + toscr_res); + out_offs = out_nbits >> 5; + out_nbits &= 31; + } + update_toscr_planes (); } - update_toscr_planes (); - estimate_last_fetch_cycle (plfstrt); last_decide_line_hpos = hpos; do_sprites (vpos, plfstrt); @@ -918,6 +1474,36 @@ static void record_color_change (int hpo curr_color_changes[next_color_change++].value = value; } +typedef int sprbuf_res_t, cclockres_t, hwres_t, bplres_t; + +static void do_playfield_collisions (void) +{ + uae_u8 *ld = line_data[next_lineno]; + int i; + + if (clxcon_bpl_enable == 0) { + clxdat |= 1; + return; + } + + for (i = thisline_decision.plfleft; i < thisline_decision.plfright; i += 2) { + int j; + uae_u32 total = 0xFFFFFFFF; + for (j = 0; j < 8; j++) { + uae_u32 t = 0; + if ((clxcon_bpl_enable & (1 << j)) == 0) + t = 0xFFFFFFFF; + else if (j < thisline_decision.nr_planes) { + t = *(uae_u32 *)(line_data[next_lineno] + 2 * i + 2 * j * MAX_WORDS_PER_LINE); + t ^= ~(((clxcon_bpl_match >> j) & 1) - 1); + } + total &= t; + } + if (total) + clxdat |= 1; + } +} + /* Sprite-to-sprite collisions are taken care of in record_sprite. This one does playfield/sprite collisions. That's the theory. In practice this doesn't work yet. I also suspect this code @@ -928,30 +1514,42 @@ static void do_sprite_collisions (void) int first = curr_drawinfo[next_lineno].first_sprite_entry; int i; unsigned int collision_mask = clxmask[clxcon >> 12]; - int plf_first_pixel = thisline_decision.plfleft * 2 + DIW_DDF_OFFSET; + int bplres = GET_RES (bplcon0); + hwres_t ddf_left = thisline_decision.plfleft * 2 << bplres; + hwres_t hw_diwlast = coord_window_to_diw_x (thisline_decision.diwlastword); + hwres_t hw_diwfirst = coord_window_to_diw_x (thisline_decision.diwfirstword); + + if (clxcon_bpl_enable == 0) { + clxdat |= 0x1FE; + return; + } for (i = 0; i < nr_sprites; i++) { struct sprite_entry *e = curr_sprite_entries + first + i; - int j; - int minpos = e->pos; - int maxpos = e->max; - - if (maxpos > thisline_decision.diwlastword) - maxpos = thisline_decision.diwlastword; - if (maxpos > thisline_decision.plfright * 2 + DIW_DDF_OFFSET) - maxpos = thisline_decision.plfright * 2 + DIW_DDF_OFFSET; - if (minpos < thisline_decision.diwfirstword) - minpos = thisline_decision.diwfirstword; - if (minpos < plf_first_pixel) - minpos = plf_first_pixel; + sprbuf_res_t j; + sprbuf_res_t minpos = e->pos; + sprbuf_res_t maxpos = e->max; + hwres_t minp1 = minpos >> sprite_buffer_res; + hwres_t maxp1 = maxpos >> sprite_buffer_res; + + if (maxp1 > hw_diwlast) + maxpos = hw_diwlast << sprite_buffer_res; + if (maxp1 > thisline_decision.plfright * 2) + maxpos = thisline_decision.plfright * 2 << sprite_buffer_res; + if (minp1 < hw_diwfirst) + minpos = hw_diwfirst << sprite_buffer_res; + if (minp1 < thisline_decision.plfleft * 2) + minpos = thisline_decision.plfleft * 2 << sprite_buffer_res; for (j = minpos; j < maxpos; j++) { int sprpix = spixels[e->first_pixel + j - e->pos] & collision_mask; int k; + int offs; if (sprpix == 0) continue; + offs = ((j << bplres) >> sprite_buffer_res) - ddf_left; sprpix = sprite_ab_merge[sprpix & 255] | (sprite_ab_merge[sprpix >> 8] << 2); sprpix <<= 1; @@ -959,19 +1557,20 @@ static void do_sprite_collisions (void) for (k = 0; k < 2; k++) { int l; int match = 1; + int planes = ((currprefs.chipset_mask & CSMASK_AGA) ? 8 : 6); - for (l = k; match && l < 6; l += 2) - if (clxcon & (64 << l)) { + for (l = k; match && l < planes; l += 2) { + if (clxcon_bpl_enable & (1 << l)) { int t = 0; if (l < thisline_decision.nr_planes) { - int offs = j - plf_first_pixel; uae_u32 *ldata = (uae_u32 *)(line_data[next_lineno] + 2 * l * MAX_WORDS_PER_LINE); uae_u32 word = ldata[offs >> 5]; t = (word >> (31 - (offs & 31))) & 1; } - if (t != ((clxcon >> l) & 1)) + if (t != ((clxcon_bpl_match >> l) & 1)) match = 0; } + } if (match) clxdat |= sprpix; sprpix <<= 4; @@ -980,51 +1579,112 @@ static void do_sprite_collisions (void) } } +static void expand_sprres (void) +{ + switch ((bplcon3 >> 6) & 3) { + case 0: /* ECS defaults (LORES,HIRES=140ns,SHRES=70ns) */ + if ((currprefs.chipset_mask & CSMASK_ECS_DENISE) && GET_RES (bplcon0) == RES_SUPERHIRES) + sprres = RES_HIRES; + else + sprres = RES_LORES; + break; + case 1: + sprres = RES_LORES; + break; + case 2: + sprres = RES_HIRES; + break; + case 3: + sprres = RES_SUPERHIRES; + break; + } +} + +STATIC_INLINE void record_sprite_1 (uae_u16 *buf, uae_u32 datab, int num, int dbl, + unsigned int mask, int do_collisions, uae_u32 collision_mask) +{ + int j = 0; + while (datab) { + unsigned int tmp = *buf; + unsigned int col = (datab & 3) << (2 * num); + tmp |= col; + if ((j & mask) == 0) + *buf++ = tmp; + if (dbl) + *buf++ = tmp; + j++; + datab >>= 2; + if (do_collisions) { + tmp &= collision_mask; + if (tmp) { + unsigned int shrunk_tmp = sprite_ab_merge[tmp & 255] | (sprite_ab_merge[tmp >> 8] << 2); + clxdat |= sprclx[shrunk_tmp]; + } + } + } +} + /* DATAB contains the sprite data; 16 pixels in two-bit packets. Bits 0/1 determine the color of the leftmost pixel, bits 2/3 the color of the next etc. This function assumes that for all sprites in a given line, SPRXP either - stays equal or increases between successive calls. */ -static void record_sprite (int line, int num, int sprxp, uae_u32 datab, unsigned int ctl) + stays equal or increases between successive calls. + + The data is recorded either in lores pixels (if ECS), or in hires pixels + (if AGA). No support for SHRES sprites. */ + +static void record_sprite (int line, int num, int sprxp, uae_u16 *data, uae_u16 *datb, unsigned int ctl) { struct sprite_entry *e = curr_sprite_entries + next_sprite_entry; int i; int word_offs; uae_u16 *buf; uae_u32 collision_mask; - int collision_enabled; + int width = sprite_width; + int dbl = 0; + unsigned int mask = 0; - if (currprefs.gfx_lores == 0) - sprxp >>= 1; + if (sprres != RES_LORES) + thisline_decision.any_hires_sprites = 1; + + if (currprefs.chipset_mask & CSMASK_AGA) { + width = (width << 1) >> sprres; + dbl = sprite_buffer_res - sprres; + mask = sprres == RES_SUPERHIRES ? 1 : 0; + } /* Try to coalesce entries if they aren't too far apart. */ - if (! next_sprite_forced && e[-1].max + 16 >= sprxp) + if (! next_sprite_forced && e[-1].max + 16 >= sprxp) { e--; - else { + } else { next_sprite_entry++; e->pos = sprxp; e->has_attached = 0; } + if (sprxp < e->pos) abort (); - e->max = sprxp + 16; + + e->max = sprxp + width; e[1].first_pixel = e->first_pixel + ((e->max - e->pos + 3) & ~3); next_sprite_forced = 0; collision_mask = clxmask[clxcon >> 12]; word_offs = e->first_pixel + sprxp - e->pos; - buf = spixels + word_offs; - while (datab) { - unsigned int tmp = *buf; - unsigned int col = (datab & 3) << (2 * num); - tmp |= col; - *buf++ = tmp; - datab >>= 2; - tmp &= collision_mask; - if (tmp) { - unsigned int shrunk_tmp = sprite_ab_merge[tmp & 255] | (sprite_ab_merge[tmp >> 8] << 2); - clxdat |= sprclx[shrunk_tmp]; - } + + for (i = 0; i < sprite_width; i += 16) { + unsigned int da = *data; + unsigned int db = *datb; + uae_u32 datab = ((sprtaba[da & 0xFF] << 16) | sprtaba[da >> 8] + | (sprtabb[db & 0xFF] << 16) | sprtabb[db >> 8]); + + buf = spixels + word_offs + (i << dbl); + if (currprefs.collision_level > 0 && collision_mask) + record_sprite_1 (buf, datab, num, dbl, mask, 1, collision_mask); + else + record_sprite_1 (buf, datab, num, dbl, mask, 0, collision_mask); + data++; + datb++; } /* We have 8 bits per pixel in spixstate, two for every sprite pair. The @@ -1034,22 +1694,17 @@ static void record_sprite (int line, int uae_u32 state = 0x01010101 << (num - 1); uae_u32 *stbuf = spixstate.words + (word_offs >> 2); uae_u8 *stb1 = spixstate.bytes + word_offs; - stb1[0] |= state; - stb1[1] |= state; - stb1[2] |= state; - stb1[3] |= state; - stb1[4] |= state; - stb1[5] |= state; - stb1[6] |= state; - stb1[7] |= state; - stb1[8] |= state; - stb1[9] |= state; - stb1[10] |= state; - stb1[11] |= state; - stb1[12] |= state; - stb1[13] |= state; - stb1[14] |= state; - stb1[15] |= state; + for (i = 0; i < width; i += 8) { + stb1[0] |= state; + stb1[1] |= state; + stb1[2] |= state; + stb1[3] |= state; + stb1[4] |= state; + stb1[5] |= state; + stb1[6] |= state; + stb1[7] |= state; + stb1 += 8; + } e->has_attached = 1; } } @@ -1058,7 +1713,9 @@ static void decide_sprites (int hpos) { int nrs[MAX_SPRITES], posns[MAX_SPRITES]; int count, i; - int point = coord_hw_to_window_x (hpos * 2); + int point = hpos * 2; + int width = sprite_width; + int window_width = (width << lores_shift) >> sprres; if (framecnt != 0 || hpos < 0x14 || nr_armed == 0 || point == last_sprite_point) return; @@ -1072,14 +1729,23 @@ static void decide_sprites (int hpos) return; #endif count = 0; - for (i = 0; i < 8; i++) { + for (i = 0; i < MAX_SPRITES; i++) { int sprxp = spr[i].xpos; + int hw_xp = (sprxp >> sprite_buffer_res); + int window_xp = coord_hw_to_window_x (hw_xp) + (DIW_DDF_OFFSET << lores_shift); int j, bestp; - if (! spr[i].armed || sprxp < 0 || sprxp <= last_sprite_point || sprxp > point) + /* ??? It is uncertain where exactly sprite data is needed. It appears + to be used some time after the actual position given in SPRxPOS. + How soon after is an open question. For Battle Squadron, using a + value of 5 seems to be enough to meet all timing constraints. + This gives it 2 more color clock cycles until the data appears on + the screen. */ + hw_xp += 5; + if (! spr[i].armed || sprxp < 0 || hw_xp <= last_sprite_point || hw_xp > point) continue; - if ((thisline_decision.diwfirstword >= 0 && sprxp + sprite_width < thisline_decision.diwfirstword) - || (thisline_decision.diwlastword >= 0 && sprxp > thisline_decision.diwlastword)) + if ((thisline_decision.diwfirstword >= 0 && window_xp + window_width < thisline_decision.diwfirstword) + || (thisline_decision.diwlastword >= 0 && window_xp > thisline_decision.diwlastword)) continue; /* Sort the sprites in order of ascending X position before recording them. */ @@ -1099,11 +1765,7 @@ static void decide_sprites (int hpos) } for (i = 0; i < count; i++) { int nr = nrs[i]; - unsigned int data = sprdata[nr]; - unsigned int datb = sprdatb[nr]; - uae_u32 datab = ((sprtaba[data & 0xFF] << 16) | sprtaba[data >> 8] - | (sprtabb[datb & 0xFF] << 16) | sprtabb[datb >> 8]); - record_sprite (next_lineno, nr, spr[nr].xpos, datab, sprctl[nr]); + record_sprite (next_lineno, nr, spr[nr].xpos, sprdata[nr], sprdatb[nr], sprctl[nr]); } last_sprite_point = point; } @@ -1232,11 +1894,9 @@ static void reset_decisions (void) if (framecnt != 0) return; + thisline_decision.any_hires_sprites = 0; thisline_decision.nr_planes = 0; - decided_bpl1mod = bpl1mod; - decided_bpl2mod = bpl2mod; - thisline_decision.plfleft = -1; thisline_decision.plflinelen = -1; @@ -1257,14 +1917,13 @@ static void reset_decisions (void) /* memset(sprite_last_drawn_at, 0, sizeof sprite_last_drawn_at); */ last_sprite_point = 0; - modulos_added = 0; - plane_decided = 0; - color_decided = 0; fetch_state = fetch_not_started; passed_plfstop = 0; memset (todisplay, 0, sizeof todisplay); memset (fetched, 0, sizeof fetched); + memset (fetched_aga0, 0, sizeof fetched_aga0); + memset (fetched_aga1, 0, sizeof fetched_aga1); memset (outword, 0, sizeof outword); last_decide_line_hpos = -1; @@ -1301,51 +1960,8 @@ static void init_hz (void) write_log ("Using %s timing\n", isntsc ? "NTSC" : "PAL"); } -#if 0 -void expand_fetchmodes (int fmode, int bplcon0) -{ - int res; - - if (bplcon0 & 0x8000) - res = 1; - else if (bplcon0 & 0x0040) - res = 2; - else - res = 0; - switch (fmode & 3) { - case 3: - fetchmode = 2; - switch (res) { - case 2: prefetch = 1<<3; fetchsize = 1<<3; fetchstart_shift = 3; break; - case 1: prefetch = 1<<3; fetchsize = 1<<4; fetchstart_shift = 4; break; - case 0: prefetch = 1<<3; fetchsize = 1<<5; fetchstart_shift = 5; break; - } - break; - case 2: - case 1: - fetchmode = 1; - switch (res) { - case 2: prefetch = 1<<2; fetchsize = 1<<3; fetchstart_shift = 2; break; - case 1: prefetch = 1<<3; fetchsize = 1<<3; fetchstart_shift = 3; break; - case 0: prefetch = 1<<3; fetchsize = 1<<4; fetchstart_shift = 4; break; - } - break; - case 0: - fetchmode = 0; - switch (res) { - case 2: prefetch = 1<<1; fetchsize = 1<<3; fetchstart_shift = 1; break; - case 1: prefetch = 1<<2; fetchsize = 1<<3; fetchstart_shift = 2; break; - case 0: prefetch = 1<<3; fetchsize = 1<<3; fetchstart_shift = 3; break; - } - break; - } - fetchstart = 1 << fetchstart_shift; -} -#endif - static void calcdiw (void) { - int fetch; int hstrt = diwstrt & 0xFF; int hstop = diwstop & 0xFF; int vstrt = diwstrt >> 8; @@ -1440,24 +2056,33 @@ static uae_u32 mousehack_helper (void) #ifdef PICASSO96 if (picasso_on) { - mousexpos = lastmx - picasso96_state.XOffset; - mouseypos = lastmy - picasso96_state.YOffset; + picasso_clip_mouse (&lastmx, &lastmy); + mousexpos = lastmx; + mouseypos = lastmy; } else #endif { + /* @@@ This isn't completely right, it doesn't deal with virtual + screen sizes larger than physical very well. */ if (lastmy >= gfxvidinfo.height) lastmy = gfxvidinfo.height - 1; + if (lastmy < 0) + lastmy = 0; + if (lastmx < 0) + lastmx = 0; + if (lastmx >= gfxvidinfo.width) + lastmx = gfxvidinfo.width - 1; mouseypos = coord_native_to_amiga_y (lastmy) << 1; mousexpos = coord_native_to_amiga_x (lastmx); } switch (m68k_dreg (regs, 0)) { - case 0: + case 0: return ievent_alive ? -1 : needmousehack (); - case 1: + case 1: ievent_alive = 10; return mousexpos; - case 2: + case 2: return mouseypos; } return 0; @@ -1602,7 +2227,7 @@ STATIC_INLINE uae_u16 ADKCONR (void) } STATIC_INLINE uae_u16 VPOSR (void) { - unsigned int csbit = ntscmode ? 0x1000 : 0; + unsigned int csbit = currprefs.ntscmode ? 0x1000 : 0; csbit |= (currprefs.chipset_mask & CSMASK_AGA) ? 0x2300 : 0; csbit |= (currprefs.chipset_mask & CSMASK_ECS_AGNUS) ? 0x2000 : 0; return (vpos >> 8) | lof | csbit; @@ -1665,7 +2290,7 @@ STATIC_INLINE void COPCON (uae_u16 a) static void DMACON (int hpos, uae_u16 v) { - int i, need_resched = 0; + int i; uae_u16 oldcon = dmacon; @@ -1678,8 +2303,6 @@ static void DMACON (int hpos, uae_u16 v) /* FIXME? Maybe we need to think a bit more about the master DMA enable * bit in these cases. */ if ((dmacon & DMA_COPPER) != (oldcon & DMA_COPPER)) { - if (eventtab[ev_copper].active) - need_resched = 1; eventtab[ev_copper].active = 0; } if ((dmacon & DMA_COPPER) > (oldcon & DMA_COPPER)) { @@ -1708,32 +2331,23 @@ static void DMACON (int hpos, uae_u16 v) if ((dmacon & (DMA_BLITPRI | DMA_BLITTER | DMA_MASTER)) != (DMA_BLITPRI | DMA_BLITTER | DMA_MASTER)) unset_special (SPCFLAG_BLTNASTY); - update_audio (); - - for (i = 0; i < 4; i++) { - struct audio_channel_data *cdp = audio_channel + i; + if (currprefs.produce_sound > 0) { + update_audio (); - cdp->dmaen = (dmacon & 0x200) && (dmacon & (1<dmaen) { - if (cdp->state == 0) { - cdp->state = 1; - cdp->pt = cdp->lc; - cdp->wper = cdp->per; - cdp->wlen = cdp->len; - cdp->data_written = 2; - cdp->evtime = eventtab[ev_hsync].evtime - cycles; - } - } else { - if (cdp->state == 1 || cdp->state == 5) { - cdp->state = 0; - cdp->last_sample = 0; - cdp->current_sample = 0; - } + for (i = 0; i < 4; i++) { + struct audio_channel_data *cdp = audio_channel + i; + int chan_ena = (dmacon & 0x200) && (dmacon & (1<dmaen == chan_ena) + continue; + cdp->dmaen = chan_ena; + if (cdp->dmaen) + audio_channel_enable_dma (cdp); + else + audio_channel_disable_dma (cdp); } + schedule_audio (); } - - if (need_resched) - events_schedule(); + events_schedule(); } /*static int trace_intena = 0;*/ @@ -1743,23 +2357,38 @@ STATIC_INLINE void INTENA (uae_u16 v) /* if (trace_intena) fprintf (stderr, "INTENA: %04x\n", v);*/ setclr (&intena,v); + /* There's stupid code out there that does + [some INTREQ bits at level 3 are set] + clear all INTREQ bits + Enable one INTREQ level 3 bit + Set level 3 handler + + If we set SPCFLAG_INT for the clear, then by the time the enable happens, + we'll have SPCFLAG_DOINT set, and the interrupt happens immediately, but + it needs to happen one insn later, when the new L3 handler has been + installed. */ + if (v & 0x8000) + set_special (SPCFLAG_INT); +} + +void INTREQ_0 (uae_u16 v) +{ + setclr (&intreq,v); set_special (SPCFLAG_INT); } + void INTREQ (uae_u16 v) { - setclr(&intreq,v); - set_special (SPCFLAG_INT); + INTREQ_0 (v); if ((v & 0x8800) == 0x0800) serdat &= 0xbfff; + rethink_cias (); } -static void ADKCON (uae_u16 v) +static void update_adkmasks (void) { unsigned long t; - update_audio (); - - setclr (&adkcon,v); t = adkcon | (adkcon >> 4); audio_channel[0].adk_mask = (((t >> 0) & 1) - 1); audio_channel[1].adk_mask = (((t >> 1) & 1) - 1); @@ -1767,6 +2396,15 @@ static void ADKCON (uae_u16 v) audio_channel[3].adk_mask = (((t >> 3) & 1) - 1); } +static void ADKCON (uae_u16 v) +{ + if (currprefs.produce_sound > 0) + update_audio (); + + setclr (&adkcon,v); + update_adkmasks (); +} + static void BEAMCON0 (uae_u16 v) { new_beamcon0 = v & 0x20; @@ -1787,33 +2425,25 @@ static void BPLPTL (int hpos, uae_u16 v, static void BPLCON0 (int hpos, uae_u16 v) { - if (! (currprefs.chipset_mask & CSMASK_AGA)) { - v &= 0xFF0E; - /* The Sanity WOC demo needs this at one place (at the end of the "Party Effect") - * Disable bitplane DMA if someone tries to do more than 4 Hires bitplanes. */ - if ((v & 0xF000) > 0xC000) - v &= 0xFFF; - /* Don't want 7 lores planes either. */ - if ((v & 0x8000) == 0 && (v & 0x7000) == 0x7000) - v &= 0xEFFF; - } + if (! (currprefs.chipset_mask & CSMASK_ECS_DENISE)) + v &= ~0x00F1; + else if (! (currprefs.chipset_mask & CSMASK_AGA)) + v &= ~0x00B1; + if (bplcon0 == v) return; - decide_line (hpos); decide_fetch (hpos); - /* if ((bplcon0 ^ v) & 0x8000)*/ - calcdiw (); bplcon0 = v; - nr_planes_from_bplcon0 = GET_PLANES (v); + curr_diagram = cycle_diagram_table[fetchmode][GET_RES(bplcon0)][GET_PLANES (v)]; - if (currprefs.chipset_mask & CSMASK_AGA) - /* It's not clear how the copper timings are affected by the number - * of bitplanes on AGA machines */ - corrected_nr_planes_from_bplcon0 = 4; - else - corrected_nr_planes_from_bplcon0 = nr_planes_from_bplcon0 << (bplcon0 & 0x8000 ? 1 : 0); + if (currprefs.chipset_mask & CSMASK_AGA) { + decide_sprites (hpos); + expand_sprres (); + } + + expand_fmodes (); } STATIC_INLINE void BPLCON1 (int hpos, uae_u16 v) @@ -1824,6 +2454,7 @@ STATIC_INLINE void BPLCON1 (int hpos, ua decide_fetch (hpos); bplcon1 = v; } + STATIC_INLINE void BPLCON2 (int hpos, uae_u16 v) { if (bplcon2 == v) @@ -1831,13 +2462,19 @@ STATIC_INLINE void BPLCON2 (int hpos, ua decide_line (hpos); bplcon2 = v; } + STATIC_INLINE void BPLCON3 (int hpos, uae_u16 v) { + if (! (currprefs.chipset_mask & CSMASK_AGA)) + return; if (bplcon3 == v) return; decide_line (hpos); + decide_sprites (hpos); bplcon3 = v; + expand_sprres (); } + STATIC_INLINE void BPLCON4 (int hpos, uae_u16 v) { if (! (currprefs.chipset_mask & CSMASK_AGA)) @@ -1873,14 +2510,6 @@ STATIC_INLINE void BPL1DAT (int hpos, ua decide_line (hpos); bpl1dat = v; - { - static int count = 0; - if (count++ > 1000) { - count = 0; - printf ("BPL1DAT %d\n", hpos); - } - } - maybe_first_bpl1dat (hpos); } /* We could do as well without those... */ @@ -1932,12 +2561,21 @@ static void DDFSTRT (int hpos, uae_u16 v decide_line (hpos); ddfstrt = v; calcdiw (); - if (ddfstop > 0xD4 && (ddfstrt & 4) == 4) - printf ("WARNING! Very strange DDF values.\n"); + if (ddfstop > 0xD4 && (ddfstrt & 4) == 4) { + static int last_warned; + last_warned = (last_warned + 1) & 4095; + if (last_warned == 0) + write_log ("WARNING! Very strange DDF values.\n"); + } } + static void DDFSTOP (int hpos, uae_u16 v) { - v &= 0xFC; + /* ??? "Virtual Meltdown" sets this to 0xD2 and expects it to behave + differently from 0xD0. RSI Megademo sets it to 0xd1 and expects it + to behave like 0xd0. Some people also write the high 8 bits and + expect them to be ignored. So mask it with 0xFE. */ + v &= 0xFE; if (ddfstop == v) return; decide_line (hpos); @@ -1946,17 +2584,36 @@ static void DDFSTOP (int hpos, uae_u16 v calcdiw (); if (fetch_state != fetch_not_started) estimate_last_fetch_cycle (hpos); - if (ddfstop > 0xD4 && (ddfstrt & 4) == 4) - printf ("WARNING! Very strange DDF values.\n"); + if (ddfstop > 0xD4 && (ddfstrt & 4) == 4) { + static int last_warned; + last_warned = (last_warned + 1) & 4095; + if (last_warned == 0) + write_log ("WARNING! Very strange DDF values.\n"); + write_log ("WARNING! Very strange DDF values.\n"); + } } static void FMODE (uae_u16 v) { if (! (currprefs.chipset_mask & CSMASK_AGA)) - return; + v = 0; fmode = v; - calcdiw (); + sprite_width = GET_SPRITEWIDTH (fmode); + switch (fmode & 3) { + case 0: + fetchmode = 0; + break; + case 1: + case 2: + fetchmode = 1; + break; + case 3: + fetchmode = 2; + break; + } + curr_diagram = cycle_diagram_table[fetchmode][GET_RES (v)][GET_PLANES (bplcon0)]; + expand_fmodes (); } static void BLTADAT (uae_u16 v) @@ -2012,12 +2669,10 @@ static void BLTDPTL (uae_u16 v) { maybe_ static void BLTSIZE (uae_u16 v) { - bltsize = v; - maybe_blit (); - blt_info.vblitsize = bltsize >> 6; - blt_info.hblitsize = bltsize & 0x3F; + blt_info.vblitsize = v >> 6; + blt_info.hblitsize = v & 0x3F; if (!blt_info.vblitsize) blt_info.vblitsize = 1024; if (!blt_info.hblitsize) blt_info.hblitsize = 64; @@ -2055,31 +2710,48 @@ STATIC_INLINE void SPRxCTL_1 (uae_u16 v, if (sprpos[num] == 0 && v == 0) { spr[num].state = SPR_stop; spr[num].on = 0; - } else + } else if (spr[num].state != SPR_waiting_stop) { spr[num].state = SPR_waiting_start; + spr[num].on = 1; + } + + sprxp = (sprpos[num] & 0xFF) * 2 + (v & 1); - sprxp = coord_hw_to_window_x ((sprpos[num] & 0xFF) * 2 + (v & 1) + DIW_DDF_OFFSET); + /* Quite a bit salad in this register... */ + if (currprefs.chipset_mask & CSMASK_AGA) { + /* We ignore the SHRES 35ns increment for now; SHRES support doesn't + work anyway, so we may as well restrict AGA sprites to a 70ns + resolution. */ + sprxp <<= 1; + sprxp |= (v >> 4) & 1; + } spr[num].xpos = sprxp; spr[num].vstart = (sprpos[num] >> 8) | ((sprctl[num] << 6) & 0x100); spr[num].vstop = (sprctl[num] >> 8) | ((sprctl[num] << 7) & 0x100); + } STATIC_INLINE void SPRxPOS_1 (uae_u16 v, int num) { int sprxp; sprpos[num] = v; - sprxp = coord_hw_to_window_x ((v & 0xFF) * 2 + (sprctl[num] & 1) + DIW_DDF_OFFSET); + sprxp = (v & 0xFF) * 2 + (sprctl[num] & 1); + + if (currprefs.chipset_mask & CSMASK_AGA) { + sprxp <<= 1; + sprxp |= (sprctl[num] >> 4) & 1; + } spr[num].xpos = sprxp; spr[num].vstart = (sprpos[num] >> 8) | ((sprctl[num] << 6) & 0x100); } STATIC_INLINE void SPRxDATA_1 (uae_u16 v, int num) { - sprdata[num] = v; + sprdata[num][0] = v; nr_armed += 1 - spr[num].armed; spr[num].armed = 1; } STATIC_INLINE void SPRxDATB_1 (uae_u16 v, int num) { - sprdatb[num] = v; + sprdatb[num][0] = v; } static void SPRxDATA (int hpos, uae_u16 v, int num) { decide_sprites (hpos); SPRxDATA_1 (v, num); } static void SPRxDATB (int hpos, uae_u16 v, int num) { decide_sprites (hpos); SPRxDATB_1 (v, num); } @@ -2108,8 +2780,18 @@ static void SPRxPTL (int hpos, uae_u16 v static void CLXCON (uae_u16 v) { clxcon = v; - clx_sprmask = (((v >> 15) << 7) | ((v >> 14) << 5) | ((v >> 13) << 3) | ((v >> 12) << 1) | 0x55); + clxcon_bpl_enable = (v >> 6) & 63; + clxcon_bpl_match = v & 63; + clx_sprmask = ((((v >> 15) & 1) << 7) | (((v >> 14) & 1) << 5) | (((v >> 13) & 1) << 3) | (((v >> 12) & 1) << 1) | 0x55); } +static void CLXCON2 (uae_u16 v) +{ + if (!(currprefs.chipset_mask & CSMASK_AGA)) + return; + clxcon2 = v; + clxcon_bpl_enable |= v & (0x40|0x80); + clxcon_bpl_match |= (v & (0x01|0x02)) << 6; + } static uae_u16 CLXDAT (void) { uae_u16 v = clxdat; @@ -2130,9 +2812,9 @@ static uae_u16 COLOR_READ (int num) cg = (current_colors.color_regs_aga[colreg] >> 8) & 0xFF; cb = current_colors.color_regs_aga[colreg] & 0xFF; if (bplcon3 & 0x200) - cval = ((cr & 15) << 12) | ((cg & 15) << 4) | ((cb & 15) << 0); + cval = ((cr & 15) << 8) | ((cg & 15) << 4) | ((cb & 15) << 0); else - cval = ((cr >> 4) << 12) | ((cg >> 4) << 4) | ((cb >> 4) << 0); + cval = ((cr >> 4) << 8) | ((cg >> 4) << 4) | ((cb >> 4) << 0); return cval; } @@ -2254,39 +2936,59 @@ static void JOYTEST (uae_u16 v) } } +/* The copper code. The biggest nightmare in the whole emulator. + + Alright. The current theory: + 1. Copper moves happen 4 cycles after state READ2 is reached. + It can't happen immediately when we reach READ2, because the + data needs time to get back from the bus. 4 cycles appears + to be the time a chip memory access takes on the Amiga. + 2. As stated in the HRM, a WAIT really does need an extra cycle + to wake up. This is implemented by _not_ falling through from + a successful wait to READ1, but by starting the next cycle. + (Note: the extra cycle for the WAIT apparently really needs a + free cycle; i.e. contention with the bitplane fetch can slow + it down). + 3. Apparently, to compensate for the extra wake up cycle, a WAIT + will use the _incremented_ horizontal position, so the WAIT + cycle normally finishes two clocks earlier than the position + it was waiting for. The extra cycle then takes us to the + position that was waited for. + If the earlier cycle is busy with a bitplane, things change a bit. + E.g., waiting for position 0x50 in a 6 plane display: In cycle + 0x4e, we fetch BPL5, so the wait wakes up in 0x50, the extra cycle + takes us to 0x54 (since 0x52 is busy), then we have READ1/READ2, + and the next register write is at 0x5c. + 4. The last cycle in a line is not usable for the copper. + 5. A 4 cycle delay also applies to the WAIT instruction. This means + that the second of two back-to-back WAITs (or a WAIT whose + condition is immediately true) takes 8 cycles. + 6. This also applies to a SKIP instruction. The copper does not + fetch the next instruction while waiting for the second word of + a WAIT or a SKIP to arrive. + 7. A SKIP also seems to need an unexplained additional two cycles + after its second word arrives; this is _not_ a memory cycle (I + think, the documentation is pretty clear on this). + 8. Two additional cycles are inserted when writing to COPJMP1/2. */ + /* Determine which cycles are available for the copper in a display * with a agiven number of planes. */ -static int cycles_for_plane[9][8] = { - { 0, -1, 0, -1, 0, -1, 0, -1 }, - { 0, -1, 0, -1, 0, -1, 0, -1 }, - { 0, -1, 0, -1, 0, -1, 0, -1 }, - { 0, -1, 0, -1, 0, -1, 0, -1 }, - { 0, -1, 0, -1, 0, -1, 0, -1 }, - { 0, -1, 0, -1, 0, -1, 1, -1 }, /* */ - { 0, -1, 1, -1, 0, -1, 1, -1 }, - { 1, -1, 1, -1, 1, -1, 1, -1 }, - { 1, -1, 1, -1, 1, -1, 1, -1 } -}; -static unsigned int waitmasktab[256]; - -STATIC_INLINE int copper_cant_read (int hpos, int planes) +STATIC_INLINE int copper_cant_read (int hpos) { int t; - /* @@@ */ - if (hpos >= (maxhpos & ~1)) + if (hpos + 1 >= maxhpos) return 1; - if (currprefs.chipset_mask & CSMASK_AGA) - /* FIXME */ + if (fetch_state == fetch_not_started || hpos < thisline_decision.plfleft) return 0; - if (fetch_state == fetch_not_started || passed_plfstop == 3 - || hpos > estimated_last_fetch_cycle) + if ((passed_plfstop == 3 && hpos >= thisline_decision.plfright) + || hpos >= estimated_last_fetch_cycle) return 0; - t = cycles_for_plane[planes][(hpos + fetch_cycle - last_fetch_hpos) & 7]; + t = curr_diagram[(hpos + cycle_diagram_shift) & fetchstart_mask]; #if 0 if (t == -1) abort (); @@ -2305,11 +3007,173 @@ STATIC_INLINE int dangerous_reg (int reg return 1; } -#define FAST_COPPER 0 +#define FAST_COPPER 1 + +/* The future, Conan? + We try to look ahead in the copper list to avoid doing continuous calls + to updat_copper (which is what happens when SPCFLAG_COPPER is set). If + we find that the same effect can be achieved by setting a delayed event + and then doing multiple copper insns in one batch, we can get a massive + speedup. + + We don't try to be precise here. All copper reads take exactly 2 cycles, + the effect of bitplane contention is ignored. Trying to get it exactly + right would be much more complex and as such carry a huge risk of getting + it subtly wrong; and it would also be more expensive - we want this code + to be fast. */ +static void predict_copper (void) +{ + uaecptr ip = cop_state.ip; + unsigned int c_hpos = cop_state.hpos; + enum copper_states state = cop_state.state; + unsigned int w1, w2, cycle_count; + + switch (state) { + case COP_read1_wr_in2: + case COP_read2_wr_in2: + case COP_read1_wr_in4: + if (dangerous_reg (cop_state.saved_i1)) + return; + state = state == COP_read2_wr_in2 ? COP_read2 : COP_read1; + break; + + case COP_read1_in2: + c_hpos += 2; + state = COP_read1; + break; + + case COP_stop: + case COP_bltwait: + case COP_wait1: + case COP_skip_in4: + case COP_skip_in2: + return; + + case COP_wait_in4: + c_hpos += 2; + /* fallthrough */ + case COP_wait_in2: + c_hpos += 2; + /* fallthrough */ + case COP_wait: + state = COP_wait; + break; + + default: + break; + } + /* Only needed for COP_wait, but let's shut up the compiler. */ + w1 = cop_state.saved_i1; + w2 = cop_state.saved_i2; + cop_state.first_sync = c_hpos; + cop_state.regtypes_modified = REGTYPE_FORCE; + + /* Get this case out of the way, so that the loop below only has to deal + with read1 and wait. */ + if (state == COP_read2) { + w1 = cop_state.i1; + if (w1 & 1) { + w2 = chipmem_wget (ip); + if (w2 & 1) + goto done; + state = COP_wait; + c_hpos += 4; + } else if (dangerous_reg (w1)) { + c_hpos += 4; + goto done; + } else { + cop_state.regtypes_modified |= regtypes[w1 & 0x1FE]; + state = COP_read1; + c_hpos += 2; + } + ip += 2; + } + + while (c_hpos + 1 < maxhpos) { + if (state == COP_read1) { + w1 = chipmem_wget (ip); + if (w1 & 1) { + w2 = chipmem_wget (ip + 2); + if (w2 & 1) + break; + state = COP_wait; + c_hpos += 6; + } else if (dangerous_reg (w1)) { + c_hpos += 6; + goto done; + } else { + cop_state.regtypes_modified |= regtypes[w1 & 0x1FE]; + c_hpos += 4; + } + ip += 4; + } else if (state == COP_wait) { + if ((w2 & 0xFE) != 0xFE) + break; + else { + unsigned int vcmp = (w1 & (w2 | 0x8000)) >> 8; + unsigned int hcmp = (w1 & 0xFE); + + unsigned int vp = vpos & (((w2 >> 8) & 0x7F) | 0x80); + if (vp < vcmp) { + /* Whee. We can wait until the end of the line! */ + c_hpos = maxhpos; + } else if (vp > vcmp || hcmp <= c_hpos) { + state = COP_read1; + /* minimum wakeup time */ + c_hpos += 2; + } else { + state = COP_read1; + c_hpos = hcmp; + } + /* If this is the current instruction, remember that we don't + need to sync CPU and copper anytime soon. */ + if (cop_state.ip == ip) { + cop_state.first_sync = c_hpos; + } + } + } else + abort (); + } + + done: + cycle_count = c_hpos - cop_state.hpos; + if (cycle_count >= 8) { + unset_special (SPCFLAG_COPPER); + eventtab[ev_copper].active = 1; + eventtab[ev_copper].oldcycles = get_cycles (); + eventtab[ev_copper].evtime = get_cycles () + cycle_count * CYCLE_UNIT; + events_schedule (); + } +} + +static void perform_copper_write (int old_hpos) +{ + int vp = vpos & (((cop_state.saved_i2 >> 8) & 0x7F) | 0x80); + int hp; + unsigned int address = cop_state.saved_i1 & 0x1FE; + + record_copper (cop_state.saved_ip - 4, old_hpos, vpos); + + if (address < (copcon & 2 ? ((currprefs.chipset_mask & CSMASK_AGA) ? 0 : 0x40u) : 0x80u)) { + cop_state.state = COP_stop; + copper_enabled_thisline = 0; + unset_special (SPCFLAG_COPPER); + return; + } + + if (address == 0x88) { + cop_state.ip = cop1lc; + cop_state.state = COP_read1_in2; + } else if (address == 0x8A) { + cop_state.ip = cop2lc; + cop_state.state = COP_read1_in2; + } else + custom_wput_1 (old_hpos, address, cop_state.saved_i2); +} static void update_copper (int until_hpos) { - int vp = vpos & (((cop_state.i2 >> 8) & 0x7F) | 0x80); + int vp = vpos & (((cop_state.saved_i2 >> 8) & 0x7F) | 0x80); int c_hpos = cop_state.hpos; if (eventtab[ev_copper].active) @@ -2334,64 +3198,113 @@ static void update_copper (int until_hpo /* So we know about the fetch state. */ decide_line (c_hpos); + switch (cop_state.state) { + case COP_read1_in2: + cop_state.state = COP_read1; + break; + case COP_read1_wr_in2: + cop_state.state = COP_read1; + perform_copper_write (old_hpos); + /* That could have turned off the copper. */ + if (! copper_enabled_thisline) + goto out; + + break; + case COP_read1_wr_in4: + cop_state.state = COP_read1_wr_in2; + break; + case COP_read2_wr_in2: + cop_state.state = COP_read2; + perform_copper_write (old_hpos); + /* That could have turned off the copper. */ + if (! copper_enabled_thisline) + goto out; + + break; + case COP_wait_in2: + cop_state.state = COP_wait1; + break; + case COP_wait_in4: + cop_state.state = COP_wait_in2; + break; + case COP_skip_in2: + { + static int skipped_before; + unsigned int vcmp, hcmp, vp1, hp1; + cop_state.state = COP_read1_in2; + + vcmp = (cop_state.saved_i1 & (cop_state.saved_i2 | 0x8000)) >> 8; + hcmp = (cop_state.saved_i1 & cop_state.saved_i2 & 0xFE); + + if (! skipped_before) { + skipped_before = 1; + write_log ("Program uses Copper SKIP instruction.\n"); + } + + vp1 = vpos & (((cop_state.saved_i2 >> 8) & 0x7F) | 0x80); + hp1 = old_hpos & (cop_state.saved_i2 & 0xFE); + + if ((vp1 > vcmp || (vp1 == vcmp && hp1 >= hcmp)) + && ((cop_state.saved_i2 & 0x8000) != 0 || ! (DMACONR() & 0x4000))) + cop_state.ignore_next = 1; + break; + } + case COP_skip_in4: + cop_state.state = COP_skip_in2; + break; + default: + break; + } + c_hpos += 2; - if (copper_cant_read (old_hpos, corrected_nr_planes_from_bplcon0)) + if (copper_cant_read (old_hpos)) continue; switch (cop_state.state) { + case COP_read1_wr_in4: + abort (); + + case COP_read1_wr_in2: case COP_read1: - cop_state.i1 = chipmem_bank.wget (cop_state.ip); + cop_state.i1 = chipmem_wget (cop_state.ip); cop_state.ip += 2; - cop_state.state = COP_read2; + cop_state.state = cop_state.state == COP_read1 ? COP_read2 : COP_read2_wr_in2; break; + case COP_read2_wr_in2: + abort (); + case COP_read2: - cop_state.i2 = chipmem_bank.wget (cop_state.ip); + cop_state.i2 = chipmem_wget (cop_state.ip); cop_state.ip += 2; - cop_state.state = COP_read1; if (cop_state.ignore_next) { cop_state.ignore_next = 0; - break; - } - /* Perform moves immediately. */ - if ((cop_state.i1 & 1) == 0) { - unsigned int address = cop_state.i1 & 0x1FE; - if (address < (copcon & 2 ? ((currprefs.chipset_mask & CSMASK_AGA) ? 0 : 0x40u) : 0x80u)) { - cop_state.state = COP_stop; - copper_enabled_thisline = 0; - unset_special (SPCFLAG_COPPER); - goto out; - } - if (address == 0x88) { - cop_state.ip = cop1lc; - } else if (address == 0x8A) { - cop_state.ip = cop2lc; - } else - custom_wput_1 (old_hpos, address, cop_state.i2); - /* That could have turned off the copper... */ - if (! copper_enabled_thisline) - goto out; + cop_state.state = COP_read1; break; } - cop_state.vcmp = (cop_state.i1 & (cop_state.i2 | 0x8000)) >> 8; - cop_state.hcmp = (cop_state.i1 & cop_state.i2 & 0xFE); - if ((cop_state.i2 & 1) == 1) { - /* Skip instruction. */ - vp = vpos & (((cop_state.i2 >> 8) & 0x7F) | 0x80); - hp = old_hpos & (cop_state.i2 & 0xFE); - - if ((vp > cop_state.vcmp || (vp == cop_state.vcmp && hp >= cop_state.hcmp)) - && ((cop_state.i2 & 0x8000) != 0 || ! (DMACONR() & 0x4000))) - cop_state.ignore_next = 1; - break; - } + cop_state.saved_i1 = cop_state.i1; + cop_state.saved_i2 = cop_state.i2; + cop_state.saved_ip = cop_state.ip; + + if (cop_state.i1 & 1) { + if (cop_state.i2 & 1) + cop_state.state = COP_skip_in4; + else + cop_state.state = COP_wait_in4; + } else + cop_state.state = COP_read1_wr_in4; + break; + + case COP_wait1: cop_state.state = COP_wait; - vp = vpos & (((cop_state.i2 >> 8) & 0x7F) | 0x80); - hp = old_hpos & (cop_state.i2 & 0xFE); + cop_state.vcmp = (cop_state.saved_i1 & (cop_state.saved_i2 | 0x8000)) >> 8; + cop_state.hcmp = (cop_state.saved_i1 & cop_state.saved_i2 & 0xFE); - if (cop_state.i1 == 0xFFFF && cop_state.i2 == 0xFFFE) { + vp = vpos & (((cop_state.saved_i2 >> 8) & 0x7F) | 0x80); + + if (cop_state.saved_i1 == 0xFFFF && cop_state.saved_i2 == 0xFFFE) { cop_state.state = COP_stop; copper_enabled_thisline = 0; unset_special (SPCFLAG_COPPER); @@ -2402,58 +3315,32 @@ static void update_copper (int until_hpo unset_special (SPCFLAG_COPPER); goto out; } - if (vp > cop_state.vcmp) - break; - - /* Only enable shortcuts if there's no masking going on. */ - if (FAST_COPPER && (cop_state.i2 & 0xFE) == 0xFE) { - /* Compute cycles remaining until c_hpos is past until_hpos - (i.e. until we'd normally break out of the loop). */ - int time_remaining = until_hpos - c_hpos; - - if (time_remaining < 0) - abort (); - - /* Compute minimum number of cycles to wait once the copper is at c_hpos. */ - cop_min_waittime = cop_state.hcmp - hp - 2; - - if (cop_min_waittime <= 0) - break; - - /* Does this still leave us before until_hpos? */ - if (cop_min_waittime <= time_remaining) { - c_hpos += cop_min_waittime; - break; - } - - /* This wait will use up all the time up to until_hpos, and then some. */ - c_hpos += time_remaining; - cop_min_waittime -= time_remaining; - - if (cop_min_waittime >= 8) { - unset_special (SPCFLAG_COPPER); - eventtab[ev_copper].active = 1; - eventtab[ev_copper].oldcycles = cycles; - eventtab[ev_copper].evtime = cycles + cop_min_waittime; - /* until_hpos is larger than hpos; add a correction for this adjustment. */ - cop_min_waittime -= 2; - events_schedule (); - goto out; - } - } - break; + /* fall through */ + do_wait: case COP_wait: if (vp < cop_state.vcmp) abort (); - hp = old_hpos & (cop_state.i2 & 0xFE); - if (vp == cop_state.vcmp && hp < cop_state.hcmp) + hp = c_hpos & (cop_state.saved_i2 & 0xFE); + if (vp == cop_state.vcmp && hp < cop_state.hcmp) { + /* Position not reached yet. */ + if (currprefs.fast_copper && (cop_state.saved_i2 & 0xFE) == 0xFE) { + int wait_finish = cop_state.hcmp - 2; + /* This will leave c_hpos untouched if it's equal to wait_finish. */ + if (wait_finish < c_hpos) + abort (); + else if (wait_finish <= until_hpos) { + c_hpos = wait_finish; + } else + c_hpos = until_hpos; + } break; + } /* Now we know that the comparisons were successful. We might still have to wait for the blitter though. */ - if ((cop_state.i2 & 0x8000) == 0 && (DMACONR() & 0x4000)) { + if ((cop_state.saved_i2 & 0x8000) == 0 && (DMACONR() & 0x4000)) { /* We need to wait for the blitter. */ cop_state.state = COP_bltwait; copper_enabled_thisline = 0; @@ -2461,53 +3348,24 @@ static void update_copper (int until_hpo goto out; } - cop_state.state = COP_wait1; - break; + record_copper (cop_state.ip - 4, old_hpos, vpos); - case COP_wait1: cop_state.state = COP_read1; break; default: - abort (); + break; } } out: cop_state.hpos = c_hpos; -#if FAST_COPPER - /* The future, Conan? */ - if (cop_state.state == COP_read1 || cop_state.state == COP_read2) { - int ip = cop_state.ip; - int word = cop_state.i1; - int cycle_count; - - if (eventtab[ev_copper].active /* || ! (regs.spcflags & SPCFLAG_COPPER) */) - abort (); - if (cop_state.state == COP_read2) { - ip += 2; - c_hpos += 2; - goto inner; - } - while (c_hpos < (maxhpos & ~1)) { - word = chipmem_bank.wget (ip); - ip += 4; - c_hpos += 4; - inner: - if ((word & 1) || dangerous_reg (word)) - break; - } - cycle_count = c_hpos - cop_state.hpos; - if (cycle_count >= 8) { - unset_special (SPCFLAG_COPPER); - eventtab[ev_copper].active = 1; - eventtab[ev_copper].oldcycles = cycles; - eventtab[ev_copper].evtime = cycles + cycle_count; - events_schedule (); - } - } -#endif + /* The test against maxhpos also prevents us from calling predict_copper + when we are being called from hsync_handler, which would not only be + stupid, but actively harmful. */ + if (currprefs.fast_copper && (regs.spcflags & SPCFLAG_COPPER) && c_hpos + 8 < maxhpos) + predict_copper (); } static void compute_spcflag_copper (void) @@ -2518,30 +3376,18 @@ static void compute_spcflag_copper (void return; if (cop_state.state == COP_wait) { - int vp = vpos & (((cop_state.i2 >> 8) & 0x7F) | 0x80); + int vp = vpos & (((cop_state.saved_i2 >> 8) & 0x7F) | 0x80); if (vp < cop_state.vcmp) return; - copper_enabled_thisline = 1; - - if (FAST_COPPER && vp == cop_state.vcmp) { - int hp = cop_state.hpos & (cop_state.i2 & 0xFE); - cop_min_waittime = cop_state.hcmp - hp - 2; - - /* If possible, compute a minimum waiting time, and set the event - timer if it's sufficiently large to be worthwhile. */ - if ((cop_state.i2 & 0xFE) == 0xFE && cop_min_waittime >= 8) { - eventtab[ev_copper].active = 1; - eventtab[ev_copper].oldcycles = cycles; - eventtab[ev_copper].evtime = cycles + cop_min_waittime; - events_schedule (); - return; - } - } } - copper_enabled_thisline = 1; - set_special (SPCFLAG_COPPER); + + if (currprefs.fast_copper) + predict_copper (); + + if (! eventtab[ev_copper].active) + set_special (SPCFLAG_COPPER); } static void copper_handler (void) @@ -2552,12 +3398,6 @@ static void copper_handler (void) if (! copper_enabled_thisline) abort (); - if (cop_state.state == COP_wait) { - cop_state.hpos += cop_min_waittime; - if (cop_state.hpos > current_hpos ()) - abort (); - } - eventtab[ev_copper].active = 0; } @@ -2577,18 +3417,30 @@ void do_copper (void) update_copper (hpos); } -static void sync_copper_with_cpu (int hpos) +/* ADDR is the address that is going to be read/written; this access is + the reason why we want to update the copper. This function is also + used from hsync_handler to finish up the line; for this case, we check + hpos against maxhpos. */ +STATIC_INLINE void sync_copper_with_cpu (int hpos, int do_schedule, unsigned int addr) { - /* Need to let the copper advance to the current position, but only if it - isn't in a waiting state. */ + /* Need to let the copper advance to the current position. */ if (eventtab[ev_copper].active) { - if (cop_state.state != COP_wait) { - eventtab[ev_copper].active = 0; - events_schedule (); - set_special (SPCFLAG_COPPER); + if (hpos != maxhpos) { + /* There might be reasons why we don't actually need to bother + updating the copper. */ + if (hpos < cop_state.first_sync) + return; + + if ((cop_state.regtypes_modified & regtypes[addr & 0x1FE]) == 0) + return; } + + eventtab[ev_copper].active = 0; + if (do_schedule) + events_schedule (); + set_special (SPCFLAG_COPPER); } - if (copper_enabled_thisline && ! eventtab[ev_copper].active) + if (copper_enabled_thisline) update_copper (hpos); } @@ -2596,6 +3448,7 @@ static void do_sprites (int currvp, int { int i; int maxspr, minspr; + int sw = sprite_width >> 3; /* I don't know whether this is right. Some programs write the sprite pointers * directly at the start of the copper list. With the test against currvp, the @@ -2622,11 +3475,11 @@ static void do_sprites (int currvp, int maxspr = hpos / 4 - 0x14 / 4; minspr = last_sprite_hpos / 4 - 0x14 / 4; - if (minspr > 8 || maxspr < 0) + if (minspr > MAX_SPRITES || maxspr < 0) return; - if (maxspr > 8) - maxspr = 8; + if (maxspr > MAX_SPRITES) + maxspr = MAX_SPRITES; if (minspr < 0) minspr = 0; @@ -2649,9 +3502,8 @@ static void do_sprites (int currvp, int } if (fetch && dmaen (DMA_SPRITE)) { - uae_u16 data1 = chipmem_bank.wget (spr[i].pt); - uae_u16 data2 = chipmem_bank.wget (spr[i].pt + 2); - spr[i].pt += 4; + uae_u16 data1 = chipmem_wget (spr[i].pt); + uae_u16 data2 = chipmem_wget (spr[i].pt + sw); if (fetch == 1) { /* Hack for X mouse auto-calibration */ @@ -2662,10 +3514,24 @@ static void do_sprites (int currvp, int } SPRxDATB_1 (data2, i); SPRxDATA_1 (data1, i); + switch (sw) + { + case 64 >> 3: + sprdata[i][3] = chipmem_wget (spr[i].pt + 6); + sprdatb[i][3] = chipmem_wget (spr[i].pt + 6 + sw); + sprdata[i][2] = chipmem_wget (spr[i].pt + 4); + sprdatb[i][2] = chipmem_wget (spr[i].pt + 4 + sw); + /* fall through */ + case 32 >> 3: + sprdata[i][1] = chipmem_wget (spr[i].pt + 2); + sprdatb[i][1] = chipmem_wget (spr[i].pt + 2 + sw); + break; + } } else { SPRxPOS_1 (data1, i); SPRxCTL_1 (data2, i); } + spr[i].pt += sw * 2; } } last_sprite_hpos = hpos; @@ -2675,7 +3541,7 @@ static void init_sprites (void) { int i; - for (i = 0; i < 8; i++) { + for (i = 0; i < MAX_SPRITES; i++) { /* ???? */ spr[i].state = SPR_stop; spr[i].on = 0; @@ -2761,8 +3627,17 @@ void init_hardware_for_drawing_frame (vo next_sprite_forced = 1; } +static void do_savestate(void); + static void vsync_handler (void) { +#if 0 + static int old_clxdat; + if (clxdat != old_clxdat) { + printf ("CLXDAT %04x\n", clxdat); + old_clxdat = clxdat; + } +#endif n_frames++; if (currprefs.m68k_speed == -1) { @@ -2785,7 +3660,10 @@ static void vsync_handler (void) if (bplcon0 & 4) lof ^= 0x8000; + if (picasso_on) + picasso_handle_vsync (); vsync_handle_redraw (lof, lof_changed); + if (quit_program > 0) return; @@ -2799,6 +3677,10 @@ static void vsync_handler (void) cnt--; } + /* Start a new set of copper records. */ + curr_cop_set ^= 1; + nr_cop_records[curr_cop_set] = 0; + /* For now, let's only allow this to change at vsync time. It gets too * hairy otherwise. */ if (beamcon0 != new_beamcon0) @@ -2852,35 +3734,27 @@ static void vsync_handler (void) ievent_alive--; if (timehack_alive > 0) timehack_alive--; - CIA_vsync_handler(); + CIA_vsync_handler (); } static void hsync_handler (void) { - int copper_was_active = eventtab[ev_copper].active; - if (copper_was_active) { - /* Could happen if horizontal wait position is too large. */ - eventtab[ev_copper].active = 0; - /* If this was a sequence of moves, we have to call update_copper if - we don't want to lose them. */ - if (cop_state.state != COP_wait) { - copper_was_active = 0; - set_special (SPCFLAG_COPPER); - } - } - if (copper_enabled_thisline && ! copper_was_active) - update_copper (maxhpos); + /* Using 0x8A makes sure that we don't accidentally trip over the + modified_regtypes check. */ + sync_copper_with_cpu (maxhpos, 0, 0x8A); finish_decisions (); - if (currprefs.collision_level > 1) - do_sprite_collisions (); - + if (thisline_decision.plfleft != -1) { + if (currprefs.collision_level > 1) + do_sprite_collisions (); + if (currprefs.collision_level > 2) + do_playfield_collisions (); + } hsync_record_line_state (next_lineno, nextline_how, thisline_changed); - eventtab[ev_hsync].evtime += cycles - eventtab[ev_hsync].oldcycles; - eventtab[ev_hsync].oldcycles = cycles; + eventtab[ev_hsync].evtime += get_cycles () - eventtab[ev_hsync].oldcycles; + eventtab[ev_hsync].oldcycles = get_cycles (); CIA_hsync_handler (); - DISK_update (); if (currprefs.produce_sound > 0) { int nr; @@ -2893,7 +3767,7 @@ static void hsync_handler (void) if (cdp->data_written == 2) { cdp->data_written = 0; - cdp->nextdat = chipmem_bank.wget(cdp->pt); + cdp->nextdat = chipmem_wget (cdp->pt); cdp->pt += 2; if (cdp->state == 2 || cdp->state == 3) { if (cdp->wlen == 1) { @@ -2901,7 +3775,7 @@ static void hsync_handler (void) cdp->wlen = cdp->len; cdp->intreq2 = 1; } else - cdp->wlen--; + cdp->wlen = (cdp->wlen - 1) & 0xFFFF; } } } @@ -2909,11 +3783,18 @@ static void hsync_handler (void) hardware_line_completed (next_lineno); - if (++vpos == (maxvpos + (lof != 0))) { + /* In theory only an equality test is needed here - but if a program + goes haywire with the VPOSW register, it can cause us to miss this, + with vpos going into the thousands (and all the nasty consequences + this has). */ + + if (++vpos >= (maxvpos + (lof != 0))) { vpos = 0; - vsync_handler(); + vsync_handler (); } + DISK_update (); + is_lastline = vpos + 1 == maxvpos + (lof != 0) && currprefs.m68k_speed == -1 && ! rpt_did_reset; if ((bplcon0 & 4) && currprefs.gfx_linedbl) @@ -2946,18 +3827,70 @@ static void hsync_handler (void) compute_spcflag_copper (); } -static void init_eventtab (void) +static void init_regtypes (void) +{ + int i; + for (i = 0; i < 512; i += 2) { + regtypes[i] = REGTYPE_ALL; + if ((i >= 0x20 && i < 0x28) || i == 0x08 || i == 0x7E) + regtypes[i] = REGTYPE_DISK; + else if (i >= 0x68 && i < 0x70) + regtypes[i] = REGTYPE_NONE; + else if (i >= 0x40 && i < 0x78) + regtypes[i] = REGTYPE_BLITTER; + else if (i >= 0xA0 && i < 0xE0 && (i & 0xF) < 0xE) + regtypes[i] = REGTYPE_AUDIO; + else if (i >= 0xA0 && i < 0xE0) + regtypes[i] = REGTYPE_NONE; + else if (i >= 0xE0 && i < 0x100) + regtypes[i] = REGTYPE_PLANE; + else if (i >= 0x120 && i < 0x180) + regtypes[i] = REGTYPE_SPRITE; + else if (i >= 0x180 && i < 0x1C0) + regtypes[i] = REGTYPE_COLOR; + else switch (i) { + case 0x02: + /* DMACONR - setting this to REGTYPE_BLITTER will cause it to + conflict with DMACON (since that is REGTYPE_ALL), and the + blitter registers (for the BBUSY bit), but nothing else, + which is (I think) what we want. */ + regtypes[i] = REGTYPE_BLITTER; + break; + case 0x04: case 0x06: case 0x2A: case 0x2C: + regtypes[i] = REGTYPE_POS; + break; + case 0x0A: case 0x0C: + case 0x12: case 0x14: case 0x16: + case 0x36: + regtypes[i] = REGTYPE_JOYPORT; + break; + case 0x104: + case 0x102: + regtypes[i] = REGTYPE_PLANE; + break; + case 0x88: case 0x8A: + case 0x8E: case 0x90: case 0x92: case 0x94: + case 0x96: + case 0x100: + regtypes[i] |= REGTYPE_FORCE; + break; + } + } +} + +void init_eventtab (void) { int i; - for(i = 0; i < ev_max; i++) { + currcycle = 0; + for (i = 0; i < ev_max; i++) { eventtab[i].active = 0; eventtab[i].oldcycles = 0; } eventtab[ev_cia].handler = CIA_handler; eventtab[ev_hsync].handler = hsync_handler; - eventtab[ev_hsync].evtime = maxhpos + cycles; + eventtab[ev_hsync].evtime = maxhpos * CYCLE_UNIT + get_cycles (); eventtab[ev_hsync].active = 1; eventtab[ev_copper].handler = copper_handler; @@ -2966,7 +3899,8 @@ static void init_eventtab (void) eventtab[ev_blitter].active = 0; eventtab[ev_disk].handler = DISK_handler; eventtab[ev_disk].active = 0; - + eventtab[ev_audio].handler = audio_evhandler; + eventtab[ev_audio].active = 0; events_schedule (); } @@ -2978,16 +3912,38 @@ void customreset (void) struct timeval tv; #endif - if ((currprefs.chipset_mask & CSMASK_AGA) == 0) { - for (i = 0; i < 32; i++) { - current_colors.color_regs_ecs[i] = 0; - current_colors.acolors[i] = xcolors[0]; - } - } else { - for (i = 0; i < 256; i++) { - current_colors.color_regs_aga[i] = 0; - current_colors.acolors[i] = CONVERT_RGB (zero); + if (! savestate_state) { + currprefs.chipset_mask = changed_prefs.chipset_mask; + if ((currprefs.chipset_mask & CSMASK_AGA) == 0) { + for (i = 0; i < 32; i++) { + current_colors.color_regs_ecs[i] = 0; + current_colors.acolors[i] = xcolors[0]; + } + } else { + for (i = 0; i < 256; i++) { + current_colors.color_regs_aga[i] = 0; + current_colors.acolors[i] = CONVERT_RGB (zero); + } } + + clx_sprmask = 0xFF; + clxdat = 0; + + /* Clear the armed flags of all sprites. */ + memset (spr, 0, sizeof spr); + nr_armed = 0; + + dmacon = intena = 0; + + copcon = 0; + DSKLEN (0, 0); + + bplcon0 = 0; + bplcon4 = 0x11; /* Get AGA chipset into ECS compatibility mode */ + bplcon3 = 0xC00; + + FMODE (0); + CLXCON (0); } n_frames = 0; @@ -2996,7 +3952,6 @@ void customreset (void) DISK_reset (); CIA_reset (); - cycles = 0; unset_special (~(SPCFLAG_BRK | SPCFLAG_MODE_CHANGE)); vpos = 0; @@ -3014,9 +3969,6 @@ void customreset (void) ievent_alive = 0; timehack_alive = 0; - clx_sprmask = 0xFF; - clxdat = 0; - curr_sprite_entries = 0; prev_sprite_entries = 0; sprite_entries[0][0].first_pixel = 0; @@ -3025,30 +3977,18 @@ void customreset (void) sprite_entries[1][1].first_pixel = MAX_SPR_PIXELS; memset (spixels, 0, sizeof spixels); memset (&spixstate, 0, sizeof spixstate); - - /* Clear the armed flags of all sprites. */ - memset (spr, 0, sizeof spr); - nr_armed = 0; - dmacon = intena = 0; bltstate = BLT_done; cop_state.state = COP_stop; diwstate = DIW_waiting_start; hdiwstate = DIW_waiting_start; - copcon = 0; - DSKLEN (0, 0); - cycles = 0; + currcycle = 0; - bplcon4 = 0x11; /* Get AGA chipset into ECS compatibility mode */ - bplcon3 = 0xC00; - - new_beamcon0 = ntscmode ? 0x00 : 0x20; + new_beamcon0 = currprefs.ntscmode ? 0x00 : 0x20; init_hz (); audio_reset (); - init_eventtab (); - init_sprites (); init_hardware_frame (); @@ -3061,6 +4001,55 @@ void customreset (void) seconds_base = tv.tv_sec; bogusframe = 1; #endif + + init_regtypes (); + + sprite_buffer_res = currprefs.chipset_mask & CSMASK_AGA ? RES_HIRES : RES_LORES; + if (savestate_state == STATE_RESTORE) { + uae_u16 v; + uae_u32 vv; + + update_adkmasks (); + INTENA (0); + INTREQ (0); +#if 0 + DMACON (0, 0); +#endif + COPJMP1 (0); + if (diwhigh) + diwhigh_written = 1; + v = bplcon0; + BPLCON0 (0, 0); + BPLCON0 (0, v); + FMODE (fmode); + if (!(currprefs.chipset_mask & CSMASK_AGA)) { + for(i = 0 ; i < 32 ; i++) { + vv = current_colors.color_regs_ecs[i]; + current_colors.color_regs_ecs[i] = -1; + record_color_change (0, i, vv); + remembered_color_entry = -1; + current_colors.color_regs_ecs[i] = vv; + current_colors.acolors[i] = xcolors[vv]; + } + } else { + for(i = 0 ; i < 256 ; i++) { + vv = current_colors.color_regs_aga[i]; + current_colors.color_regs_aga[i] = -1; + record_color_change (0, i, vv); + remembered_color_entry = -1; + current_colors.color_regs_aga[i] = vv; + current_colors.acolors[i] = CONVERT_RGB(vv); + } + } + CLXCON (clxcon); + CLXCON2 (clxcon2); + calcdiw (); + write_log ("State restored\n"); + dumpcustom (); + for (i = 0; i < 8; i++) + nr_armed += spr[i].armed != 0; + } + expand_sprres (); } void dumpcustom (void) @@ -3076,7 +4065,6 @@ void dumpcustom (void) if (total_skipped) write_log ("Skipped frames: %d\n", total_skipped); } - dump_audio_bench (); /*for (i=0; i<256; i++) if (blitcount[i]) fprintf (stderr, "minterm %x = %d\n",i,blitcount[i]); blitter debug */ } @@ -3110,10 +4098,6 @@ static void gen_custom_tables (void) sprtabb[i] = sprtaba[i] * 2; sprite_ab_merge[i] = (((i & 15) ? 1 : 0) | ((i & 240) ? 2 : 0)); - - for (j = 0; j < 511; j = (j << 1) | 1) - if ((i & ~j) == 0) - waitmasktab[i] = ~j; } for (i = 0; i < 16; i++) { clxmask[i] = (((i & 1) ? 0xF : 0x3) @@ -3162,7 +4146,9 @@ void custom_init (void) mousestate = unknown_mouse; if (needmousehack ()) - mousehack_setfollow(); + mousehack_setfollow (); + + create_cycle_diagram_table (); } /* Custom chip memory bank */ @@ -3177,7 +4163,7 @@ static void custom_bput (uaecptr, uae_u3 addrbank custom_bank = { custom_lget, custom_wget, custom_bget, custom_lput, custom_wput, custom_bput, - default_xlate, default_check + default_xlate, default_check, NULL }; STATIC_INLINE uae_u32 REGPARAM2 custom_wget_1 (uaecptr addr) @@ -3220,7 +4206,7 @@ STATIC_INLINE uae_u32 REGPARAM2 custom_w uae_u32 REGPARAM2 custom_wget (uaecptr addr) { - sync_copper_with_cpu (current_hpos ()); + sync_copper_with_cpu (current_hpos (), 1, addr); return custom_wget_1 (addr); } @@ -3246,10 +4232,10 @@ void REGPARAM2 custom_wput_1 (int hpos, case 0x026: DSKDAT (value); break; case 0x02A: VPOSW (value); break; - case 0x2E: COPCON (value); break; + case 0x02E: COPCON (value); break; case 0x030: SERDAT (value); break; case 0x032: SERPER (value); break; - case 0x34: POTGO (value); break; + case 0x034: POTGO (value); break; case 0x040: BLTCON0 (value); break; case 0x042: BLTCON1 (value); break; @@ -3349,6 +4335,7 @@ void REGPARAM2 custom_wput_1 (int hpos, case 0x108: BPL1MOD (hpos, value); break; case 0x10A: BPL2MOD (hpos, value); break; + case 0x10E: CLXCON2 (value); break; case 0x110: BPL1DAT (hpos, value); break; case 0x112: BPL2DAT (value); break; @@ -3407,7 +4394,7 @@ void REGPARAM2 custom_wput (uaecptr addr int hpos = current_hpos (); special_mem |= S_WRITE; - sync_copper_with_cpu (hpos); + sync_copper_with_cpu (hpos, 1, addr); custom_wput_1 (hpos, addr, value); } @@ -3428,3 +4415,352 @@ void REGPARAM2 custom_lput(uaecptr addr, custom_wput (addr & 0xfffe, value >> 16); custom_wput ((addr + 2) & 0xfffe, (uae_u16)value); } + +void custom_prepare_savestate (void) +{ + /* force blitter to finish, no support for saving full blitter state yet */ + if (eventtab[ev_blitter].active) { + unsigned int olddmacon = dmacon; + dmacon |= DMA_BLITTER; /* ugh.. */ + blitter_handler (); + dmacon = olddmacon; + } +} + +#define RB restore_u8 () +#define RW restore_u16 () +#define RL restore_u32 () + +uae_u8 *restore_custom (uae_u8 *src) +{ + uae_u16 dsklen, dskbytr, dskdatr; + int dskpt; + int i; + + audio_reset (); + + currprefs.chipset_mask = RL; + RW; /* 000 ? */ + RW; /* 002 DMACONR */ + RW; /* 004 VPOSR */ + RW; /* 006 VHPOSR */ + dskdatr = RW; /* 008 DSKDATR */ + RW; /* 00A JOY0DAT */ + RW; /* 00C JOY1DAT */ + clxdat = RW; /* 00E CLXDAT */ + RW; /* 010 ADKCONR */ + RW; /* 012 POT0DAT* */ + RW; /* 014 POT1DAT* */ + RW; /* 016 POTINP* */ + RW; /* 018 SERDATR* */ + dskbytr = RW; /* 01A DSKBYTR */ + RW; /* 01C INTENAR */ + RW; /* 01E INTREQR */ + dskpt = RL; /* 020-022 DSKPT */ + dsklen = RW; /* 024 DSKLEN */ + RW; /* 026 DSKDAT */ + RW; /* 028 REFPTR */ + lof = RW; /* 02A VPOSW */ + RW; /* 02C VHPOSW */ + COPCON(RW); /* 02E COPCON */ + RW; /* 030 SERDAT* */ + RW; /* 032 SERPER* */ + POTGO(RW); /* 034 POTGO */ + RW; /* 036 JOYTEST* */ + RW; /* 038 STREQU */ + RW; /* 03A STRVHBL */ + RW; /* 03C STRHOR */ + RW; /* 03E STRLONG */ + BLTCON0(RW); /* 040 BLTCON0 */ + BLTCON1(RW); /* 042 BLTCON1 */ + BLTAFWM(RW); /* 044 BLTAFWM */ + BLTALWM(RW); /* 046 BLTALWM */ + BLTCPTH(RL); /* 048-04B BLTCPT */ + BLTBPTH(RL); /* 04C-04F BLTBPT */ + BLTAPTH(RL); /* 050-053 BLTAPT */ + BLTDPTH(RL); /* 054-057 BLTDPT */ + RW; /* 058 BLTSIZE */ + RW; /* 05A BLTCON0L */ + oldvblts = RW; /* 05C BLTSIZV */ + RW; /* 05E BLTSIZH */ + BLTCMOD(RW); /* 060 BLTCMOD */ + BLTBMOD(RW); /* 062 BLTBMOD */ + BLTAMOD(RW); /* 064 BLTAMOD */ + BLTDMOD(RW); /* 066 BLTDMOD */ + RW; /* 068 ? */ + RW; /* 06A ? */ + RW; /* 06C ? */ + RW; /* 06E ? */ + BLTCDAT(RW); /* 070 BLTCDAT */ + BLTBDAT(RW); /* 072 BLTBDAT */ + BLTADAT(RW); /* 074 BLTADAT */ + RW; /* 076 ? */ + RW; /* 078 ? */ + RW; /* 07A ? */ + RW; /* 07C LISAID */ + DSKSYNC(RW); /* 07E DSKSYNC */ + cop1lc = RL; /* 080/082 COP1LC */ + cop2lc = RL; /* 084/086 COP2LC */ + RW; /* 088 ? */ + RW; /* 08A ? */ + RW; /* 08C ? */ + diwstrt = RW; /* 08E DIWSTRT */ + diwstop = RW; /* 090 DIWSTOP */ + ddfstrt = RW; /* 092 DDFSTRT */ + ddfstop = RW; /* 094 DDFSTOP */ + dmacon = RW & ~(0x2000|0x4000); /* 096 DMACON */ + CLXCON(RW); /* 098 CLXCON */ + intena = RW; /* 09A INTENA */ + intreq = RW; /* 09C INTREQ */ + adkcon = RW; /* 09E ADKCON */ + for (i = 0; i < 8; i++) + bplpt[i] = RL; + bplcon0 = RW; /* 100 BPLCON0 */ + bplcon1 = RW; /* 102 BPLCON1 */ + bplcon2 = RW; /* 104 BPLCON2 */ + bplcon3 = RW; /* 106 BPLCON3 */ + bpl1mod = RW; /* 108 BPL1MOD */ + bpl2mod = RW; /* 10A BPL2MOD */ + bplcon4 = RW; /* 10C BPLCON4 */ + clxcon2 = RW; /* 10E CLXCON2* */ + for(i = 0; i < 8; i++) + RW; /* BPLXDAT */ + for(i = 0; i < 32; i++) + current_colors.color_regs_ecs[i] = RW; /* 180 COLORxx */ + RW; /* 1C0 ? */ + RW; /* 1C2 ? */ + RW; /* 1C4 ? */ + RW; /* 1C6 ? */ + RW; /* 1C8 ? */ + RW; /* 1CA ? */ + RW; /* 1CC ? */ + RW; /* 1CE ? */ + RW; /* 1D0 ? */ + RW; /* 1D2 ? */ + RW; /* 1D4 ? */ + RW; /* 1D6 ? */ + RW; /* 1D8 ? */ + RW; /* 1DA ? */ + new_beamcon0 = RW; /* 1DC BEAMCON0 */ + RW; /* 1DE ? */ + RW; /* 1E0 ? */ + RW; /* 1E2 ? */ + RW; /* 1E4 ? */ + RW; /* 1E6 ? */ + RW; /* 1E8 ? */ + RW; /* 1EA ? */ + RW; /* 1EC ? */ + RW; /* 1EE ? */ + RW; /* 1F0 ? */ + RW; /* 1F2 ? */ + RW; /* 1F4 ? */ + RW; /* 1F6 ? */ + RW; /* 1F8 ? */ + RW; /* 1FA ? */ + fmode = RW; /* 1FC FMODE */ + RW; /* 1FE ? */ + + DISK_restore_custom (dskpt, dsklen, dskdatr, dskbytr); + + return src; +} + + +#define SB save_u8 +#define SW save_u16 +#define SL save_u32 + +extern uae_u16 serper; + +uae_u8 *save_custom (int *len) +{ + uae_u8 *dstbak, *dst; + int i; + uae_u32 dskpt; + uae_u16 dsklen, dsksync, dskdatr, dskbytr; + + DISK_save_custom (&dskpt, &dsklen, &dsksync, &dskdatr, &dskbytr); + dstbak = dst = malloc (8+256*2); + SL (currprefs.chipset_mask); + SW (0); /* 000 ? */ + SW (dmacon); /* 002 DMACONR */ + SW (VPOSR()); /* 004 VPOSR */ + SW (VHPOSR()); /* 006 VHPOSR */ + SW (dskdatr); /* 008 DSKDATR */ + SW (JOY0DAT()); /* 00A JOY0DAT */ + SW (JOY1DAT()); /* 00C JOY1DAT */ + SW (clxdat); /* 00E CLXDAT */ + SW (ADKCONR()); /* 010 ADKCONR */ + SW (POT0DAT()); /* 012 POT0DAT */ + SW (POT0DAT()); /* 014 POT1DAT */ + SW (0) ; /* 016 POTINP * */ + SW (0); /* 018 SERDATR * */ + SW (dskbytr); /* 01A DSKBYTR */ + SW (INTENAR()); /* 01C INTENAR */ + SW (INTREQR()); /* 01E INTREQR */ + SL (dskpt); /* 020-023 DSKPT */ + SW (dsklen); /* 024 DSKLEN */ + SW (0); /* 026 DSKDAT */ + SW (0); /* 028 REFPTR */ + SW (lof); /* 02A VPOSW */ + SW (0); /* 02C VHPOSW */ + SW (copcon); /* 02E COPCON */ + SW (serper); /* 030 SERDAT * */ + SW (serdat); /* 032 SERPER * */ + SW (potgo_value); /* 034 POTGO */ + SW (0); /* 036 JOYTEST * */ + SW (0); /* 038 STREQU */ + SW (0); /* 03A STRVBL */ + SW (0); /* 03C STRHOR */ + SW (0); /* 03E STRLONG */ + SW (bltcon0); /* 040 BLTCON0 */ + SW (bltcon1); /* 042 BLTCON1 */ + SW (blt_info.bltafwm); /* 044 BLTAFWM */ + SW (blt_info.bltalwm); /* 046 BLTALWM */ + SL (bltcpt); /* 048-04B BLTCPT */ + SL (bltbpt); /* 04C-04F BLTCPT */ + SL (bltapt); /* 050-043 BLTCPT */ + SL (bltdpt); /* 054-057 BLTCPT */ + SW (0); /* 058 BLTSIZE */ + SW (0); /* 05A BLTCON0L (use BLTCON0 instead) */ + SW (oldvblts); /* 05C BLTSIZV */ + SW (blt_info.hblitsize); /* 05E BLTSIZH */ + SW (blt_info.bltcmod); /* 060 BLTCMOD */ + SW (blt_info.bltbmod); /* 062 BLTBMOD */ + SW (blt_info.bltamod); /* 064 BLTAMOD */ + SW (blt_info.bltdmod); /* 066 BLTDMOD */ + SW (0); /* 068 ? */ + SW (0); /* 06A ? */ + SW (0); /* 06C ? */ + SW (0); /* 06E ? */ + SW (blt_info.bltcdat); /* 070 BLTCDAT */ + SW (blt_info.bltbdat); /* 072 BLTBDAT */ + SW (blt_info.bltadat); /* 074 BLTADAT */ + SW (0); /* 076 ? */ + SW (0); /* 078 ? */ + SW (0); /* 07A ? */ + SW (DENISEID()); /* 07C DENISEID/LISAID */ + SW (dsksync); /* 07E DSKSYNC */ + SL (cop1lc); /* 080-083 COP1LC */ + SL (cop2lc); /* 084-087 COP2LC */ + SW (0); /* 088 ? */ + SW (0); /* 08A ? */ + SW (0); /* 08C ? */ + SW (diwstrt); /* 08E DIWSTRT */ + SW (diwstop); /* 090 DIWSTOP */ + SW (ddfstrt); /* 092 DDFSTRT */ + SW (ddfstop); /* 094 DDFSTOP */ + SW (dmacon); /* 096 DMACON */ + SW (clxcon); /* 098 CLXCON */ + SW (intena); /* 09A INTENA */ + SW (intreq); /* 09C INTREQ */ + SW (adkcon); /* 09E ADKCON */ + for (i = 0; i < 8; i++) + SL (bplpt[i]); /* 0E0-0FE BPLxPT */ + SW (bplcon0); /* 100 BPLCON0 */ + SW (bplcon1); /* 102 BPLCON1 */ + SW (bplcon2); /* 104 BPLCON2 */ + SW (bplcon3); /* 106 BPLCON3 */ + SW (bpl1mod); /* 108 BPL1MOD */ + SW (bpl2mod); /* 10A BPL2MOD */ + SW (bplcon4); /* 10C BPLCON4 */ + SW (clxcon2); /* 10E CLXCON2 */ + for (i = 0;i < 8; i++) + SW (0); /* 110 BPLxDAT */ + for ( i = 0; i < 32; i++) + SW (current_colors.color_regs_ecs[i]); /* 180-1BE COLORxx */ + SW (0); /* 1C0 */ + SW (0); /* 1C2 */ + SW (0); /* 1C4 */ + SW (0); /* 1C6 */ + SW (0); /* 1C8 */ + SW (0); /* 1CA */ + SW (0); /* 1CC */ + SW (0); /* 1CE */ + SW (0); /* 1D0 */ + SW (0); /* 1D2 */ + SW (0); /* 1D4 */ + SW (0); /* 1D6 */ + SW (0); /* 1D8 */ + SW (0); /* 1DA */ + SW (beamcon0); /* 1DC BEAMCON0 */ + SW (0); /* 1DE */ + SW (0); /* 1E0 */ + SW (0); /* 1E2 */ + SW (0); /* 1E4 */ + SW (0); /* 1E6 */ + SW (0); /* 1E8 */ + SW (0); /* 1EA */ + SW (0); /* 1EC */ + SW (0); /* 1EE */ + SW (0); /* 1F0 */ + SW (0); /* 1F2 */ + SW (0); /* 1F4 */ + SW (0); /* 1F6 */ + SW (0); /* 1F8 */ + SW (0); /* 1FA */ + SW (fmode); /* 1FC FMODE */ + SW (0xffff); /* 1FE */ + + *len = dst - dstbak; + return dstbak; +} + +uae_u8 *restore_custom_agacolors (uae_u8 *src) +{ + int i; + + for (i = 0; i < 256; i++) + current_colors.color_regs_aga[i] = RL; + return src; +} + +uae_u8 *save_custom_agacolors (int *len) +{ + uae_u8 *dstbak, *dst; + int i; + + dstbak = dst = malloc (256*4); + for (i = 0; i < 256; i++) + SL (current_colors.color_regs_aga[i]); + *len = dst - dstbak; + return dstbak; +} + +uae_u8 *restore_custom_sprite (uae_u8 *src, int num) +{ + spr[num].pt = RL; /* 120-13E SPRxPT */ + sprpos[num] = RW; /* 1x0 SPRxPOS */ + sprctl[num] = RW; /* 1x2 SPRxPOS */ + sprdata[num][0] = RW; /* 1x4 SPRxDATA */ + sprdatb[num][0] = RW; /* 1x6 SPRxDATB */ + sprdata[num][1] = RW; + sprdatb[num][1] = RW; + sprdata[num][2] = RW; + sprdatb[num][2] = RW; + sprdata[num][3] = RW; + sprdatb[num][3] = RW; + spr[num].armed = RB; + return src; +} + +uae_u8 *save_custom_sprite(int *len, int num) +{ + uae_u8 *dstbak, *dst; + + dstbak = dst = malloc (25); + SL (spr[num].pt); /* 120-13E SPRxPT */ + SW (sprpos[num]); /* 1x0 SPRxPOS */ + SW (sprctl[num]); /* 1x2 SPRxPOS */ + SW (sprdata[num][0]); /* 1x4 SPRxDATA */ + SW (sprdatb[num][0]); /* 1x6 SPRxDATB */ + SW (sprdata[num][1]); + SW (sprdatb[num][1]); + SW (sprdata[num][2]); + SW (sprdatb[num][2]); + SW (sprdata[num][3]); + SW (sprdatb[num][3]); + SB (spr[num].armed ? 1 : 0); + *len = dst - dstbak; + return dstbak; +}