diff --git a/.gitignore b/.gitignore index 30c6424..ad24269 100644 --- a/.gitignore +++ b/.gitignore @@ -10,3 +10,5 @@ FLOPPIES/ gost .vscode/ .tmp/ +go.work +go.work.sum diff --git a/CHANGELOG.md b/CHANGELOG.md index baf7adf..3cd9c7a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,29 @@ ## Unreleased +### Changed + +- Fetches and TOS-vector reads from the boot ROM now bypass the bus via a + read-only flat memory window (`m68kemu` `SetFastMemory`), cutting an EmuTOS + boot benchmark by ~8%. Wait-state accounting is unchanged, so cycle counts, + register state, and the debug-trace suite are byte-for-byte identical to the + bus path; `machine.disableFastMemory` forces every access back through the bus. +- `EnableTrace` installs CPU observation callbacks through the single + `m68kemu` `SetHooks` call instead of separate setters. +- `Machine.RequestInterrupt` and `devices.Interrupt.Vector` now take a plain + `uint8` vector; pass `cpu.AutoVector` / `devices.AutoVector` (zero) to + auto-vector, replacing the previous `*uint8`. +- The single-range peripherals (ACIA, GLUE, MFP, PSG, FDC, Blitter, STE sound, + cartridge ROM) expose `AddressRange()` and drop their `Contains` boilerplate; + the bus page-maps them instead of scanning. + +### Dependencies + +- `github.com/jenska/m68kemu` updated to v1.5.0 for the `WithDeferredReset`, + `SetHooks`/`Hooks`, and `SetFastMemory` APIs, a polished public surface + (plain-value `RequestInterrupt`, `*Bus` on `NewCPU`, fewer exported + internals), and optional `Device.Contains`. + ## v0.4.0 - 2026-09-04 ### Added diff --git a/configs/atari-1040st-tos100.json b/configs/atari-1040st-tos100.json new file mode 100644 index 0000000..c23bbc3 --- /dev/null +++ b/configs/atari-1040st-tos100.json @@ -0,0 +1,8 @@ +{ + "preset": "st", + "model": "st", + "ram-size": 1048576, + "color-monitor": true, + "rom": "TOS/TOS100GE.IMG", + "scale": 2 +} diff --git a/docs/m68kemu-api-proposal.md b/docs/m68kemu-api-proposal.md new file mode 100644 index 0000000..bd86937 --- /dev/null +++ b/docs/m68kemu-api-proposal.md @@ -0,0 +1,374 @@ +# m68kemu API changes to simplify gost integration + +Status: partially implemented +Author: gost maintainers +Target: `github.com/jenska/m68kemu` (currently v1.4.0) +Scope: additive where possible; one semantic change (item 2) behind an opt-in + +## 0. Progress + +Landed in `m68kemu` (branch `api-additive-hooks-fastram`) and wired into `gost`: + +- **3.6 `WithDeferredReset`** — implemented in `m68kemu`; not yet adopted by + `gost` (the current construction order already leaves devices in a valid + post-cold-reset state, so it buys little until the scheduler work). +- **3.5 `SetHooks`** — implemented and adopted; `EnableTrace` now installs the + whole callback set in one call. +- **3.1 fast memory** — implemented as `SetFastMemory(...FastRegion)` (multi + region, with a `ReadOnly` flag and bus-matching wait-state accounting). + `gost` maps a read-only window over each isolated ROM mirror only. A low-RAM + window was prototyped and dropped: the shifter's variable video-DMA + contention penalty cannot be reproduced on a flat region, so it made cycle + counts drift from the bus model. ROM has no such penalty, so the ROM window + is cycle-exact (verified by a 45-frame boot-parity test) and still removes + ~8% of an EmuTOS-boot benchmark, since EmuTOS executes entirely from ROM. + +Also landed, as a separate surface-polish pass (`m68kemu` gost is the sole +consumer): `RequestInterrupt` takes a plain `uint8` vector with an `AutoVector` +constant instead of `*uint8` (the minimum of **3.3**); `NewCPU` takes `*Bus`; +`SetPreTracer` / `SetInterruptTracer` folded into `SetHooks`; and +`InterruptController`, `MappedDevice`, `ScheduledEvent`, `WaitHook`, +`Bus.SetWaitHook` unexported. + +Primitives landed in `m68kemu` (branch `api-scheduler-irq`) but **not adopted by +`gost`**: + +- **3.2 `CycleScheduler.SetClockRatio`** and **3.3 `CPU.SetIRQSource`** — both + implemented, tested, opt-in (the queue and the 1:1 scheduler are unchanged + when unused). +- Wiring `gost` onto the scheduler was tried and reverted. It is behaviourally + correct — the debug-trace suite and the fast/bus cycle-parity tests were + unchanged — but ~2× slower: the current devices advance in an "advance me by + N cycles" style, so as `CycleListener`s they do their per-quantum work (GLUE + scanline loop, MFP GPIP-edge scan, interrupt drain) on *every instruction* + instead of every ~512-cycle quantum. The scheduler pays off only after the + device layer is rewritten event-driven (each device schedules its next state + change), which is its own project. Until then `gost` keeps the quantum loop + and `cpuCyclesForHardwareCycles`. + +- **3.4 optional `Contains`** — landed. `m68kemu`'s `Device` no longer requires + `Contains`; a device is located by `AddressRangeDevice` (page-mapped) and/or + `ContainsDevice`. `gost`'s eight single-range peripherals now expose + `AddressRange()` and drop `Contains`; the shifter, ROM, RAM, overlay, + memory-config, and multi-range regions keep `Contains`. The bus resolves each + check once, so the multi-device lookup is slightly faster. Debug-trace suite + unchanged. + +Not started: the device-layer event-driven rewrite that 3.2/3.3 adoption +depends on. + +## 1. Motivation + +`gost` drives `m68kemu` as the CPU core of an Atari ST. Three subsystems that +`m68kemu` already ships are reimplemented inside `gost` because the library +versions do not fit the ST's timing and interrupt model, and a fourth +(the interpreter fast path) is unreachable from a multi-device bus. + +| gost implements | m68kemu already has | why the library version is unused | +|---|---|---| +| `clocked []Clocked`, `advanceDevices`, `EventPredictor`, `nextStepQuantum`, `cpuCyclesForHardwareCycles`, `cpuCycleCarry` | `CycleScheduler`, `CycleListener`, `Schedule`/`ScheduleAfter` | no CPU-clock ↔ device-clock ratio; no "advance to next scheduled event" from inside `RunCycles` | +| `irqSources`, `InterruptSource.DrainInterrupts`, `dispatchInterrupts`, `maskedAutovectorPulse` | `InterruptController`, `CPU.RequestInterrupt` | controller queues every request indefinitely and never coalesces; a masked autovector pulse is delivered late instead of dropped, so `gost` inspects `cpu.Registers().SR` by hand to discard it | +| `Machine.RunUntil` — a per-instruction loop that re-aggregates `RunResult` | `CPU.RunUntil` | cannot advance devices or sample interrupts between instructions, so `gost` forces `MaxInstructions = 1` and defeats the internal loop | +| — | `Bus.fastRAM`, `cpu.fastFetchMem` single-RAM fast path | only enabled when the bus holds exactly one device and it is a `*m68kemu.RAM` (`bus.go`, `refreshTopology`); `gost`'s bus has ~15 devices and its own `devices.RAM` | + +Relevant `gost` files: `internal/emulator/machine_runtime.go`, +`internal/emulator/machine_builder.go`, `internal/emulator/machine_io.go`, +`internal/devices/device.go`, `internal/devices/glue.go`, +`internal/devices/mfp.go`. + +## 2. Current data flow + +### 2.1 Frame stepping (`machine_runtime.go`) + +``` +StepFrame: + shifter.BeginFrame() + remaining = frameCycles // 8 MHz "hardware" cycles + while remaining > 0: + dispatchInterrupts() // drain every irqSource -> cpu.RequestInterrupt + quantum = min(remaining, 512, nextDeviceEventCycles()) + cpuQuant = cpuCyclesForHardwareCycles(quantum) // rational scale + carry + cpu.RunCycles(cpuQuant) + advanceDevices(quantum) // every Clocked.Advance(quantum) + shifter.AdvanceFrame(quantum) + dispatchInterrupts() + remaining -= quantum + shifter.EndFrame() +``` + +`nextDeviceEventCycles()` polls every `EventPredictor` for the soonest state +change so the quantum never steps over an HBL boundary. `cpuCycleCarry` holds the +sub-cycle remainder of the `CPUClockHz / ClockHz` conversion. + +### 2.2 Interrupts + +`GLUE` (`glue.go`) appends a level-2 pulse per scanline and a level-4 pulse at +frame end into a slice. `MFP` does the same for its timers. `dispatchInterrupts` +calls `DrainInterrupts()` on each source, then for every returned `Interrupt`: + +```go +func (m *Machine) maskedAutovectorPulse(irq devices.Interrupt) bool { + if irq.Vector != nil { // vectored (MFP) interrupts are never dropped + return false + } + mask := uint8((m.cpu.Registers().SR >> 8) & 0x7) + return irq.Level <= mask // autovector pulse masked right now -> drop it +} +``` + +This hand-rolled masking exists only because `InterruptController.Request` +enqueues the pulse and delivers it whenever the mask next drops, which is wrong +for the ST's edge-triggered autovector lines. + +### 2.3 `RunUntil` (debugger / test stepping) + +`Machine.RunUntil` copies the caller's `RunUntilOptions`, sets +`MaxInstructions = 1`, calls `cpu.RunUntil` in a loop, and after each instruction +runs `advanceDevices(advanced)` + `dispatchInterrupts()`, then merges +`result.Instructions/Cycles/PC/Exception/BusAccess/Interrupt` back into a running +`RunResult` — about 50 lines that duplicate the library's own loop. + +## 3. Proposals + +Ordered by payoff. Items 1, 3, 4, 5, 6 are backward compatible. Item 2 changes +`CycleListener` timing only for callers that opt in with `SetClockRatio`. + +### 3.1 Caller-supplied fast-memory window + +```go +// SetFastRAM installs a flat byte slice that the interpreter reads and writes +// directly for addresses in [base, base+len(mem)), bypassing the bus. The caller +// guarantees the region is plain RAM: no access side effects, no MMU remapping, +// stable for the lifetime of the mapping. Passing a zero-length slice clears it. +// Addresses outside the window fall through to the bus unchanged. +func (c *CPU) SetFastRAM(base uint32, mem []byte) +``` + +Cleaner alternative — keep the guarantee on the device: + +```go +// A device may implement FastMemory to expose a directly-addressable slice. +// The bus picks the widest such region and hands it to the core. +type FastMemory interface { + FastSlice() (base uint32, mem []byte, ok bool) +} +``` + +`gost` maps only the always-present low ST RAM (below the MMU bank-switch, +"absent", and high-mirror ranges handled in `devices/ram.go`). TOS executes +almost entirely from low RAM, so this restores most of the throughput lost to the +`deviceForAddress -> page-map binary search -> MappedDevice bounds check -> +interface call -> RAM.translate()` chain that every fetch and RAM word currently +pays. + +Interaction: the fast window must be consulted **before** breakpoints and bus +tracing are disabled — i.e. it participates in the same `refreshRunModes` gate +that already governs `fastFetchMem`. When any execute/read/write breakpoint or +bus tracer is active, the core must route through the bus so those hooks still +fire. + +### 3.2 Clock ratio on the scheduler + +```go +// SetClockRatio makes scheduler time run in device cycles rather than CPU +// cycles. After SetClockRatio(deviceHz, cpuHz), CycleListener.AdvanceCycles +// deltas, Now(), and Schedule() "at" values are all expressed in device cycles, +// and the fractional remainder of the conversion is retained internally. +// The default 1:1 ratio preserves current behaviour. +func (s *CycleScheduler) SetClockRatio(deviceHz, cpuHz uint64) +``` + +With this, `gost` registers `glue`, `mfp`, `acia`, `fdc`, `psg`, `steSound`, and +`shifter` as `CycleListener`s and deletes: `advanceDevices`, `nextStepQuantum`, +`nextDeviceEventCycles`, `cpuCyclesForHardwareCycles`, `cpuCycleCarry`, +`stepQuantumCycles`, and the `Clocked` and `EventPredictor` interfaces. Devices +that predict events (`GLUE.NextEventCycles`, `MFP.NextEventCycles`) instead call +`scheduler.ScheduleAfter(delta, fn)` when their state last changed; the scheduler +already fires callbacks at the exact cycle and bounds `Advance` to the next +event, which is what `nextDeviceEventCycles` approximated. + +Open question: `CycleScheduler.Advance` is currently called from +`cpu.addCycles`, i.e. after each instruction's cycle cost is booked. That is fine +for device advancement. The shifter needs writes to `FF82xx` registers to take +effect at the right raster position; per-instruction granularity (typ. 4-40 +cycles) is finer than today's 512-cycle quantum, so this is an improvement, but +the shifter's `BeginFrame`/`EndFrame` bracketing still has to be driven by +`gost` around `RunCycles`. + +### 3.3 Pull-based interrupt line + +```go +// IRQSource reports the interrupt line state sampled at each instruction +// boundary, before the next opcode is fetched. level 0 means no request. +// When autovector is true, vector is ignored and the CPU uses 24+level. +type IRQSource interface { + PendingIRQ() (level uint8, vector uint8, autovector bool) +} + +func (c *CPU) SetIRQSource(IRQSource) +``` + +This is the Musashi / UAE model: the line is level-sensitive. A request that is +masked when sampled simply is not taken yet and is re-sampled next boundary; if +the device lowers the line first, it is never taken. ST-style HBL/VBL edge +behaviour becomes the device's decision — `GLUE` lowers its line after the CPU +acknowledges (observed via the existing `InterruptCallback`). + +`gost` deletes `DrainInterrupts`, `dispatchInterrupts`, `maskedAutovectorPulse`, +the `irqSources` slice, and the `Interrupt` / `InterruptSource` types. In their +place, one aggregator: + +```go +func (m *Machine) PendingIRQ() (uint8, uint8, bool) { + // highest of GLUE / MFP / ACIA / FDC lines; MFP wins ties with its vector +} +``` + +Because interrupts are now sampled inside the core, `cpu.RunUntil` advances +devices (via 3.2) and interrupts correctly on its own, so **`Machine.RunUntil` +is deleted** and callers use `m.cpu.RunUntil(options)` directly. + +Minimum viable version if the full redesign is too large for one release: + +```go +const AutoVector uint8 = 0 // sentinel for RequestInterrupt + +func (c *CPU) RequestInterrupt(level, vector uint8) error // was (level uint8, vector *uint8) +``` + +plus coalescing in `InterruptController` so a device that re-requests the same +level every boundary does not grow the queue without bound. + +### 3.4 Optional `Contains` for range devices + +Every `gost` device implements both `Contains(uint32) bool` and the +`AddressRangeDevice` interface `AddressRange() (start, end uint32)`. The +hand-written `Contains` methods are a recurring bug source — see the high-mirror +and MMU-size logic in `devices/ram.go:Contains`. + +Proposal: if a device implements `AddressRangeDevice` but not `Device.Contains`, +the bus synthesises containment from the range. Split the interface: + +```go +type Device interface { + Read(Size, uint32) (uint32, error) + Write(Size, uint32, uint32) error + Reset() +} + +type ContainsDevice interface { // optional; only for non-contiguous decode + Contains(address uint32) bool +} +``` + +Most `gost` devices then shrink to `AddressRange` + `Read` + `Write` + `Reset`. + +### 3.5 One hooks struct + +```go +type Hooks struct { + Trace TraceCallback + PreTrace PreTraceCallback + Exception ExceptionCallback + Bus BusAccessCallback + Interrupt InterruptCallback +} + +func (c *CPU) SetHooks(Hooks) // replaces the five SetXxxTracer setters +``` + +`gost`'s `EnableTrace` currently nils four setters and then re-sets a subset on +every mode change, each call re-running `refreshDebugModes` / `refreshRunModes`. +A single struct swap is atomic and runs the refresh once. Keep the individual +setters as thin wrappers for compatibility. + +### 3.6 Deferred reset in `NewCPU` + +`NewCPU` calls `c.Reset()` internally, which reads the reset vector from a bus +whose devices `gost` has not finished wiring. `gost` immediately calls +`Machine.Reset` again after construction. Add: + +```go +type Option func(*config) +func WithDeferredReset() Option +func NewCPU(bus AddressBus, opts ...Option) (CPU, error) +``` + +so construction does not touch the bus and the first real reset is the caller's. + +## 4. Effect on gost + +Deleted (~250-300 lines): + +- `machine_runtime.go`: `nextStepQuantum`, `nextDeviceEventCycles`, + `cpuCyclesForHardwareCycles`, `advanceDevices`, `dispatchInterrupts`, + `maskedAutovectorPulse`, `Machine.RunUntil`, fields `cpuCycleCarry` +- `device.go`: `Clocked`, `EventPredictor`, `InterruptSource`, `Interrupt` +- `machine_builder.go`: `clockedDevices`, the `irqSources` slice literal +- `glue.go` / `mfp.go`: `DrainInterrupts`, `draining` scratch buffers, + `NextEventCycles` (replaced by `ScheduleAfter` calls) + +`StepFrame` after the change: + +```go +func (m *Machine) StepFrame() (bool, error) { + m.shifter.BeginFrame() + if err := m.cpu.RunCycles(m.frameCycles); err != nil { + return false, err + } + frame := m.shifter.EndFrame() + if m.traceShifter() { + m.traceShifterFrame(frame) + } + m.frameCounter++ + return frame, nil +} +``` + +Setup in `NewMachineWithCartridge`: + +```go +sched := cpu.NewCycleScheduler() +sched.SetClockRatio(cfg.ClockHz, cfg.CPUClockHz) +for _, d := range []cpu.CycleListener{glue, mfp, acia, fdc, psg, shifter} { + sched.AddListener(d) +} +if steSound != nil { sched.AddListener(steSound) } +processor.SetScheduler(sched) +processor.SetIRQSource(machineIRQ{glue, mfp, acia, fdc}) +if lo := ram.FlatLowRegion(); len(lo) != 0 { + processor.SetFastRAM(0x000000, lo) +} +``` + +## 5. Compatibility and rollout + +| Item | Break? | Suggested release | +|---|---|---| +| 3.1 `SetFastRAM` | additive | minor | +| 3.2 `SetClockRatio` | additive; default 1:1 unchanged | minor | +| 3.3 `IRQSource` | additive; old `RequestInterrupt` kept | minor, deprecate queue path | +| 3.3 min. `RequestInterrupt` non-pointer | signature break | major, or new `RequestIRQ` name | +| 3.4 optional `Contains` | additive if `ContainsDevice` is a new optional interface | minor | +| 3.5 `SetHooks` | additive | minor | +| 3.6 `WithDeferredReset` | additive | minor | + +Recommended: ship 3.1, 3.2, 3.5, 3.6 in one minor release; land 3.3 and 3.4 in +the next once `gost` has migrated its device layer. + +## 6. Risks + +- **3.2**: moving device time inside `RunCycles` means a device callback that + itself touches the bus (DMA) runs mid-instruction-budget rather than at a + quantum edge. `gost`'s blitter and FDC DMA already assume they can read ST RAM + at any point, so this should be safe, but needs a regression pass against the + floppy and blitter test suites. +- **3.3**: the ST relies on HBL firing every scanline even under load. A + level-sensitive line that the device forgets to lower would stall the guest in + the level-2 handler. The `GLUE` change must lower the line in the same place it + currently clears `pending`. +- **3.1**: if the fast window ever overlaps an address the MMU can remap + (bank switch, `memoryAddressAbsent`), reads would bypass that logic silently. + `gost` must expose only the region below `MemoryConfig.LogicalSize()`'s + smallest possible value, and re-call `SetFastRAM` (or clear it) if the machine + ever gains runtime-reconfigurable low RAM. diff --git a/go.mod b/go.mod index 19cdbb0..42e8033 100644 --- a/go.mod +++ b/go.mod @@ -6,7 +6,7 @@ require ( github.com/ebitenui/ebitenui v0.7.3 github.com/hajimehoshi/ebiten/v2 v2.9.11 github.com/jenska/m68kdasm v1.1.0 - github.com/jenska/m68kemu v1.4.0 + github.com/jenska/m68kemu v1.5.0 github.com/jenska/ym2149 v1.1.0 golang.org/x/image v0.45.0 ) diff --git a/go.sum b/go.sum index fb8b9d8..7ff84b4 100644 --- a/go.sum +++ b/go.sum @@ -24,8 +24,8 @@ github.com/jenska/m68kasm v1.4.0 h1:/VDRRGtmLKynjQJxFmLsvp0nDEPtfIClSFB0+e+m7js= github.com/jenska/m68kasm v1.4.0/go.mod h1:0791j2jGF9H6JrtHde3qYZxCt2pJbQwT3VP7x/UZqYU= github.com/jenska/m68kdasm v1.1.0 h1:z50gR+U1fvrcz3LDv1li7x4nedvx6a/JRkyIQ++/XQU= github.com/jenska/m68kdasm v1.1.0/go.mod h1:fCE5ioFP52IewNAcb7wV5Pnj1hxTCA/82xb2lFaU1tc= -github.com/jenska/m68kemu v1.4.0 h1:6OmmTY48hu+SVLIIqpzIiP5LJjd+pGQCxnimIeHFmgY= -github.com/jenska/m68kemu v1.4.0/go.mod h1:kTiZoKMbJ1FAAHyTPRaMYtLq2WM+ZkTymnFWrvk5UlM= +github.com/jenska/m68kemu v1.5.0 h1:TzkQ+x8PGTn28y9yz2VT3etrN4DDZNlHIsogDz14dAc= +github.com/jenska/m68kemu v1.5.0/go.mod h1:kTiZoKMbJ1FAAHyTPRaMYtLq2WM+ZkTymnFWrvk5UlM= github.com/jenska/ym2149 v1.1.0 h1:1NDy8hOJY018LhcV2KymZ6lZE9tJB9wLlIzPwFwNIuA= github.com/jenska/ym2149 v1.1.0/go.mod h1:qQtXDjwXPrQ/UWIVY2hbyswTlwIYxnCHsRuJ7jiNyE0= github.com/jezek/xgb v1.3.1 h1:NQCAEfQyzN+3RjWUSHBuVIxQcy2YfG3/mNvKfs/0rEg= diff --git a/internal/devices/acia.go b/internal/devices/acia.go index faaf7a3..cccb438 100644 --- a/internal/devices/acia.go +++ b/internal/devices/acia.go @@ -35,8 +35,8 @@ func NewACIA(aciaIRQ func(bool)) *ACIA { } // Contains reports whether the given address is serviced by the ACIA. -func (a *ACIA) Contains(address uint32) bool { - return address >= aciaBase && address < aciaBase+aciaSize +func (a *ACIA) AddressRange() (uint32, uint32) { + return aciaBase, aciaBase + aciaSize - 1 } // WaitStates returns the fixed ACIA bus latency. diff --git a/internal/devices/acia_test.go b/internal/devices/acia_test.go index a58d19c..ed99035 100644 --- a/internal/devices/acia_test.go +++ b/internal/devices/acia_test.go @@ -162,7 +162,7 @@ func TestACIAKeyboardSignalsMFPInterruptOnReceive(t *testing.T) { if len(irqs) != 1 { t.Fatalf("expected one MFP interrupt, got %d", len(irqs)) } - if irqs[0].Vector == nil || *irqs[0].Vector != 0x46 { + if irqs[0].Vector != 0x46 { t.Fatalf("unexpected ACIA MFP vector: %+v", irqs[0].Vector) } @@ -198,7 +198,7 @@ func TestACIAMIDISignalsMFPInterruptOnReceive(t *testing.T) { if len(irqs) != 1 { t.Fatalf("expected one MFP interrupt, got %d", len(irqs)) } - if irqs[0].Vector == nil || *irqs[0].Vector != 0x46 { + if irqs[0].Vector != 0x46 { t.Fatalf("unexpected ACIA MFP vector: %+v", irqs[0].Vector) } diff --git a/internal/devices/blitter.go b/internal/devices/blitter.go index 7705281..0e37e01 100644 --- a/internal/devices/blitter.go +++ b/internal/devices/blitter.go @@ -33,8 +33,8 @@ func NewBlitter(ram *RAM) *Blitter { return b } -func (b *Blitter) Contains(address uint32) bool { - return address >= blitterBase && address < blitterBase+blitterSize +func (b *Blitter) AddressRange() (uint32, uint32) { + return blitterBase, blitterBase + blitterSize - 1 } func (b *Blitter) Read(size cpu.Size, address uint32) (uint32, error) { @@ -74,8 +74,12 @@ func (b *Blitter) Reset() { clear(b.regs[:]) } +func (b *Blitter) contains(address uint32) bool { + return address >= blitterBase && address < blitterBase+blitterSize +} + func (b *Blitter) offsetFor(address uint32, size cpu.Size) (uint32, error) { - if !b.Contains(address) { + if !b.contains(address) { return 0, cpu.BusError(address) } offset := address - blitterBase diff --git a/internal/devices/cartridge_rom.go b/internal/devices/cartridge_rom.go index a73bc7f..6a1d0e9 100644 --- a/internal/devices/cartridge_rom.go +++ b/internal/devices/cartridge_rom.go @@ -28,8 +28,8 @@ func NewCartridgeROM(image []byte) (*CartridgeROM, error) { return &CartridgeROM{data: append([]byte(nil), image...)}, nil } -func (c *CartridgeROM) Contains(address uint32) bool { - return address >= cartridgeROMBase && address < cartridgeROMBase+cartridgeROMSize +func (c *CartridgeROM) AddressRange() (uint32, uint32) { + return cartridgeROMBase, cartridgeROMBase + cartridgeROMSize - 1 } func (c *CartridgeROM) Read(size cpu.Size, address uint32) (uint32, error) { diff --git a/internal/devices/device.go b/internal/devices/device.go index 789c5e3..36f4cc9 100644 --- a/internal/devices/device.go +++ b/internal/devices/device.go @@ -6,12 +6,16 @@ import ( cpu "github.com/jenska/m68kemu" ) +// AutoVector is the Interrupt.Vector value that asks the CPU to auto-vector the +// request (vector 24+Level) rather than take a device-supplied vector. +const AutoVector = cpu.AutoVector + // Interrupt models a pending CPU interrupt from a device to the CPU. type Interrupt struct { // Level is the interrupt priority level (1-7). Level uint8 - // Vector is a pointer to the exception vector address if applicable, nil for autovector. - Vector *uint8 + // Vector is the exception vector number, or AutoVector to auto-vector. + Vector uint8 } // Clocked represents a device that advances its internal state with CPU cycles. diff --git a/internal/devices/fdc.go b/internal/devices/fdc.go index aa375eb..2e2d53a 100644 --- a/internal/devices/fdc.go +++ b/internal/devices/fdc.go @@ -150,8 +150,8 @@ func NewFDC(ram *RAM, irq func(bool)) *FDC { return f } -func (f *FDC) Contains(address uint32) bool { - return address >= fdcBase && address < fdcBase+fdcSize +func (f *FDC) AddressRange() (uint32, uint32) { + return fdcBase, fdcBase + fdcSize - 1 } func (f *FDC) WaitStates(cpu.Size, uint32) uint32 { @@ -1099,7 +1099,7 @@ func (f *FDC) commandSectorCount(cmd byte) (count int, multi bool) { func (f *FDC) queueInterrupt() { vector := f.vector - f.pending = append(f.pending, Interrupt{Level: 5, Vector: &vector}) + f.pending = append(f.pending, Interrupt{Level: 5, Vector: vector}) if f.irq != nil { f.irq(true) } diff --git a/internal/devices/glue.go b/internal/devices/glue.go index 0a377f2..ae744de 100644 --- a/internal/devices/glue.go +++ b/internal/devices/glue.go @@ -35,8 +35,8 @@ func NewGLUE(cfg ...*config.Config) *GLUE { return g } -func (g *GLUE) Contains(address uint32) bool { - return address >= glueBase && address < glueBase+glueSize +func (g *GLUE) AddressRange() (uint32, uint32) { + return glueBase, glueBase + glueSize - 1 } func (g *GLUE) Read(size cpu.Size, address uint32) (uint32, error) { diff --git a/internal/devices/glue_test.go b/internal/devices/glue_test.go index 4edaef4..0c796be 100644 --- a/internal/devices/glue_test.go +++ b/internal/devices/glue_test.go @@ -55,7 +55,7 @@ func TestGLUEQueuesHBLAtScanlineBoundary(t *testing.T) { if len(irqs) != 1 { t.Fatalf("expected one HBL interrupt, got %d", len(irqs)) } - if irqs[0].Level != 2 || irqs[0].Vector != nil { + if irqs[0].Level != 2 || irqs[0].Vector != AutoVector { t.Fatalf("unexpected HBL interrupt: %+v", irqs[0]) } } @@ -70,7 +70,7 @@ func TestGLUEQueuesVBLAtFrameBoundary(t *testing.T) { t.Fatalf("expected GLUE frame interrupts") } last := irqs[len(irqs)-1] - if last.Level != 4 || last.Vector != nil { + if last.Level != 4 || last.Vector != AutoVector { t.Fatalf("expected final frame interrupt to be VBL autovector, got %+v", last) } } diff --git a/internal/devices/mfp.go b/internal/devices/mfp.go index bd8f9d6..c20f3d8 100644 --- a/internal/devices/mfp.go +++ b/internal/devices/mfp.go @@ -102,8 +102,8 @@ func NewMFP(cfg *config.Config) *MFP { return m } -func (m *MFP) Contains(address uint32) bool { - return address >= mfpBase && address < mfpBase+mfpSize +func (m *MFP) AddressRange() (uint32, uint32) { + return mfpBase, mfpBase + mfpSize - 1 } func (m *MFP) WaitStates(cpu.Size, uint32) uint32 { @@ -213,7 +213,7 @@ func (m *MFP) DrainInterrupts() []Interrupt { } vector := m.vectorBase + uint8(channel) - return []Interrupt{{Level: 6, Vector: &vector}} + return []Interrupt{{Level: 6, Vector: vector}} } func (m *MFP) NextEventCycles() (uint64, bool) { diff --git a/internal/devices/mfp_test.go b/internal/devices/mfp_test.go index 4990297..98b64df 100644 --- a/internal/devices/mfp_test.go +++ b/internal/devices/mfp_test.go @@ -33,7 +33,7 @@ func TestMFPTimerQueuesInterrupt(t *testing.T) { if irqs[0].Level != 6 { t.Fatalf("unexpected interrupt level: got %d want 6", irqs[0].Level) } - if irqs[0].Vector == nil || *irqs[0].Vector != 0x4D { + if irqs[0].Vector != 0x4D { t.Fatalf("unexpected vector: %+v", irqs[0].Vector) } } @@ -62,7 +62,7 @@ func TestMFPTimerCQueuesInterrupt(t *testing.T) { if len(irqs) != 1 { t.Fatalf("expected 1 interrupt, got %d", len(irqs)) } - if irqs[0].Vector == nil || *irqs[0].Vector != 0x45 { + if irqs[0].Vector != 0x45 { t.Fatalf("unexpected vector: %+v", irqs[0].Vector) } } @@ -155,7 +155,7 @@ func TestMFPSoftwareEOIBlocksLowerPriorityInterrupts(t *testing.T) { if len(irqs) != 1 { t.Fatalf("expected 1 interrupt, got %d", len(irqs)) } - if irqs[0].Vector == nil || *irqs[0].Vector != 0x4D { + if irqs[0].Vector != 0x4D { t.Fatalf("unexpected vector for highest priority interrupt: %+v", irqs[0].Vector) } @@ -179,7 +179,7 @@ func TestMFPSoftwareEOIBlocksLowerPriorityInterrupts(t *testing.T) { if len(irqs) != 1 { t.Fatalf("expected 1 interrupt after software eoi, got %d", len(irqs)) } - if irqs[0].Vector == nil || *irqs[0].Vector != 0x48 { + if irqs[0].Vector != 0x48 { t.Fatalf("unexpected vector after software eoi: %+v", irqs[0].Vector) } } @@ -240,7 +240,7 @@ func TestMFPTimerAccumulatesFractionalCPUClock(t *testing.T) { if len(irqs) != 1 { t.Fatalf("expected 1 interrupt after 14 CPU cycles, got %d", len(irqs)) } - if irqs[0].Vector == nil || *irqs[0].Vector != 0x45 { + if irqs[0].Vector != 0x45 { t.Fatalf("unexpected vector: %+v", irqs[0].Vector) } } @@ -275,7 +275,7 @@ func TestMFPAutoEOITimerCRepeats(t *testing.T) { if len(irqs) != 1 { t.Fatalf("expected recurring timer c interrupt under auto-EOI, got %d", len(irqs)) } - if irqs[0].Vector == nil || *irqs[0].Vector != 0x45 { + if irqs[0].Vector != 0x45 { t.Fatalf("unexpected recurring timer c vector: %+v", irqs[0].Vector) } } @@ -346,7 +346,7 @@ func TestMFPSoftwareEOIPreventsDuplicateTimerDispatchBeforeServiceClear(t *testi if len(irqs) != 1 { t.Fatalf("expected pending timer c interrupt after service clear, got %d", len(irqs)) } - if irqs[0].Vector == nil || *irqs[0].Vector != 0x45 { + if irqs[0].Vector != 0x45 { t.Fatalf("unexpected vector after service clear: %+v", irqs[0].Vector) } } @@ -520,7 +520,7 @@ func TestMFPGPIPAERDefaultDetectsFallingACIAEdge(t *testing.T) { if len(irqs) != 1 { t.Fatalf("expected falling ACIA edge interrupt, got %d", len(irqs)) } - if irqs[0].Vector == nil || *irqs[0].Vector != 0x46 { + if irqs[0].Vector != 0x46 { t.Fatalf("unexpected falling ACIA edge vector: %+v", irqs[0].Vector) } @@ -556,7 +556,7 @@ func TestMFPGPIPAERDetectsRisingACIAEdgeWhenConfigured(t *testing.T) { if len(irqs) != 1 { t.Fatalf("expected rising ACIA edge interrupt, got %d", len(irqs)) } - if irqs[0].Vector == nil || *irqs[0].Vector != 0x46 { + if irqs[0].Vector != 0x46 { t.Fatalf("unexpected rising ACIA edge vector: %+v", irqs[0].Vector) } } @@ -607,7 +607,7 @@ func TestMFPGPIPAERDetectsFallingRTCEdgeOnGPIP5(t *testing.T) { if len(irqs) != 1 { t.Fatalf("expected falling RTC edge interrupt, got %d", len(irqs)) } - if irqs[0].Vector == nil || *irqs[0].Vector != 0x47 { + if irqs[0].Vector != 0x47 { t.Fatalf("unexpected falling RTC edge vector: %+v", irqs[0].Vector) } @@ -719,7 +719,7 @@ func TestMFPRS232ReceiveByteSetsStatusAndInterrupt(t *testing.T) { if len(irqs) != 1 { t.Fatalf("expected receive interrupt, got %d", len(irqs)) } - if irqs[0].Vector == nil || *irqs[0].Vector != 0x4C { + if irqs[0].Vector != 0x4C { t.Fatalf("unexpected receive vector: %+v", irqs[0].Vector) } @@ -811,7 +811,7 @@ func TestMFPRS232TransmitCapturesOutputAndInterrupts(t *testing.T) { if len(irqs) != 1 { t.Fatalf("expected transmit interrupt, got %d", len(irqs)) } - if irqs[0].Vector == nil || *irqs[0].Vector != 0x4A { + if irqs[0].Vector != 0x4A { t.Fatalf("unexpected transmit vector: %+v", irqs[0].Vector) } } diff --git a/internal/devices/psg.go b/internal/devices/psg.go index b864664..b44d49f 100644 --- a/internal/devices/psg.go +++ b/internal/devices/psg.go @@ -32,8 +32,8 @@ func NewPSG(cpuClockHz uint64) *PSG { } } -func (p *PSG) Contains(address uint32) bool { - return address >= psgBase && address < psgBase+psgSize +func (p *PSG) AddressRange() (uint32, uint32) { + return psgBase, psgBase + psgSize - 1 } func (p *PSG) WaitStates(cpu.Size, uint32) uint32 { diff --git a/internal/devices/rom.go b/internal/devices/rom.go index f8dce61..f88ff4f 100644 --- a/internal/devices/rom.go +++ b/internal/devices/rom.go @@ -86,6 +86,17 @@ func (r *ROM) Bytes() []byte { return append([]byte(nil), r.data...) } +// Image returns the live ROM backing bytes. The contents are immutable, so the +// slice is safe to hand to a read-only fast-memory region. +func (r *ROM) Image() []byte { + return r.data +} + +// Aliases returns the base addresses at which the ROM image is mapped. +func (r *ROM) Aliases() []uint32 { + return append([]uint32(nil), r.aliases...) +} + func (r *ROM) Slice(address uint32, size cpu.Size) (uint32, error) { return r.Read(size, address) } diff --git a/internal/devices/ste_sound.go b/internal/devices/ste_sound.go index e255ae7..60b255b 100644 --- a/internal/devices/ste_sound.go +++ b/internal/devices/ste_sound.go @@ -70,8 +70,8 @@ func NewAbsentSTESound() *BusErrorRegion { return NewBusErrorRegion(AddressRange{Start: steSoundBase, End: steSoundBase + steSoundSize}) } -func (s *STESound) Contains(address uint32) bool { - return address >= steSoundBase && address < steSoundBase+steSoundSize +func (s *STESound) AddressRange() (uint32, uint32) { + return steSoundBase, steSoundBase + steSoundSize - 1 } func (s *STESound) WaitStates(cpu.Size, uint32) uint32 { @@ -340,12 +340,16 @@ func (s *STESound) playing() bool { return s.control&steSoundControlEnable != 0 } +func (s *STESound) contains(address uint32) bool { + return address >= steSoundBase && address < steSoundBase+steSoundSize +} + func (s *STESound) accessInRange(address uint32, count int) bool { if count <= 0 { return false } end := address + uint32(count) - 1 - return s.Contains(address) && s.Contains(end) + return s.contains(address) && s.contains(end) } func steSoundAccessSize(size cpu.Size) (int, error) { diff --git a/internal/emulator/machine_fastram_bench_test.go b/internal/emulator/machine_fastram_bench_test.go new file mode 100644 index 0000000..e785ecc --- /dev/null +++ b/internal/emulator/machine_fastram_bench_test.go @@ -0,0 +1,35 @@ +package emulator + +import ( + "testing" + + "github.com/jenska/gost/internal/assets" +) + +// benchmarkBootFrames steps a freshly booted EmuTOS machine for a fixed number +// of frames. Most of that work is the CPU interpreting TOS code out of ROM with +// data in low RAM, so it is a reasonable proxy for overall emulation throughput. +func benchmarkBootFrames(b *testing.B, frames int, disableFastMem bool) { + b.Helper() + rom := assets.DefaultROM() + + b.ReportAllocs() + for b.Loop() { + machine, err := NewMachine(DefaultConfig(), rom) + if err != nil { + b.Fatalf("NewMachine: %v", err) + } + machine.disableFastMemory = disableFastMem + if err := machine.Reset(); err != nil { + b.Fatalf("Reset: %v", err) + } + for frame := 0; frame < frames; frame++ { + if _, err := machine.StepFrame(); err != nil { + b.Fatalf("StepFrame %d: %v", frame, err) + } + } + } +} + +func BenchmarkBootFramesFastMem(b *testing.B) { benchmarkBootFrames(b, 120, false) } +func BenchmarkBootFramesBusOnly(b *testing.B) { benchmarkBootFrames(b, 120, true) } diff --git a/internal/emulator/machine_fastram_test.go b/internal/emulator/machine_fastram_test.go new file mode 100644 index 0000000..23977a2 --- /dev/null +++ b/internal/emulator/machine_fastram_test.go @@ -0,0 +1,129 @@ +package emulator + +import ( + "testing" + + "github.com/jenska/gost/internal/assets" + cpu "github.com/jenska/m68kemu" +) + +func TestRomAliasIsIsolated(t *testing.T) { + const k256 = 256 * 1024 + const k192 = 192 * 1024 + + cases := []struct { + name string + base uint32 + length uint32 + want bool + }{ + {"emutos 256K at E00000", 0xE00000, k256, true}, + {"tos 1.x 192K at FC0000", 0xFC0000, k192, true}, + {"256K at FC0000 reaches I/O", 0xFC0000, k256, false}, + {"reaches cartridge probe", 0xF90000, k256, false}, + {"below ROM space", 0x000000, k256, false}, + {"zero length", 0xE00000, 0, false}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if got := romAliasIsIsolated(tc.base, tc.length); got != tc.want { + t.Fatalf("romAliasIsIsolated(%06x, %d) = %v, want %v", tc.base, tc.length, got, tc.want) + } + }) + } +} + +func hasFastRegionAt(m *Machine, address uint32) bool { + for _, r := range m.fastRegions { + if address >= r.Base && uint64(address) < uint64(r.Base)+uint64(len(r.Mem)) { + return true + } + } + return false +} + +// TestFastRegionsCoverRomEntry verifies the reset PC lands in a fast region. +func TestFastRegionsCoverRomEntry(t *testing.T) { + m := mustMachine(t, loopROM(nil)) + if err := m.Reset(); err != nil { + t.Fatalf("Reset: %v", err) + } + if !hasFastRegionAt(m, m.Registers().PC) { + t.Fatalf("no fast region covers reset PC %06x; regions=%+v", m.Registers().PC, m.fastRegions) + } + for _, r := range m.fastRegions { + if !r.ReadOnly { + t.Fatalf("fast region at %06x is not read-only", r.Base) + } + } +} + +// TestFastRegionCycleCountMatchesBus runs a short ROM-resident program that +// mixes word and long fetches with long RAM load/store, once with the ROM fast +// region and once purely on the bus. The total cycle count must be identical. +func TestFastRegionCycleCountMatchesBus(t *testing.T) { + // FC0008: MOVE.L $00001200,D0 2039 0000 1200 + // FC000E: MOVE.L D0,$00001300 23C0 0000 1300 + // FC0014: MOVEQ #1,D1 7201 + // FC0016: ADD.L D1,D0 D081 + // FC0018: BRA * 60FE + prog := []byte{ + 0x20, 0x39, 0x00, 0x00, 0x12, 0x00, + 0x23, 0xC0, 0x00, 0x00, 0x13, 0x00, + 0x72, 0x01, + 0xD0, 0x81, + 0x60, 0xFE, + } + + run := func(disable bool) uint64 { + m, err := NewMachine(DefaultConfig(), loopROM(prog)) + if err != nil { + t.Fatalf("NewMachine: %v", err) + } + m.disableFastMemory = disable + if err := m.Reset(); err != nil { + t.Fatalf("Reset: %v", err) + } + if !disable && !hasFastRegionAt(m, m.Registers().PC) { + t.Fatalf("expected a ROM fast region over the PC") + } + if _, err := m.RunUntil(cpu.RunUntilOptions{MaxInstructions: 8}); err != nil { + t.Fatalf("RunUntil: %v", err) + } + return m.Cycles() + } + + if fast, bus := run(false), run(true); fast != bus { + t.Fatalf("cycle count differs: fast=%d bus=%d", fast, bus) + } +} + +// TestFastRegionBootParity is the regression guard for the fast-memory path: a +// real EmuTOS boot must land on exactly the same cycle count, PC and SR whether +// the ROM fast region is used or every fetch goes through the bus. +func TestFastRegionBootParity(t *testing.T) { + run := func(disable bool) (uint64, uint32, uint16) { + m, err := NewMachine(DefaultConfig(), assets.DefaultROM()) + if err != nil { + t.Fatalf("NewMachine: %v", err) + } + m.disableFastMemory = disable + if err := m.Reset(); err != nil { + t.Fatalf("Reset: %v", err) + } + for frame := 0; frame < 45; frame++ { + if _, err := m.StepFrame(); err != nil { + t.Fatalf("StepFrame %d: %v", frame, err) + } + } + r := m.Registers() + return m.Cycles(), r.PC, r.SR + } + + bc, bpc, bsr := run(true) + fc, fpc, fsr := run(false) + if bc != fc || bpc != fpc || bsr != fsr { + t.Fatalf("boot diverged: bus=(cyc %d pc %06x sr %04x) fast=(cyc %d pc %06x sr %04x)", + bc, bpc, bsr, fc, fpc, fsr) + } +} diff --git a/internal/emulator/machine_io.go b/internal/emulator/machine_io.go index ddda0aa..f20346b 100644 --- a/internal/emulator/machine_io.go +++ b/internal/emulator/machine_io.go @@ -60,6 +60,12 @@ type ( cpuCycleCarry uint64 traceWriter io.Writer frameCounter uint64 + // fastRegions is the reusable buffer of flat-memory regions handed to the + // CPU via SetFastMemory (installed once per reset, ROM mirrors only). + fastRegions []cpu.FastRegion + // disableFastMemory forces every access back through the bus. Used by + // benchmarks and by anyone debugging a suspected fast-region fault. + disableFastMemory bool } ) @@ -177,7 +183,9 @@ func (m *Machine) EjectFloppy(drive int) error { return m.fdc.EjectDiskFromDrive(drive) } -func (m *Machine) RequestInterrupt(level uint8, vector *uint8) error { +// RequestInterrupt queues a CPU interrupt at the given level (1-7). Pass +// cpu.AutoVector for vector to auto-vector it. +func (m *Machine) RequestInterrupt(level, vector uint8) error { return m.cpu.RequestInterrupt(level, vector) } diff --git a/internal/emulator/machine_runtime.go b/internal/emulator/machine_runtime.go index b847710..eda9f2c 100644 --- a/internal/emulator/machine_runtime.go +++ b/internal/emulator/machine_runtime.go @@ -14,7 +14,54 @@ func (m *Machine) Reset() error { m.memoryConfig.ColdReset() m.cpuCycleCarry = 0 m.frameCounter = 0 - return m.cpu.Reset() + if err := m.cpu.Reset(); err != nil { + return err + } + m.installFastMemory() + return nil +} + +// installFastMemory hands the CPU a read-only flat window over each isolated ROM +// mirror, letting instruction fetches and TOS-vector reads bypass the bus. +// EmuTOS executes entirely from ROM, so this covers the bulk of the fetch +// traffic; the ROM has no address-dependent wait states, so a fast-region access +// costs exactly what the bus path would (bus 4 + ROM 4 per word transfer). +// +// Low RAM is deliberately not exposed: the shifter charges a variable video-DMA +// contention penalty on RAM access that a flat region cannot reproduce, which +// would make cycle counts diverge from the bus model. +func (m *Machine) installFastMemory() { + m.fastRegions = m.fastRegions[:0] + if !m.disableFastMemory { + image := m.rom.Image() + for _, base := range m.rom.Aliases() { + if !romAliasIsIsolated(base, uint32(len(image))) { + continue + } + m.fastRegions = append(m.fastRegions, cpu.FastRegion{ + Base: base, Mem: image, WaitStates: 8, ReadOnly: true, + }) + } + } + m.cpu.SetFastMemory(m.fastRegions...) +} + +// romAliasIsIsolated reports whether a ROM mirror of the given length at base is +// backed only by ROM across its whole span - i.e. it does not reach into the +// cartridge probe window or the hardware register area, where other bus devices +// must win. Only isolated mirrors are safe to serve from a flat fast region. +func romAliasIsIsolated(base, length uint32) bool { + end := uint64(base) + uint64(length) // exclusive; may exceed the 24-bit space + if base < secondaryROMAlias || length == 0 { + return false + } + if base < 0xFA0010 && end > 0xFA0000 { // cartridge / RTC probe region + return false + } + if end > 0xFF8000 { // memory-mapped I/O + return false + } + return true } func (m *Machine) StepFrame() (bool, error) { @@ -173,7 +220,7 @@ func (m *Machine) dispatchInterrupts() { // interrupts behave for software that briefly raises IPL during critical // sections; vectored (MFP) interrupts are never dropped this way. func (m *Machine) maskedAutovectorPulse(irq devices.Interrupt) bool { - if irq.Vector != nil { + if irq.Vector != devices.AutoVector { return false } mask := uint8((m.cpu.Registers().SR >> 8) & 0x7) diff --git a/internal/emulator/machine_test.go b/internal/emulator/machine_test.go index 61a9c14..e9983dc 100644 --- a/internal/emulator/machine_test.go +++ b/internal/emulator/machine_test.go @@ -361,7 +361,7 @@ func TestMachineInterruptHandling(t *testing.T) { t.Fatalf("load interrupt handler: %v", err) } - if err := machine.RequestInterrupt(6, nil); err != nil { + if err := machine.RequestInterrupt(6, cpu.AutoVector); err != nil { t.Fatalf("request interrupt: %v", err) } @@ -905,7 +905,7 @@ func TestMachineKeepsMaskedVectoredInterruptPending(t *testing.T) { machine := mustMachine(t, rom) vector := uint8(64) machine.irqSources = []devices.InterruptSource{ - &testIRQSource{pending: []devices.Interrupt{{Level: 6, Vector: &vector}}}, + &testIRQSource{pending: []devices.Interrupt{{Level: 6, Vector: vector}}}, } handlerAddress := uint32(0x00002000) diff --git a/internal/emulator/machine_trace.go b/internal/emulator/machine_trace.go index 5458a23..0f26cf1 100644 --- a/internal/emulator/machine_trace.go +++ b/internal/emulator/machine_trace.go @@ -74,34 +74,34 @@ func (m *Machine) EnableTrace(mode string, writer io.Writer) { writer = io.Discard } m.traceWriter = writer - m.cpu.SetTracer(nil) - m.cpu.SetBusTracer(nil) - m.cpu.SetExceptionTracer(nil) m.shifter.SetDebug(false) + // EnableTrace fully owns the CPU observation hooks: every call replaces the + // whole set, so an unhandled mode clears them. + var hooks cpu.Hooks switch TraceMode(mode) { case TraceModeSimple: - m.cpu.SetTracer(func(info cpu.TraceInfo) { + hooks.Trace = func(info cpu.TraceInfo) { fmt.Fprintf(m.traceWriter, "pc=%06x sr=%04x cycles=%d\n", info.PC, info.SR, m.cpu.Cycles()) - }) + } case TraceModeSimpleVerbose: logger := cpu.NewVerboseLogger(m.cpu, m.bus, m.traceWriter, cpu.VerboseLoggerOptions{ IncludeCycles: true, }) - m.cpu.SetTracer(logger.Trace) + hooks.Trace = logger.Trace case TraceModeBootSimple: - m.enableBootTrace(false) + hooks = m.bootTraceHooks(false) case TraceModeBootVerbose: - m.enableBootTrace(true) + hooks = m.bootTraceHooks(true) case TraceModeShifterSimple, TraceModeShifterVerbose: m.shifter.SetDebug(true) - default: - m.cpu.SetTracer(nil) } + m.cpu.SetHooks(hooks) } -func (m *Machine) enableBootTrace(verbose bool) { - m.cpu.SetBusTracer(func(info cpu.BusAccessInfo) { +func (m *Machine) bootTraceHooks(verbose bool) cpu.Hooks { + var hooks cpu.Hooks + hooks.Bus = func(info cpu.BusAccessInfo) { address := info.Address & 0xFFFFFF if info.InstructionFetch || !bootTraceAddressSet[address] { return @@ -116,8 +116,8 @@ func (m *Machine) enableBootTrace(verbose bool) { regs.A[7], traceValueString(info.Size, info.Value), ) - }) - m.cpu.SetExceptionTracer(func(info cpu.ExceptionInfo) { + } + hooks.Exception = func(info cpu.ExceptionInfo) { if info.FaultValid { fmt.Fprintf(m.traceWriter, "exception vector=%d pc=%06x newpc=%06x opcode=%04x fault=%06x sr=%04x newsr=%04x\n", info.Vector, info.PC&0xFFFFFF, info.NewPC&0xFFFFFF, info.Opcode, info.FaultAddress&0xFFFFFF, info.SR, info.NewSR) @@ -125,23 +125,23 @@ func (m *Machine) enableBootTrace(verbose bool) { } fmt.Fprintf(m.traceWriter, "exception vector=%d pc=%06x newpc=%06x opcode=%04x sr=%04x newsr=%04x\n", info.Vector, info.PC&0xFFFFFF, info.NewPC&0xFFFFFF, info.Opcode, info.SR, info.NewSR) - }) + } if verbose { logger := cpu.NewVerboseLogger(m.cpu, m.bus, m.traceWriter, cpu.VerboseLoggerOptions{ IncludeRegisters: true, IncludeCycles: true, }) - m.cpu.SetTracer(func(info cpu.TraceInfo) { + hooks.Trace = func(info cpu.TraceInfo) { pc := info.PC & 0xFFFFFF if m.tracePCInRange(pc) { logger.Trace(info) } - }) - return + } + return hooks } - m.cpu.SetTracer(func(info cpu.TraceInfo) { + hooks.Trace = func(info cpu.TraceInfo) { pc := info.PC & 0xFFFFFF if !m.tracePCInRange(pc) { return @@ -160,7 +160,8 @@ func (m *Machine) enableBootTrace(verbose bool) { info.Registers.A[7], m.decodeTraceInstruction(info), ) - }) + } + return hooks } func (m *Machine) tracePCInRange(pc uint32) bool {