diff --git a/CHANGELOG.md b/CHANGELOG.md index 5ec6d8eba..6d5e6ce68 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,48 @@ version at the top when it merges. ## Unreleased +- Every figure below is from a tree with nothing uncommitted, and that matters + here more than usual: the build stamps `-dirty` into the version hash inside + every image whenever `git status` is not empty, and it lands unevenly -- 8 + bytes on the rp0 arm against 4 on the rp3 arm in the default drawer, so a + dirty tree moves the delta. Commit or stash before measuring a size here. +- `-mregparm=3` is what every drawer builds with, so our own calls pass their + arguments in registers. `bsdsocket.library` 353,404 -> 329,464 loaded bytes, + 370,880 -> 346,608 file, 343,420 -> 319,480 of code; `anxnet.device` 43,584 -> + 40,144 loaded, 45,704 -> 42,228 file; the resident pair -27,380. +- Default drawer, 137 images: 7,917,004 -> 7,707,384 loaded (-209,620, -2.65%), + 6,772,936 -> 6,560,528 file (-212,408, -3.14%). Minimal -93,460 loaded, micro + -87,364. Loaded is CODE + DATA + BSS, what `LoadSeg` needs room for. Measured + against the same tree built with `-DAMINETXDUO_REGPARM=0`. +- Five test images grew 4-20 bytes: `tests/perf/chipscreen` +20, + `tests/perf/n68kmv` +16, `tests/tools/AamProbe`, `ResolveBreak` and `PtrProbe` + +4 each. No shipping image grew, and no image is in one arm only. +- No LVO changed: every one is entered through a register the NDK headers pin. + 207 foreign direct calls in `bsdsocket.library` at regparm 0, 201 at regparm 3, + and the only sites pushing in neither arm are libgcc's internal `jsr`. +- The boundaries the compiler does not build keep the stack convention, pinned + where declared: `main()` (`crt0.o` and `src/tools/tool_startup.S` push `argv` + then `argc`; `include/aminetxduo/asm_main.h`), the assembly in `src/net68k` and + `src/crypto68k` reading `4(sp)`, `8(sp)`, `12(sp)` (`asm_abi.h`), `libc.a`, + `amiga.lib`, and the OS entry points whose registers the headers pin by name. +- Gate: `-Wmissing-prototypes` with `-Werror=implicit-function-declaration`, and + an `#error` in `include/aminetxduo/asm_main.h` unless the m68k build is C23 or + later -- where `()` means `(void)` and an unprototyped `f(a, b)` is a hard + error. `-Wstrict-prototypes` was removed: it cannot fire on the m68k arm, and + the three host sites it does fire on are the NDK's and the vendored tree's. +- `src/net68k/n68k_memcpy_hook.c` includes ``, so its stack convention + is stated by a declaration and not left to GCC's recognition of the name + `memcpy`. Two clean trees differing only in that include leave both shipping + images byte-identical but for the 7 hash bytes the version string carries. +- `tests/fuzz/CMakeLists.txt` puts `include/` on `fuzz_tls_crypto`'s path for + `src/crypto68k/crypto68k.h`'s ``. That target exists + only where `sizeof(void*) == 4`, so CI was the first to build it: it found the + gap in host32 and sanitize32, not on any host or cross arm. +- Host tier: 434 targets, 0 warnings, ctest 493/493. 32-bit host tier: 24/24 and + within its duration budget. Emulator tier: green on both arms. +- Build consequence: an image must come from one configuration. A `build/` + directory made before this change needs its objects rebuilt, not relinked. + - `CheckNetConfig` names the file a default-gateway finding was read from even when that file is an interface file. `load_gateway()` falls back to a `GATEWAY=` in the first interface file when neither diff --git a/CMakePresets.json b/CMakePresets.json index 7360a0497..f6350ff53 100644 --- a/CMakePresets.json +++ b/CMakePresets.json @@ -13,7 +13,8 @@ "toolchainFile": "${sourceDir}/cmake/toolchain-m68k-amigaos.cmake", "binaryDir": "${sourceDir}/build/${presetName}", "cacheVariables": { - "CMAKE_BUILD_TYPE": "Release" + "CMAKE_BUILD_TYPE": "Release", + "CMAKE_PROJECT_INCLUDE": "${sourceDir}/cmake/ci-warnings.cmake" } }, { diff --git a/cmake/ci-warnings.cmake b/cmake/ci-warnings.cmake index 4d9d754ee..247bee64b 100644 --- a/cmake/ci-warnings.cmake +++ b/cmake/ci-warnings.cmake @@ -1,92 +1,97 @@ # Turn compiler warnings into build failures, for OUR sources only. # -# Usage (no edit to any CMakeLists.txt is needed; CMake includes this file at -# the end of the top-level project() call): +# Included at the end of the top-level project() call. Every preset sets +# CMAKE_PROJECT_INCLUDE to this file (CMakePresets.json, hidden `amiga` base), +# so there is no way to configure one of this project's own drawers without the +# gate. tools/ci.sh passes the same option explicitly; that is redundant and +# kept only because the ci.sh host arm configures a directory no preset +# describes. # -# cmake -S . -B build -DCMAKE_PROJECT_INCLUDE=cmake/ci-warnings.cmake +# Override with -DAMINETXDUO_WARNING_FLAGS="-Wall;-Wextra", turn the whole +# thing off with -DAMINETXDUO_WERROR=OFF. # -# Override the flag set with -DAMINETXDUO_WARNING_FLAGS="-Wall;-Wextra", and -# turn the whole thing off again with -DAMINETXDUO_WERROR=OFF. -# -# WHY IT IS NOT JUST add_compile_options(-Wall -Wextra -Werror) -# -# Most of what this project compiles is not this project: ThreadX, NetX Duo, -# nx_crypto and nx_secure are vendored verbatim as submodules and are not -# warning-clean under -Wextra. A global flag would therefore fail the build -# on code we have a standing rule never to modify. -# -# Filtering by target does not work either, `threadx`, `netxduo`, -# `netxduo_addons` and `crypto68k_ref` are declared in OUR CMakeLists.txt -# files but compile vendored sources. So the filter is per SOURCE FILE: -# every source whose path contains /third_party/ is left alone, everything -# else gets the flags. Nothing has to be kept in a list, so a new component -# is covered the day it is added. +# Filtering is per SOURCE FILE, not per target: most of what this project +# compiles is not this project (the ThreadX, NetX Duo, nx_crypto and nx_secure +# submodules are vendored verbatim and are not warning-clean under -Wextra), +# while targets like `threadx` and `netxduo` are declared in OUR CMakeLists.txt +# and compile vendored sources. Paths containing /third_party/ are left alone; +# nothing has to be kept in a list. # # The work is deferred to the end of the top-level directory because targets do -# not exist yet at the point this file is included, add_subdirectory() has -# not run. set_property(SOURCE ... TARGET_DIRECTORY ...) is what makes it -# legal to reach into a target declared in another directory. +# not exist yet when this file is included. set_property(SOURCE ... +# TARGET_DIRECTORY ...) is what makes reaching into another directory legal. # # SPDX-License-Identifier: MIT option(AMINETXDUO_WERROR "Fail the build on any warning in our own sources" ON) -# -Wmissing-prototypes is here so a function that is not static has to say what -# it is somewhere a caller can see. It found 34: eight that were only ever used -# in their own file and are static now, one dead accessor, two asm-facing entry -# points nothing declared, a C fallback a macro renamed out from under its own -# prototype, and a handful of files that simply did not include the header -# already declaring what they defined -- config_advice.c defined ami_cfg_advice() -# without including the header that tells callers its shape. +# -Wmissing-prototypes: a function that is not static has to say what it is +# where a caller can see it. Found 34. Not a size lever -- single-unit LTO +# already saw everything, so static unlocked nothing the linker did not have; +# what it buys is the compiler checking every definition against what callers +# were told. # -# It is not a size lever. Measured across the whole change: bsdsocket.library -# 339,468 -> 339,476 bytes, +8. Single-unit LTO already saw everything, so -# static unlocked no internalization the linker did not have. What it buys is -# the compiler checking every definition against what callers were told. -set(AMINETXDUO_WARNING_FLAGS "-Wall;-Wextra" CACHE STRING - "Warning flags applied to sources outside third_party/") - -# SHIPPING SOURCES ONLY -- src/ and port/, not tests/. +# -Wstrict-prototypes is NOT here, deliberately, and what replaced it is a hard +# error in the language rather than a warning here. # -# -Wmissing-prototypes says a function that is not static has to declare itself -# somewhere a caller can see. That is exactly right for code that ships, and -# it found real things: a definition whose header was never included -# (config_advice.c defined ami_cfg_advice() without including config.h, so -# nothing checked it against what every caller is told), a C fallback a macro -# had renamed out from under its own prototype, and eight functions never used -# outside their own file. +# It was added for the mixing hazard -mregparm introduces: `VOID f();' declares +# no prototype, so `f(a, b)' travels on the stack while a definition compiled +# for -mregparm=3 reads registers, and the callee returns a wrong number. Two +# things make the flag the wrong tool. It cannot fire where the hazard exists: +# the m68k compiler is GCC 16 and compiles as C23 (__STDC_VERSION__ 202311L with +# no -std given), where `()' MEANS `(void)' and `f(a, b)' is "too many arguments +# to function" -- a hard error for a direct call and for a call through a +# function pointer alike. And where it does fire it cannot be satisfied: a full +# host build (gcc 14, C17) reports 26 diagnostics from just three sites, none of +# them ours to change -- the NDK's own `VOID (*putChProc)()' in exec_protos.h, +# which forces the cast at src/bsdsocket/loghook.c:163; the NDK-impersonating +# src/netdev/test/shim/exec/interrupts.h; and the vendored +# third_party/netxduo/nx_secure/inc/nx_secure_tls_api.h. # -# It is NOT right for tests. tests/fuzz/fuzz_dns.c DEFINES -# _tx_thread_system_suspend() to stand in for ThreadX's, and there is no header -# it could be declared in that would not be a forgery of upstream's. A test -# that impersonates a symbol on purpose is not the bug this flag looks for, and -# the findings there are per-configuration noise: each CI arm compiles a -# different set of stubs. -set(AMINETXDUO_SHIPPING_WARNING_FLAGS "-Wmissing-prototypes" CACHE STRING - "Warning flags applied to src/ and port/ only") - -# Per-file escapes, as pairs. Every entry is a -# bug someone has to fix, so each one says what it is; this list should shrink. +# The guarantee is therefore asserted where it is relied on: +# include/aminetxduo/asm_main.h #errors unless the m68k build is C23 or later. +# That covers a hand-written compile line too, which no flag here would. # -# IT IS EMPTY, and the last entry to leave is worth recording, because it did -# not leave for the reason its own note gave. +# -Werror=implicit-function-declaration: C23 makes it an error anyway; naming it +# stops a future -std from taking that away quietly. # -# src/config/test/test_config.c held -Wno-error=address for CHECK_STR's -# `(got) ? (got) : "(null)"` printf argument, which is -Waddress when `got` is -# an array. That form was replaced by an or_null() helper at some point after -# the escape was written, and nobody took the escape back out: measured on gcc -# 14.2 with this file's own flags, the ternary form gives 7 -Werror=address and -# the or_null form gives none. SO THE ESCAPE HAD BEEN DEAD, and an escape that -# is dead is worse than one that is needed -- it is a hole nothing is watching, -# ready for the next real warning in that file to fall through silently. +# MEASURED 2026-09-30, full host build with -Werror on (gcc 14.2, C17): +# 0 -Wmissing-prototypes from shipping sources, 0 -Wcast-function-type, 0 +# implicit-function-declaration. Everything -Wmissing-prototypes finds is a +# test, which is why it is off for them below. +set(AMINETXDUO_WARNING_FLAGS + "-Wall;-Wextra;-Werror=implicit-function-declaration;-Wmissing-prototypes" + CACHE STRING "Warning flags applied to sources outside third_party/") + +# NON-SHIPPING -- anything not under src/ or port/, plus the host tests under +# src//test/. # -# CHECK_STR is a function now, so the null guard is expressed once instead of -# at 132 expansion sites. That does not make the guard fire for array callers; -# nothing can, an array is not null. It means the compiler is not asked to -# prove the same tautology 132 times, which is the thing that needed an escape. +# -Wmissing-prototypes is right for code that ships and wrong for tests. A test +# that impersonates a symbol on purpose (tests/fuzz/fuzz_dns.c DEFINES +# _tx_thread_system_suspend()) is not the bug the warning looks for, and the +# findings are per-configuration noise: each CI arm compiles a different set of +# stubs. One rule, not an entry per file. # -# Adding an entry here is fine. Leaving a dead one is not: check that the -# warning still fires before assuming an entry is load-bearing. +# -Wno-cast-function-type used to be here, for the five tests that install a Hook +# by casting one -- the NDK declares `h_Entry' as ULONG (*)(VOID) while a Hook +# function is really ULONG (*)(struct Hook *, APTR, APTR). It is gone because +# it has no subject left: main replaced all five casts with union punning +# (4607a0c6, test_netmon_host.c and its siblings now assign through a `HookEntry' +# union), no shipping source ever cast h_Entry -- src/ only READS it -- and a +# full host build with the escape removed reports 0 -Wcast-function-type. An +# escape nothing can trip is a hole nothing is watching; see the note on +# AMINETXDUO_WARNING_EXEMPT below. +set(AMINETXDUO_NONSHIPPING_WARNING_FLAGS + "-Wno-missing-prototypes" + CACHE STRING "Warning flags applied to sources that do not ship") + +# Per-file escapes, as pairs. Each entry is a bug +# somebody has to fix, so it says what it is; this list should shrink. IT IS +# EMPTY: the last entry, -Wno-error=address for src/config/test/test_config.c, +# had been dead -- the ternary it was written for was replaced by an or_null() +# helper and nobody took the escape out. A dead escape is worse than a needed +# one, a hole nothing is watching. Check a warning still fires before assuming +# an entry is load-bearing. set(AMINETXDUO_WARNING_EXEMPT) function(_aminetxduo_warnings_apply_dir dir) @@ -115,12 +120,13 @@ function(_aminetxduo_warnings_apply_dir dir) set(_ours "") set(_shipping "") + set(_nonshipping "") foreach(_s IN LISTS _srcs) if(NOT IS_ABSOLUTE "${_s}") set(_s "${_sdir}/${_s}") endif() - # Generator expressions and generated sources are skipped: they are - # not ours to police and cannot be path-matched reliably. + # Generator expressions and generated sources cannot be + # path-matched reliably, so they are not policed. if(_s MATCHES "\\$<") continue() endif() @@ -128,12 +134,15 @@ function(_aminetxduo_warnings_apply_dir dir) continue() endif() # tests/atf/ holds FreeBSD's tests/sys/netinet sources byte for - # byte, under their own BSD-2-Clause header, so the same rule + # byte under their own BSD-2-Clause header, so the same rule # applies: not ours, not modifiable, not warning-clean against the # Roadshow NDK's prototypes (sendto() takes APTR where FreeBSD's - # takes const void *). The shim beside them -- atf-c.h, - # atf_main.c, atf-prelude.h -- is ours and is not exempt. - if(_s MATCHES "/tests/atf/" AND NOT _s MATCHES "/tests/atf/atf") + # takes const void *). That is tcp_socket.c and whatever lands + # beside it, so the exemption names the files that are OURS + # instead of guessing from a filename prefix. The shim is atf-c.h, + # atf-prelude.h and atf_main.c, and it is not exempt. + if(_s MATCHES "/tests/atf/" + AND NOT _s MATCHES "/tests/atf/(atf-c\\.h|atf-prelude\\.h|atf_main\\.c)$") continue() endif() list(APPEND _ours "${_s}") @@ -144,21 +153,25 @@ function(_aminetxduo_warnings_apply_dir dir) if((_s MATCHES "/src/" OR _s MATCHES "/port/") AND NOT _s MATCHES "/test/") list(APPEND _shipping "${_s}") + else() + list(APPEND _nonshipping "${_s}") endif() endforeach() if(_ours) # APPEND, not set_source_files_properties(): src/crypto68k already - # puts COMPILE_OPTIONS on its two .S files and overwriting them + # puts COMPILE_OPTIONS on its two .S files and overwriting those # would drop the -m68020 the assembler needs. set_property(SOURCE ${_ours} TARGET_DIRECTORY ${_t} APPEND PROPERTY COMPILE_OPTIONS ${_flags}) - # The shipping-only half: src/ and port/, never tests/. - if(_shipping AND AMINETXDUO_SHIPPING_WARNING_FLAGS) - set_property(SOURCE ${_shipping} TARGET_DIRECTORY ${_t} + # Everything that does not ship gets the two flags that only make + # sense for shipping code turned back off. Set after the main list + # so the -Wno- wins. + if(_nonshipping AND AMINETXDUO_NONSHIPPING_WARNING_FLAGS) + set_property(SOURCE ${_nonshipping} TARGET_DIRECTORY ${_t} APPEND PROPERTY COMPILE_OPTIONS - ${AMINETXDUO_SHIPPING_WARNING_FLAGS}) + ${AMINETXDUO_NONSHIPPING_WARNING_FLAGS}) endif() # ... then the escapes, which have to come after to win. diff --git a/cmake/toolchain-m68k-amigaos.cmake b/cmake/toolchain-m68k-amigaos.cmake index fdf3c7d88..33bdaa56e 100644 --- a/cmake/toolchain-m68k-amigaos.cmake +++ b/cmake/toolchain-m68k-amigaos.cmake @@ -345,7 +345,103 @@ endif() string(REPLACE ";" " " AMIGA_ARCH_FLAGS_STR "${AMIGA_ARCH_FLAGS}") -set(CMAKE_C_FLAGS_INIT "${AMIGA_ARCH_FLAGS_STR} -fomit-frame-pointer -fno-strict-aliasing") +# ------------------------------------------------------- argument passing -- +# +# Arguments used to travel on the stack: a push per argument at every call site +# and a read at the callee, six to twelve bytes each, thousands of times. +# +# -mregparm=N is PER CLASS, not one list: the first N integer arguments go to +# d0..d(N-1) and the first N pointer arguments to a0..a(N-1), so N=3 is six +# slots, and a class that runs out spills into what the other class left. +# Measured with a probe calling ext(a,b,c,d,e,f): six ints give d0,d1,d2,a0,a1,a2, +# six pointers give a0,a1,a2,d0,d1,d2, (int,char*,int,char*) gives d0,a0,d1,a1. +# "d0/d1/d2" is the wrong shorthand -- a pointer can arrive in d0 and an int in +# a0, and a callee that assumes otherwise returns a wrong number, not a crash. +# That is why every such callee is pinned rather than taught the register set. +# +# Measured against the same tree built with AMINETXDUO_REGPARM=0, the flag the +# only difference (m68k-amigaos-gcc 16.2.0b, -flto), in LOADED bytes -- CODE + +# DATA + BSS, what LoadSeg has to find room for -- and in file bytes. The two +# images a machine keeps resident, default drawer: +# +# bsdsocket.library 353,412 -> 329,468 loaded 370,888 -> 346,612 file +# anxnet.device 43,592 -> 40,148 loaded 45,712 -> 42,232 file +# +# so the pair costs 27,388 bytes less loaded (23,944 + 3,444). bsdsocket.library +# loses 23,944 loaded bytes in default (343,428 -> 319,484 of code, DATA and BSS +# unchanged), 15,888 in minimal (231,140 -> 215,252) and 13,356 in micro +# (198,248 -> 184,892). Over every image BOTH arms link: 137 images 7,906,460 +# -> 7,696,820 loaded (-209,640, -2.65%), 123 minimal -93,460 (-2.12%), 122 +# micro -87,356 (-2.06%). Four test images grew, by 4 to 16 bytes +# (tests/perf/n68kmv +16, tests/perf/chipscreen +16, tests/tools/PtrProbe +4, +# ResolveBreak +4); nothing that ships did, and no image is present in one arm +# only. +# +# SAFE ONLY WHERE THE CONVENTION IS PINNED at both ends, which it is at every +# boundary this project has but one: +# +# - library and device entry points (bsdsocket_vectors.h, library.c, the +# netdev entries) pin with `__asm("d0")` / `__asm("a6")`, which GCC honours +# whatever -mregparm says; +# - user-supplied hooks (loghook.c, errno.c, netmonitor.c) pin a0/a1/a2, so a +# caller compiled the ordinary way is still called correctly; +# - the hand-written routines in src/net68k and src/crypto68k read the stack, +# and every C declaration of one carries AMIGA_ASM_ARGS +# (__attribute__((__stkparm__))); see include/aminetxduo/asm_abi.h. +# +# The one exception is ami_rt_cpu_select(), deliberately: it follows whatever +# convention the build uses so that src/common/ami_udivdi3.c and its callers +# stay header-free. Safe because its one non-C caller covers both at once -- +# tool_startup.S .Lrtgo loads the flags into d0/d1 AND pushes the same +# registers. It is the only such boundary. +# +# main() IS THE ONE THAT GOT AWAY, and -include asm_main.h below is its pin. +# Nothing in this tree calls it: crt0.o and tool_startup.S both push argv then +# argc and `jsr _main`, neither is ours to edit or compiled with these flags, so +# a C main() under -mregparm=3 reads argc out of d0 instead. Measured on the +# first build that carried the option: iperf's _main at 0x207e was `tst.l d0`; +# it is `tst.l 8(a5)` now, same address. A command that reads argc == 0 thinks +# Workbench launched it. The pin is a force-included header rather than +# -Dmain=... because the -D reaches /bin/sh with unquoted parentheses and the +# configure dies before gcc runs; see the header for the rest. It is gated on +# the same option, so AMINETXDUO_REGPARM=0 reproduces the old convention +# exactly, and 0 is a supported build. +set(AMINETXDUO_REGPARM "3" CACHE STRING + "Integer arguments passed in registers d0-d2 (0 disables, as before)") +set_property(CACHE AMINETXDUO_REGPARM PROPERTY STRINGS 0 1 2 3) + +if(AMINETXDUO_REGPARM GREATER 0) + get_filename_component(_amiga_top "${CMAKE_CURRENT_LIST_DIR}/.." ABSOLUTE) + set(_amiga_regparm_flags + "-mregparm=${AMINETXDUO_REGPARM} -include ${_amiga_top}/include/aminetxduo/asm_main.h") +else() + set(_amiga_regparm_flags "") +endif() + +set(CMAKE_C_FLAGS_INIT + "${AMIGA_ARCH_FLAGS_STR} -fomit-frame-pointer -fno-strict-aliasing ${_amiga_regparm_flags}") + +# -mregparm and the pin travel in CMAKE_C_FLAGS, and CMAKE_C_FLAGS_INIT is +# consulted only when the cache is created: -DAMINETXDUO_REGPARM=0 on an +# existing directory would leave -mregparm=3 on the command line while the +# cache said 0, so the tree would report one convention and build the other. +# Unlike the CPU guard above, this one repairs rather than refuses. Both flags +# reach the preprocessor and the C compiler and nothing else -- the assembler +# never sees them -- so rewriting CMAKE_C_FLAGS is the whole change. The first +# configure has no cache entry and is left to CMAKE_C_FLAGS_INIT, which is +# already right. Stripping the pin matters as much as stripping -mregparm: +# removing one and leaving the other is the same lie with the halves swapped. +if(DEFINED CACHE{CMAKE_C_FLAGS}) + set(_amiga_cf "${CMAKE_C_FLAGS}") + string(REGEX REPLACE " ?-mregparm=[0-9]+" "" _amiga_cf "${_amiga_cf}") + string(REGEX REPLACE " ?-include +[^ ]+asm_main\\.h" "" _amiga_cf "${_amiga_cf}") + string(STRIP "${_amiga_cf} ${_amiga_regparm_flags}" _amiga_cf) + if(NOT _amiga_cf STREQUAL CMAKE_C_FLAGS) + message(STATUS "regparm ${AMINETXDUO_REGPARM}: CMAKE_C_FLAGS is now '${_amiga_cf}'") + set(CMAKE_C_FLAGS "${_amiga_cf}" CACHE STRING "C compiler flags" FORCE) + endif() +endif() + # -Os, everywhere, and stated rather than inherited. CMake's Compiler/GNU # module APPENDS its own "-O3 -DNDEBUG" after CMAKE_C_FLAGS_RELEASE_INIT and # the last -O on the command line wins, so this line has to name the level it diff --git a/include/aminetxduo/asm_abi.h b/include/aminetxduo/asm_abi.h new file mode 100644 index 000000000..3d604fb96 --- /dev/null +++ b/include/aminetxduo/asm_abi.h @@ -0,0 +1,53 @@ +/* + * AmiNetXDuo, the one marker for a call boundary that lives in hand-written + * assembly. + * + * SPDX-License-Identifier: MIT + */ + +#ifndef AMINETXDUO_ASM_ABI_H +#define AMINETXDUO_ASM_ABI_H + +/* + * AMIGA_ASM_ARGS -- pin the amigaos convention, every argument on the stack, on + * a function the C compiler does not build. + * + * The routines in src/net68k and src/crypto68k read 4(sp), 8(sp), 12(sp). No + * compiler option reaches them, so the C side is what is held still: under + * -mregparm=3 the call would hand them d0, d1, d2 and leave the stack holding + * whatever was there before -- a wrong number, not a trap. + * + * AMIGA_ASM_ARGS ULONG n68k_sum_longwords(const ULONG *p, ULONG count); + * AMIGA_ASM_ARGS ULONG (*n68k_vec_sum)(const ULONG *, ULONG); + * + * It goes BEFORE the declarator, as the toolchain's own headers spell it + * (`__stdargs int memcmp(...)`), because a trailing attribute is rejected on a + * DEFINITION and some of these names have a C definition in configurations with + * no assembly. A pointer to a pinned function needs its own pin: it is a + * different type. + * + * regparm(0) is the natural spelling of "stack, please" and does nothing on this + * backend -- GCC 16.2.0b m68k ignores it; regparm(1..3) all take effect. A + * trailing ellipsis would also work, but it changes what the function IS. + * + * No-op off m68k, so a host test supplying its own C body sees the prototype it + * saw before. + */ + +#if defined(__m68k__) && (defined(__GNUC__) || defined(__clang__)) +# if defined(__has_attribute) && __has_attribute(__stkparm__) +# define AMIGA_ASM_ARGS __attribute__((__stkparm__)) +# else +/* + * Not "define it away": dropping the pin here would compile, link, and pass + * arguments in d0/d1/d2 to assembly that reads the stack, which is a wrong + * number rather than a crash. This toolchain spells the attribute + * `__stkparm__` (its own libc headers do), so say so and stop. + */ +# error "m68k compiler without __attribute__((__stkparm__)): the assembly call boundaries cannot be pinned" +# endif +#else +# define AMIGA_ASM_ARGS +#endif + +#endif /* AMINETXDUO_ASM_ABI_H */ diff --git a/include/aminetxduo/asm_main.h b/include/aminetxduo/asm_main.h new file mode 100644 index 000000000..f1b6be63b --- /dev/null +++ b/include/aminetxduo/asm_main.h @@ -0,0 +1,74 @@ +/* + * AmiNetXDuo, pin C main() to the convention the startup code calls it with. + * + * SPDX-License-Identifier: MIT + */ + +/* + * FORCE-INCLUDED BY cmake/toolchain-m68k-amigaos.cmake, only when -mregparm + * is in force. Nothing should include it. + * + * Nothing in this tree calls main(). The toolchain's crt0.o does, and so does + * src/tools/tool_startup.S: both push argv, then argc, then `jsr _main`. Under + * -mregparm=3 a C main() reads argc out of d0 instead -- whatever the startup + * left there -- so a command reads argc == 0 and believes Workbench launched + * it. 87 of the 130 main-bearing images here link crt0.o, which is not ours to + * edit, so the C side is pinned instead. + * + * The macro is FUNCTION-LIKE for one reason: an object-like + * `#define main __attribute__((__stkparm__)) main` also expands where main is + * an EXPRESSION, and 16 files here do that -- + * `ami_crash_set_reference((APTR)main, "main")`. A function-like macro needs a + * following `(` and so fires only on a parameter list. A CALL to main(...) is + * the one construct it does not tolerate; that is a compile error, not a silent + * miscompile. String literals and comments are single tokens and never + * matched. + * + * Not spelled -Dmain=... on the command line: CMake writes flags into the make + * recipe as raw text, so the parentheses reach /bin/sh unquoted and the + * compiler never runs. A path has no shell metacharacters. + * + * __stdargs is the same attribute -- the compiler predefines it as + * `__attribute__((__stkparm__))` -- and this uses the expansion because it + * needs no space. + */ + +#if defined(__m68k__) && (defined(__GNUC__) || defined(__clang__)) +# if defined(__has_attribute) && __has_attribute(__stkparm__) +# define main(...) __attribute__((__stkparm__)) main(__VA_ARGS__) +# else +# error "m68k compiler without __attribute__((__stkparm__)): main() cannot be pinned to the startup's convention" +# endif +#endif + +/* + * AND THE LANGUAGE HAS TO BE C23 OR LATER, which is the other half of the same + * pin and the reason no warning flag has to guard this convention. + * + * A declaration of the form `VOID f();' is a different thing in the two + * standards, and the difference is exactly the bug above. In C17 it declares + * f with NO prototype, so `f(a, b)' is accepted and the arguments travel on the + * stack -- a call the callee, compiled for -mregparm=3, reads out of registers. + * In C23 `()' MEANS `(void)': f is declared to take no arguments at all, and + * `f(a, b)' is "too many arguments to function", a hard error, for a direct + * call and for a call through a function pointer alike (probed with gcc 14 on + * `extern void f(); f(1,2);' and `extern void (*p)(); p(3,4);'). + * + * The m68k compiler defaults to C23 -- m68k-amigaos-gcc 16.2 reports + * __STDC_VERSION__ 202311L with no -std on the command line -- so an + * unprototyped call cannot be written in this build in the first place and + * there is nothing left for a warning to catch. That is why -Wstrict-prototypes + * is NOT in cmake/ci-warnings.cmake: it can never fire on this arm, and on the + * host arm (gcc 14, C17) it fires on the NDK's own callback spelling -- + * `VOID (*putChProc)()' in exec_protos.h, which forces the cast in + * src/bsdsocket/loghook.c:163 -- where nothing about it is ours to change. + * + * The default is the whole guarantee, so it is asserted rather than assumed: a + * build that pins an older -std, or a future compiler that drops C23, would + * reopen the hole silently and produce an image that mixes the two conventions. + */ +#if defined(__m68k__) && (defined(__GNUC__) || defined(__clang__)) +# if !defined(__STDC_VERSION__) || (__STDC_VERSION__ < 202311L) +# error "the m68k build must be C23 or later: `()' has to mean `(void)', or a call through an unprototyped declaration passes arguments on the stack while the definition reads them from registers" +# endif +#endif diff --git a/src/common/ami_udivdi3.c b/src/common/ami_udivdi3.c index 0a40f8293..59513bcb0 100644 --- a/src/common/ami_udivdi3.c +++ b/src/common/ami_udivdi3.c @@ -64,6 +64,14 @@ typedef long s32; static int ami_rt_020; static int ami_rt_mulul; +/* + * The one thing called here from assembly. src/tools/tool_startup.S .Lrtgo + * loads the two flags into d0/d1 and pushes them, and this follows the build's + * convention rather than being pinned to one -- so it needs no header, which + * is what keeps this file as free of them as it looks. The comment at that + * call site and the argument-passing note in cmake/toolchain-m68k-amigaos.cmake + * are the other two thirds of it. + */ void ami_rt_cpu_select(int have_68020, int have_mulul); void ami_rt_cpu_select(int have_68020, int have_mulul) { diff --git a/src/common/crashguard.c b/src/common/crashguard.c index b314c1722..99983bd52 100644 --- a/src/common/crashguard.c +++ b/src/common/crashguard.c @@ -9,6 +9,7 @@ */ #include "aminetxduo/crashguard.h" +#include "aminetxduo/asm_abi.h" #include "aminetxduo/compat.h" #include @@ -272,7 +273,9 @@ static volatile ULONG ami_alert_inflight; finish first, so the two never overlap (F-080). */ static volatile BOOL ami_alert_draining; -VOID ami_alert_report(ULONG num); +/* The trampoline's asm() pushes the alert number on the stack and jsr's + this, so it must read 4(sp) -- see aminetxduo/asm_abi.h. */ +AMIGA_ASM_ARGS VOID ami_alert_report(ULONG num); VOID ami_alert_trampoline(VOID); __asm__( @@ -409,8 +412,8 @@ static VOID ami_alert_flush(VOID) /* `used': the only caller is the `jsr _ami_alert_report' in the trampoline's asm() above. Not static, which is not protection -- a whole-program view is entitled to privatise and then drop it. */ -VOID ami_alert_report(ULONG num) __attribute__((used)); -VOID ami_alert_report(ULONG num) +AMIGA_ASM_ARGS VOID ami_alert_report(ULONG num) __attribute__((used)); +AMIGA_ASM_ARGS VOID ami_alert_report(ULONG num) { struct Task *task = SysBase->ThisTask; const char *name = "?"; diff --git a/src/crypto68k/c68k_25519.c b/src/crypto68k/c68k_25519.c index 147571786..35146b6a9 100644 --- a/src/crypto68k/c68k_25519.c +++ b/src/crypto68k/c68k_25519.c @@ -11,6 +11,7 @@ #include "c68k_25519.h" #include "c68k_variant.h" +#include "aminetxduo/asm_abi.h" /* , not exec/types.h: this file must compile on the build host. */ #include @@ -69,7 +70,7 @@ static void fe_fold(fe r, uint32_t c) } } -static void fe_add_c(fe r, const fe a, const fe b) +static AMIGA_ASM_ARGS void fe_add_c(fe r, const fe a, const fe b) { uint64_t t = 0; int i; @@ -82,7 +83,7 @@ static void fe_add_c(fe r, const fe a, const fe b) fe_fold(r, (uint32_t)t); } -static void fe_sub_c(fe r, const fe a, const fe b) +static AMIGA_ASM_ARGS void fe_sub_c(fe r, const fe a, const fe b) { uint64_t t = 0; int i; @@ -117,20 +118,20 @@ static void fe_sub_c(fe r, const fe a, const fe b) */ #if defined(C68K_MV) -extern void c68k_fe_add_asm_mulu(fe r, const fe a, const fe b); -extern void c68k_fe_sub_asm_mulu(fe r, const fe a, const fe b); -extern void c68k_fe_add_asm_mulw(fe r, const fe a, const fe b); -extern void c68k_fe_sub_asm_mulw(fe r, const fe a, const fe b); +extern AMIGA_ASM_ARGS void c68k_fe_add_asm_mulu(fe r, const fe a, const fe b); +extern AMIGA_ASM_ARGS void c68k_fe_sub_asm_mulu(fe r, const fe a, const fe b); +extern AMIGA_ASM_ARGS void c68k_fe_add_asm_mulw(fe r, const fe a, const fe b); +extern AMIGA_ASM_ARGS void c68k_fe_sub_asm_mulw(fe r, const fe a, const fe b); -void (*c68k_vec_fe_add)(fe, const fe, const fe) = fe_add_c; -void (*c68k_vec_fe_sub)(fe, const fe, const fe) = fe_sub_c; +AMIGA_ASM_ARGS void (*c68k_vec_fe_add)(fe, const fe, const fe) = fe_add_c; +AMIGA_ASM_ARGS void (*c68k_vec_fe_sub)(fe, const fe, const fe) = fe_sub_c; #define fe_add (*c68k_vec_fe_add) #define fe_sub (*c68k_vec_fe_sub) #elif defined(C68K_ASM_25519) || defined(C68K_ASM_MULW) -extern void c68k_fe_add_asm(fe r, const fe a, const fe b); -extern void c68k_fe_sub_asm(fe r, const fe a, const fe b); +extern AMIGA_ASM_ARGS void c68k_fe_add_asm(fe r, const fe a, const fe b); +extern AMIGA_ASM_ARGS void c68k_fe_sub_asm(fe r, const fe a, const fe b); #define fe_add c68k_fe_add_asm #define fe_sub c68k_fe_sub_asm #else @@ -164,7 +165,7 @@ void c68k_25519_fe_sub_ref(uint32_t r[8], const uint32_t a[8], * Operand scanning, 64 MULU.L, then the 2^256 = 38 fold. The accumulator * cannot overflow: (2^32-1)^2 + 2*(2^32-1) is exactly 2^64-1. */ -static void fe_mul_c(fe r, const fe a, const fe b) +static AMIGA_ASM_ARGS void fe_mul_c(fe r, const fe a, const fe b) { uint32_t t[16]; uint64_t v; @@ -202,12 +203,12 @@ static void fe_mul_c(fe r, const fe a, const fe b) * src/crypto68k/CMakeLists.txt refuses both at once. */ #if defined(C68K_MV) -extern void c68k_fe_mul_asm_mulu(fe r, const fe a, const fe b); -extern void c68k_fe_mul_asm_mulw(fe r, const fe a, const fe b); -void (*c68k_vec_fe_mul)(fe, const fe, const fe) = fe_mul_c; +extern AMIGA_ASM_ARGS void c68k_fe_mul_asm_mulu(fe r, const fe a, const fe b); +extern AMIGA_ASM_ARGS void c68k_fe_mul_asm_mulw(fe r, const fe a, const fe b); +AMIGA_ASM_ARGS void (*c68k_vec_fe_mul)(fe, const fe, const fe) = fe_mul_c; #define fe_mul (*c68k_vec_fe_mul) #elif defined(C68K_ASM_25519) || defined(C68K_ASM_MULW) -extern void c68k_fe_mul_asm(fe r, const fe a, const fe b); +extern AMIGA_ASM_ARGS void c68k_fe_mul_asm(fe r, const fe a, const fe b); #define fe_mul c68k_fe_mul_asm #else #define fe_mul fe_mul_c @@ -240,7 +241,7 @@ int c68k_25519_fe_mul_is_asm(void) * diagonal squares added. Checked against fe_mul(r,a,a) on random inputs -- * published vectors cannot catch a transcription error here. */ -static void fe_sqr_c(fe r, const fe a) +static AMIGA_ASM_ARGS void fe_sqr_c(fe r, const fe a) { uint32_t t[16]; uint64_t v; @@ -292,12 +293,12 @@ static void fe_sqr_c(fe r, const fe a) } #if defined(C68K_MV) -extern void c68k_fe_sqr_asm_mulu(fe r, const fe a); -extern void c68k_fe_sqr_asm_mulw(fe r, const fe a); -void (*c68k_vec_fe_sqr)(fe, const fe) = fe_sqr_c; +extern AMIGA_ASM_ARGS void c68k_fe_sqr_asm_mulu(fe r, const fe a); +extern AMIGA_ASM_ARGS void c68k_fe_sqr_asm_mulw(fe r, const fe a); +AMIGA_ASM_ARGS void (*c68k_vec_fe_sqr)(fe, const fe) = fe_sqr_c; #define fe_sqr (*c68k_vec_fe_sqr) #elif defined(C68K_ASM_25519) || defined(C68K_ASM_MULW) -extern void c68k_fe_sqr_asm(fe r, const fe a); +extern AMIGA_ASM_ARGS void c68k_fe_sqr_asm(fe r, const fe a); #define fe_sqr c68k_fe_sqr_asm #else #define fe_sqr fe_sqr_c diff --git a/src/crypto68k/c68k_aes.c b/src/crypto68k/c68k_aes.c index f0b7ea2c3..a5d824c9b 100644 --- a/src/crypto68k/c68k_aes.c +++ b/src/crypto68k/c68k_aes.c @@ -8,6 +8,7 @@ #include "c68k_aes.h" #include "c68k_variant.h" +#include "aminetxduo/asm_abi.h" /* ------------------------------------------------------------- variant --- */ @@ -726,10 +727,10 @@ UINT r; * switch and not two code paths. */ #ifdef C68K_ASM_AES -extern VOID c68k_aes_core_enc_t4_asm(const ULONG *rk, UINT nr, ULONG *st); -extern VOID c68k_aes_core_dec_t4_asm(const ULONG *rk, UINT nr, ULONG *st); -extern VOID c68k_aes_core_enc_t1_asm(const ULONG *rk, UINT nr, ULONG *st); -extern VOID c68k_aes_core_dec_t1_asm(const ULONG *rk, UINT nr, ULONG *st); +extern AMIGA_ASM_ARGS VOID c68k_aes_core_enc_t4_asm(const ULONG *rk, UINT nr, ULONG *st); +extern AMIGA_ASM_ARGS VOID c68k_aes_core_dec_t4_asm(const ULONG *rk, UINT nr, ULONG *st); +extern AMIGA_ASM_ARGS VOID c68k_aes_core_enc_t1_asm(const ULONG *rk, UINT nr, ULONG *st); +extern AMIGA_ASM_ARGS VOID c68k_aes_core_dec_t1_asm(const ULONG *rk, UINT nr, ULONG *st); #endif static VOID c68k_aes_enc_dispatch(const ULONG *rk, UINT nr, ULONG *st) diff --git a/src/crypto68k/c68k_chacha20.c b/src/crypto68k/c68k_chacha20.c index bfcb0ffdd..93505863b 100644 --- a/src/crypto68k/c68k_chacha20.c +++ b/src/crypto68k/c68k_chacha20.c @@ -14,6 +14,7 @@ #include "c68k_chacha20.h" #include "c68k_variant.h" +#include "aminetxduo/asm_abi.h" /* "expand 32-byte k", the four constant words, little-endian. */ @@ -108,7 +109,7 @@ UINT i; * AMINETXDUO_CRYPTO68K_ASM=OFF takes the C, and so do the 68000 and the 68060. */ #ifdef C68K_ASM_CHACHA20 -extern VOID c68k_chacha20_core_asm(const ULONG *in, ULONG *out); +extern AMIGA_ASM_ARGS VOID c68k_chacha20_core_asm(const ULONG *in, ULONG *out); #define C68K_CHACHA20_CORE c68k_chacha20_core_asm #else #define C68K_CHACHA20_CORE c68k_chacha20_core_c @@ -177,7 +178,7 @@ ULONG k; /* The same seven instructions a word, hand-written: see c68k_chacha20.S for what -Os made of the loop above. Same #ifndef/#ifdef structure as the limb primitives in c68k_prim.c. */ -extern VOID c68k_chacha20_xor_block_asm(const ULONG *ks, const UCHAR *in, +extern AMIGA_ASM_ARGS VOID c68k_chacha20_xor_block_asm(const ULONG *ks, const UCHAR *in, UCHAR *out); #define c68k_chacha20_xor_block c68k_chacha20_xor_block_asm #endif /* C68K_ASM */ diff --git a/src/crypto68k/c68k_cpu.c b/src/crypto68k/c68k_cpu.c index 4ab58e124..02f439a98 100644 --- a/src/crypto68k/c68k_cpu.c +++ b/src/crypto68k/c68k_cpu.c @@ -16,17 +16,17 @@ #include -extern c68k_limb c68k_addmul_1_mulu(c68k_limb *r, const c68k_limb *b, UINT n, +extern AMIGA_ASM_ARGS c68k_limb c68k_addmul_1_mulu(c68k_limb *r, const c68k_limb *b, UINT n, c68k_limb a); -extern c68k_limb c68k_addmul_1_mulw(c68k_limb *r, const c68k_limb *b, UINT n, +extern AMIGA_ASM_ARGS c68k_limb c68k_addmul_1_mulw(c68k_limb *r, const c68k_limb *b, UINT n, c68k_limb a); -extern c68k_limb c68k_div_2by1_mulu(c68k_limb hi, c68k_limb lo, c68k_limb d, +extern AMIGA_ASM_ARGS c68k_limb c68k_div_2by1_mulu(c68k_limb hi, c68k_limb lo, c68k_limb d, c68k_limb *rem); -extern c68k_limb c68k_div_2by1_c(c68k_limb hi, c68k_limb lo, c68k_limb d, +extern AMIGA_ASM_ARGS c68k_limb c68k_div_2by1_c(c68k_limb hi, c68k_limb lo, c68k_limb d, c68k_limb *rem); -extern VOID c68k_poly1305_blocks_asm(C68K_POLY1305 *ctx, const UCHAR *m, +extern AMIGA_ASM_ARGS VOID c68k_poly1305_blocks_asm(C68K_POLY1305 *ctx, const UCHAR *m, ULONG blocks, ULONG hibit); /* The four X25519 field routines are set where their portable twins are, @@ -42,11 +42,11 @@ extern VOID c68k_p256_cpu_select(UINT wide); * linked by programs that never select -- a bench, a test, a Shell command -- * and the answer they get has to be correct. */ -c68k_limb (*c68k_vec_addmul_1)(c68k_limb *, const c68k_limb *, UINT, +AMIGA_ASM_ARGS c68k_limb (*c68k_vec_addmul_1)(c68k_limb *, const c68k_limb *, UINT, c68k_limb) = c68k_addmul_1_c; -c68k_limb (*c68k_vec_div_2by1)(c68k_limb, c68k_limb, c68k_limb, +AMIGA_ASM_ARGS c68k_limb (*c68k_vec_div_2by1)(c68k_limb, c68k_limb, c68k_limb, c68k_limb *) = c68k_div_2by1_c; -VOID (*c68k_vec_poly1305_blocks)(C68K_POLY1305 *, const UCHAR *, ULONG, +AMIGA_ASM_ARGS VOID (*c68k_vec_poly1305_blocks)(C68K_POLY1305 *, const UCHAR *, ULONG, ULONG) = c68k_poly1305_blocks_c; static UINT c68k_selected = C68K_ASM_NONE; diff --git a/src/crypto68k/c68k_p256.c b/src/crypto68k/c68k_p256.c index 3bd093ed7..47230508f 100644 --- a/src/crypto68k/c68k_p256.c +++ b/src/crypto68k/c68k_p256.c @@ -14,6 +14,7 @@ #include "c68k_variant.h" #include "nx_crypto_huge_number.h" +#include "aminetxduo/asm_abi.h" /* @@ -42,24 +43,24 @@ static const c68k_limb c68k_p256_one[C68K_P256_LIMBS] = */ #ifdef C68K_MV -extern c68k_limb c68k_p256_add_raw_mv0(c68k_limb *r, const c68k_limb *a, +extern AMIGA_ASM_ARGS c68k_limb c68k_p256_add_raw_mv0(c68k_limb *r, const c68k_limb *a, const c68k_limb *b); -extern c68k_limb c68k_p256_add_raw_mv20(c68k_limb *r, const c68k_limb *a, +extern AMIGA_ASM_ARGS c68k_limb c68k_p256_add_raw_mv20(c68k_limb *r, const c68k_limb *a, const c68k_limb *b); -extern c68k_limb c68k_p256_sub_raw_mv0(c68k_limb *r, const c68k_limb *a, +extern AMIGA_ASM_ARGS c68k_limb c68k_p256_sub_raw_mv0(c68k_limb *r, const c68k_limb *a, const c68k_limb *b); -extern c68k_limb c68k_p256_sub_raw_mv20(c68k_limb *r, const c68k_limb *a, +extern AMIGA_ASM_ARGS c68k_limb c68k_p256_sub_raw_mv20(c68k_limb *r, const c68k_limb *a, const c68k_limb *b); -extern INT c68k_p256_reduce_core_mv0(c68k_limb *r, const c68k_limb *t); -extern INT c68k_p256_reduce_core_mv20(c68k_limb *r, const c68k_limb *t); +extern AMIGA_ASM_ARGS INT c68k_p256_reduce_core_mv0(c68k_limb *r, const c68k_limb *t); +extern AMIGA_ASM_ARGS INT c68k_p256_reduce_core_mv20(c68k_limb *r, const c68k_limb *t); -static c68k_limb (*c68k_vec_p256_add_raw)(c68k_limb *, const c68k_limb *, +static AMIGA_ASM_ARGS c68k_limb (*c68k_vec_p256_add_raw)(c68k_limb *, const c68k_limb *, const c68k_limb *) = c68k_p256_add_raw_mv0; -static c68k_limb (*c68k_vec_p256_sub_raw)(c68k_limb *, const c68k_limb *, +static AMIGA_ASM_ARGS c68k_limb (*c68k_vec_p256_sub_raw)(c68k_limb *, const c68k_limb *, const c68k_limb *) = c68k_p256_sub_raw_mv0; -static INT (*c68k_vec_p256_reduce_core)(c68k_limb *, const c68k_limb *) = +static AMIGA_ASM_ARGS INT (*c68k_vec_p256_reduce_core)(c68k_limb *, const c68k_limb *) = c68k_p256_reduce_core_mv0; #define C68K_P256_ADD_RAW (*c68k_vec_p256_add_raw) diff --git a/src/crypto68k/c68k_poly1305.c b/src/crypto68k/c68k_poly1305.c index 9a9b1369b..5555e9c2b 100644 --- a/src/crypto68k/c68k_poly1305.c +++ b/src/crypto68k/c68k_poly1305.c @@ -16,6 +16,7 @@ #include "c68k_variant.h" #include "c68k_poly1305.h" +#include "aminetxduo/asm_abi.h" /* A little-endian longword at an arbitrary address. @@ -101,7 +102,7 @@ ULONG t0, t1, t2, t3; * terms that fall off the top of the 130-bit accumulator, folded into the * multiplier instead of into a separate pass. */ -VOID c68k_poly1305_blocks_c(C68K_POLY1305 *ctx, const UCHAR *m, +AMIGA_ASM_ARGS VOID c68k_poly1305_blocks_c(C68K_POLY1305 *ctx, const UCHAR *m, ULONG blocks, ULONG hibit) { @@ -187,7 +188,7 @@ ULONG64 d0, d1, d2, d3, d4; * so do the 68000 and the 68060. */ #ifdef C68K_ASM_POLY1305 -extern VOID c68k_poly1305_blocks_asm(C68K_POLY1305 *ctx, const UCHAR *m, +extern AMIGA_ASM_ARGS VOID c68k_poly1305_blocks_asm(C68K_POLY1305 *ctx, const UCHAR *m, ULONG blocks, ULONG hibit); /* * One binary for every CPU takes the vector instead: the inner loop is a diff --git a/src/crypto68k/c68k_poly1305.h b/src/crypto68k/c68k_poly1305.h index b4f4d59dc..57a4fbcdc 100644 --- a/src/crypto68k/c68k_poly1305.h +++ b/src/crypto68k/c68k_poly1305.h @@ -11,6 +11,7 @@ #define AMINETXDUO_C68K_POLY1305_H #include "nx_crypto.h" +#include "aminetxduo/asm_abi.h" #ifdef __cplusplus extern "C" { @@ -68,7 +69,7 @@ VOID c68k_poly1305_finish(C68K_POLY1305 *ctx, UCHAR *tag); VOID c68k_poly1305_blocks(C68K_POLY1305 *ctx, const UCHAR *m, ULONG blocks, ULONG hibit); -VOID c68k_poly1305_blocks_c(C68K_POLY1305 *ctx, const UCHAR *m, ULONG blocks, +AMIGA_ASM_ARGS VOID c68k_poly1305_blocks_c(C68K_POLY1305 *ctx, const UCHAR *m, ULONG blocks, ULONG hibit); /* NX_CRYPTO_TRUE when the block function of this build is the assembly. In an diff --git a/src/crypto68k/c68k_prim.c b/src/crypto68k/c68k_prim.c index 59ea550b2..9d110927e 100644 --- a/src/crypto68k/c68k_prim.c +++ b/src/crypto68k/c68k_prim.c @@ -199,8 +199,8 @@ c68k_limb old; /* The macro above renames the C implementation AFTER crypto68k.h declared the plain name, so the definition below is a different symbol with no prototype. Declare it through the same macro, so one line covers both spellings. */ -c68k_limb c68k_div_2by1(c68k_limb hi, c68k_limb lo, c68k_limb d, - c68k_limb *rem); +AMIGA_ASM_ARGS c68k_limb c68k_div_2by1(c68k_limb hi, c68k_limb lo, c68k_limb d, + c68k_limb *rem); #endif #if !defined(C68K_ASM) || defined(C68K_MV) diff --git a/src/crypto68k/crypto68k.h b/src/crypto68k/crypto68k.h index e5e495ba2..44cfb6545 100644 --- a/src/crypto68k/crypto68k.h +++ b/src/crypto68k/crypto68k.h @@ -13,10 +13,26 @@ #include "nx_crypto_huge_number.h" +#include "aminetxduo/asm_abi.h" + #ifdef __cplusplus extern "C" { #endif +/* + * AMIGA_ASM_ARGS on a declaration below means the definition is in + * the c68k_*.S files in src/crypto68k -- c68k_prim.S, c68k_dispatch.S, + * c68k_25519.S, c68k_p256.S, c68k_prim_mulw.S -- and reads its arguments + * from the stack. See aminetxduo/asm_abi.h for what the pin is. + * + * A few of these names are assembly in one build and portable C in another + * (c68k_addmul_1, c68k_add, c68k_add_carry, c68k_sub, c68k_cmp, + * c68k_div_2by1 all have a C body under the per-file gates in + * c68k_variant.h). The pin is on the name rather than on the definition, so + * those C bodies take the stack convention too -- they are the fallback for a + * configuration that has no assembly, so it costs nothing that ships. + */ + /* * Limb type. The assembly is written for EXACTLY this: 32-bit limbs, * big-endian host, little-endian limb order (limb 0 least significant). @@ -35,7 +51,8 @@ typedef HN_UBASE c68k_limb; * the top limb (a FULL LIMB, not a single bit). The 64-bit intermediate * cannot overflow: (2^32-1)^2 + 2*(2^32-1) = 2^64-1 exactly. */ -c68k_limb c68k_addmul_1(c68k_limb *r, const c68k_limb *b, UINT n, c68k_limb a); +AMIGA_ASM_ARGS c68k_limb c68k_addmul_1(c68k_limb *r, const c68k_limb *b, UINT n, + c68k_limb a); /* The portable C version, always present under its own name whichever build option is in force, so the benchmark can time both in one run. */ @@ -45,16 +62,16 @@ c68k_limb c68k_addmul_1_c(c68k_limb *r, const c68k_limb *b, UINT n, c68k_limb a) * dst[j] = src[j] + carry, for j in 0..n-1. Returns the final carry (0 or 1 * after the first limb). dst can alias src. */ -c68k_limb c68k_add_carry(c68k_limb *dst, const c68k_limb *src, UINT n, - c68k_limb carry); +AMIGA_ASM_ARGS c68k_limb c68k_add_carry(c68k_limb *dst, const c68k_limb *src, UINT n, + c68k_limb carry); /* r[0..n-1] += b[0..n-1]. Returns the carry out (0 or 1). */ -c68k_limb c68k_add(c68k_limb *r, const c68k_limb *b, UINT n); +AMIGA_ASM_ARGS c68k_limb c68k_add(c68k_limb *r, const c68k_limb *b, UINT n); /* * r[0..n-1] -= b[0..n-1]. Returns the borrow out (0 or 1). */ -c68k_limb c68k_sub(c68k_limb *r, const c68k_limb *b, UINT n); +AMIGA_ASM_ARGS c68k_limb c68k_sub(c68k_limb *r, const c68k_limb *b, UINT n); /* r[0..n-1] -= a * b[0..n-1]. Returns the borrow out (a full limb). */ c68k_limb c68k_submul_1(c68k_limb *r, const c68k_limb *b, UINT n, c68k_limb a); @@ -65,8 +82,8 @@ c68k_limb c68k_submul_1(c68k_limb *r, const c68k_limb *b, UINT n, c68k_limb a); * DIVU.L 64/32 is unimplemented on a 68060, so AMINETXDUO_CRYPTO68K_ASM must * never be enabled for a 68060 build. */ -c68k_limb c68k_div_2by1(c68k_limb hi, c68k_limb lo, c68k_limb d, - c68k_limb *rem); +AMIGA_ASM_ARGS c68k_limb c68k_div_2by1(c68k_limb hi, c68k_limb lo, c68k_limb d, + c68k_limb *rem); /* * rem = u mod m by Knuth's algorithm D over 32-bit limbs. m[m_len-1] must be @@ -91,7 +108,7 @@ VOID c68k_mont_setup_rr(c68k_limb *rr, const c68k_limb *m, UINT m_len, c68k_limb *setup); /* Unsigned compare of two n-limb values. -1, 0 or 1. */ -INT c68k_cmp(const c68k_limb *a, const c68k_limb *b, UINT n); +AMIGA_ASM_ARGS INT c68k_cmp(const c68k_limb *a, const c68k_limb *b, UINT n); /* * Which limb primitives were compiled in: 0 portable C, 1 c68k_prim.S (68020), @@ -118,10 +135,14 @@ UINT c68k_cpu_class(VOID); c68k_prim.S is 68000 code called by name everywhere. */ #ifdef C68K_MV -extern c68k_limb (*c68k_vec_addmul_1)(c68k_limb *, const c68k_limb *, UINT, - c68k_limb); -extern c68k_limb (*c68k_vec_div_2by1)(c68k_limb, c68k_limb, c68k_limb, - c68k_limb *); +/* c68k_dispatch.S jumps through these two, so the pointer TYPE carries the + pin as well: an indirect call is compiled from the type it is made + through. The targets are c68k_prim.S / c68k_prim_mulw.S bodies. */ +extern AMIGA_ASM_ARGS c68k_limb (*c68k_vec_addmul_1)(c68k_limb *, + const c68k_limb *, UINT, + c68k_limb); +extern AMIGA_ASM_ARGS c68k_limb (*c68k_vec_div_2by1)(c68k_limb, c68k_limb, + c68k_limb, c68k_limb *); #define C68K_ADDMUL_1 (*c68k_vec_addmul_1) #define C68K_DIV_2BY1 (*c68k_vec_div_2by1) diff --git a/src/net68k/n68k_cpu.c b/src/net68k/n68k_cpu.c index 9949bf72e..526c3aa92 100644 --- a/src/net68k/n68k_cpu.c +++ b/src/net68k/n68k_cpu.c @@ -17,31 +17,34 @@ #include -extern ULONG n68k_sum_longwords_mv0(const ULONG *p, ULONG count); -extern ULONG n68k_sum_longwords_mv20(const ULONG *p, ULONG count); -extern ULONG n68k_sum_longwords_mv40(const ULONG *p, ULONG count); -extern ULONG n68k_sum_longwords_mv60(const ULONG *p, ULONG count); +/* Every variant is a body in n68k_checksum.S or n68k_copy.S, so every one of + them carries AMIGA_ASM_ARGS, and so does the pointer type each is stored + into. See aminetxduo/asm_abi.h. */ +extern AMIGA_ASM_ARGS ULONG n68k_sum_longwords_mv0(const ULONG *p, ULONG count); +extern AMIGA_ASM_ARGS ULONG n68k_sum_longwords_mv20(const ULONG *p, ULONG count); +extern AMIGA_ASM_ARGS ULONG n68k_sum_longwords_mv40(const ULONG *p, ULONG count); +extern AMIGA_ASM_ARGS ULONG n68k_sum_longwords_mv60(const ULONG *p, ULONG count); -extern ULONG n68k_copy_sum_longwords_mv0(ULONG *to, const ULONG *from, - ULONG count); -extern ULONG n68k_copy_sum_longwords_mv20(ULONG *to, const ULONG *from, - ULONG count); -extern ULONG n68k_copy_sum_longwords_mv40(ULONG *to, const ULONG *from, - ULONG count); -extern ULONG n68k_copy_sum_longwords_mv60(ULONG *to, const ULONG *from, - ULONG count); +extern AMIGA_ASM_ARGS ULONG n68k_copy_sum_longwords_mv0(ULONG *to, const ULONG *from, + ULONG count); +extern AMIGA_ASM_ARGS ULONG n68k_copy_sum_longwords_mv20(ULONG *to, const ULONG *from, + ULONG count); +extern AMIGA_ASM_ARGS ULONG n68k_copy_sum_longwords_mv40(ULONG *to, const ULONG *from, + ULONG count); +extern AMIGA_ASM_ARGS ULONG n68k_copy_sum_longwords_mv60(ULONG *to, const ULONG *from, + ULONG count); -extern VOID n68k_copy_bytes_mv0(UCHAR *to, const UCHAR *from, ULONG len); -extern VOID n68k_copy_bytes_mv20(UCHAR *to, const UCHAR *from, ULONG len); -extern VOID n68k_copy_bytes_mv40(UCHAR *to, const UCHAR *from, ULONG len); -extern VOID n68k_copy_bytes_mv60(UCHAR *to, const UCHAR *from, ULONG len); +extern AMIGA_ASM_ARGS VOID n68k_copy_bytes_mv0(UCHAR *to, const UCHAR *from, ULONG len); +extern AMIGA_ASM_ARGS VOID n68k_copy_bytes_mv20(UCHAR *to, const UCHAR *from, ULONG len); +extern AMIGA_ASM_ARGS VOID n68k_copy_bytes_mv40(UCHAR *to, const UCHAR *from, ULONG len); +extern AMIGA_ASM_ARGS VOID n68k_copy_bytes_mv60(UCHAR *to, const UCHAR *from, ULONG len); /* n68k_dispatch.S jumps through these three. They are written once, before the first packet, and read on every one after that. */ -ULONG (*n68k_vec_sum)(const ULONG *, ULONG) = n68k_sum_longwords_mv0; -ULONG (*n68k_vec_copy_sum)(ULONG *, const ULONG *, ULONG) = +AMIGA_ASM_ARGS ULONG (*n68k_vec_sum)(const ULONG *, ULONG) = n68k_sum_longwords_mv0; +AMIGA_ASM_ARGS ULONG (*n68k_vec_copy_sum)(ULONG *, const ULONG *, ULONG) = n68k_copy_sum_longwords_mv0; -VOID (*n68k_vec_copy)(UCHAR *, const UCHAR *, ULONG) = n68k_copy_bytes_mv0; +AMIGA_ASM_ARGS VOID (*n68k_vec_copy)(UCHAR *, const UCHAR *, ULONG) = n68k_copy_bytes_mv0; VOID n68k_cpu_select(ULONG attnflags) { diff --git a/src/net68k/n68k_iocopy.h b/src/net68k/n68k_iocopy.h index 2bfc93a09..cebf3c22d 100644 --- a/src/net68k/n68k_iocopy.h +++ b/src/net68k/n68k_iocopy.h @@ -20,38 +20,43 @@ #include +#include "aminetxduo/asm_abi.h" + #ifdef __cplusplus extern "C" { #endif +/* Every one of these is n68k_iocopy.S, so every one takes AMIGA_ASM_ARGS -- + see aminetxduo/asm_abi.h for what that is. */ + /* Memory to memory, or memory to a mapped card buffer. Both sides advance. */ -VOID n68k_copy_longs(volatile void *to, const volatile void *from, - ULONG longs); +AMIGA_ASM_ARGS VOID n68k_copy_longs(volatile void *to, const volatile void *from, + ULONG longs); /* A data port that does not advance a host address: `port` is one address. */ -VOID n68k_port_in(void *to, const volatile void *port, ULONG blocks); -VOID n68k_port_out(volatile void *port, const void *from, ULONG blocks); +AMIGA_ASM_ARGS VOID n68k_port_in(void *to, const volatile void *port, ULONG blocks); +AMIGA_ASM_ARGS VOID n68k_port_out(volatile void *port, const void *from, ULONG blocks); /* The same, for a 16-bit port that is one address and not mirrored: the reads stay word-wide, the host side moves a block at a time. */ -VOID n68k_port_in_w(void *to, const volatile void *port, ULONG blocks); +AMIGA_ASM_ARGS VOID n68k_port_in_w(void *to, const volatile void *port, ULONG blocks); /* Drain + longword ones-complement sum in one pass; the sum has exactly n68k_copy_sum_longwords() semantics so the verifier cannot tell who produced it. Destination must be even. */ -ULONG n68k_port_in_w_sum(void *to, const volatile void *port, ULONG bytes); -VOID n68k_port_out_w(volatile void *port, const void *from, ULONG blocks); +AMIGA_ASM_ARGS ULONG n68k_port_in_w_sum(void *to, const volatile void *port, ULONG bytes); +AMIGA_ASM_ARGS VOID n68k_port_out_w(volatile void *port, const void *from, ULONG blocks); /* The same fusion for a 32-bit mirrored data window. `longs` is a count of longwords and there is no tail: the caller takes the 1..3 bytes that do not fill one off the 16-bit port, which is what the plain long drain does too. Destination must be longword aligned. */ -ULONG n68k_port_in_l_sum(void *to, const volatile void *port, ULONG longs); +AMIGA_ASM_ARGS ULONG n68k_port_in_l_sum(void *to, const volatile void *port, ULONG longs); /* Memory to memory with the sum, for a mapped buffer that advances (the ZZ9000's receive window). The SOURCE must be longword aligned: that is the side on the slow bus. The destination may be 2 mod 4. */ -ULONG n68k_copy_longs_sum(void *to, const volatile void *from, ULONG longs); +AMIGA_ASM_ARGS ULONG n68k_copy_longs_sum(void *to, const volatile void *from, ULONG longs); #ifdef __cplusplus } diff --git a/src/net68k/n68k_memcpy_hook.c b/src/net68k/n68k_memcpy_hook.c index a8816d943..ce6a25a81 100644 --- a/src/net68k/n68k_memcpy_hook.c +++ b/src/net68k/n68k_memcpy_hook.c @@ -5,12 +5,30 @@ * version is not safe for them either. memmove() covers that case and is * left alone. * + * IS LOAD-BEARING HERE, not a tidiness. Under -mregparm=3 the + * compiler calls memcpy by NAME with every argument on the stack, whatever + * the caller looks like: `*p = g' leaves `pea 64.w / pea _g / move.l a0,-(sp) + * / jsr _memcpy', and an explicit call does the same. A definition of memcpy + * reads those arguments off the stack for the same reason -- the name is + * recognized -- and not because anything here says so. Measured on + * m68k-amigaos-gcc 16.2.0b: the identical body with -fno-builtin instead + * reads d0/a0/a1, so the recognition is the only thing holding the two sides + * together. + * + * The include makes it stated rather than recognized. This toolchain's own + * string.h declares `__stdargs void *memcpy(...)' -- the same + * __attribute__((__stkparm__)) this tree spells AMIGA_ASM_ARGS -- so the + * definition inherits the stack convention from the declaration, and stays + * right if -fno-builtin is ever added. It costs nothing: the hook is already + * stack-convention, so the emitted code is unchanged. + * * SPDX-License-Identifier: MIT */ #include "net68k.h" #include +#include void *memcpy(void *dst, const void *src, size_t n) { diff --git a/src/net68k/net68k.h b/src/net68k/net68k.h index da3e2765c..264f29e9c 100644 --- a/src/net68k/net68k.h +++ b/src/net68k/net68k.h @@ -15,6 +15,8 @@ #include "nx_api.h" +#include "aminetxduo/asm_abi.h" + #ifdef __cplusplus extern "C" { #endif @@ -40,7 +42,7 @@ extern "C" { * below. NetX Duo enters the loop only on a packet's prepend pointer, which * the pool keeps longword aligned. */ -ULONG n68k_sum_longwords(const ULONG *p, ULONG count); +AMIGA_ASM_ARGS ULONG n68k_sum_longwords(const ULONG *p, ULONG count); /* * The replacement for _nx_ip_checksum_compute(). Same signature, semantics @@ -59,7 +61,7 @@ USHORT n68k_ip_checksum_compute(NX_PACKET *packet_ptr, ULONG protocol, * the measurements and the reason C cannot reach it. Off that path it is a * plain loop, present so that a host build links. */ -VOID n68k_copy_bytes(UCHAR *to, const UCHAR *from, ULONG len); +AMIGA_ASM_ARGS VOID n68k_copy_bytes(UCHAR *to, const UCHAR *from, ULONG len); /* * Point the three routines above at the forms this machine wants. The @@ -86,9 +88,12 @@ VOID n68k_cpu_select(ULONG attnflags); */ #ifdef N68K_MV_MULTI -extern ULONG (*n68k_vec_sum)(const ULONG *, ULONG); -extern ULONG (*n68k_vec_copy_sum)(ULONG *, const ULONG *, ULONG); -extern VOID (*n68k_vec_copy)(UCHAR *, const UCHAR *, ULONG); +/* The vectors point at assembly, so the pointer TYPE carries the pin as well: + an indirect call is compiled from the type it is made through, and a plain + prototype here would put the arguments in d0/d1/d2 at every one of them. */ +extern AMIGA_ASM_ARGS ULONG (*n68k_vec_sum)(const ULONG *, ULONG); +extern AMIGA_ASM_ARGS ULONG (*n68k_vec_copy_sum)(ULONG *, const ULONG *, ULONG); +extern AMIGA_ASM_ARGS VOID (*n68k_vec_copy)(UCHAR *, const UCHAR *, ULONG); #define N68K_SUM_LONGWORDS (*n68k_vec_sum) #define N68K_COPY_SUM_LONGWORDS (*n68k_vec_copy_sum) @@ -113,7 +118,7 @@ extern VOID (*n68k_vec_copy)(UCHAR *, const UCHAR *, ULONG); * longword aligned. The caller decides that, because it is a fact about the * frame in front of it. */ -ULONG n68k_copy_sum_longwords(ULONG *to, const ULONG *from, ULONG count); +AMIGA_ASM_ARGS ULONG n68k_copy_sum_longwords(ULONG *to, const ULONG *from, ULONG count); #ifdef AMINETXDUO_RX_VERIFY diff --git a/src/netdev/test/test_netdev_lance.c b/src/netdev/test/test_netdev_lance.c index ad92f41e4..b9f141bf2 100644 --- a/src/netdev/test/test_netdev_lance.c +++ b/src/netdev/test/test_netdev_lance.c @@ -37,6 +37,7 @@ static VOID mock_csr_put(NetdevNic *nic, UWORD csr, UWORD value); #define LANCE_RDP_PUT(nic, val) mock_csr_put((nic), LE_CSR0, (val)) #include "lance.c" +#include "aminetxduo/asm_abi.h" static union { @@ -62,7 +63,7 @@ static VOID expect_u32(const char *what, ULONG got, ULONG want) /* The transmit copy is linked into lance.c but these interrupt tests never call it. Keep the definition honest for any later transmit fixture. */ -VOID n68k_copy_longs(volatile void *to, const volatile void *from, ULONG longs) +AMIGA_ASM_ARGS VOID n68k_copy_longs(volatile void *to, const volatile void *from, ULONG longs) { volatile ULONG *dst = (volatile ULONG *)to; const volatile ULONG *src = (const volatile ULONG *)from; diff --git a/src/netdev/test/test_netdev_lance_csr.c b/src/netdev/test/test_netdev_lance_csr.c index fd7cff7b1..5704a37e5 100644 --- a/src/netdev/test/test_netdev_lance_csr.c +++ b/src/netdev/test/test_netdev_lance_csr.c @@ -29,10 +29,11 @@ /* No LANCE_CSR_GET/PUT here: that is the entire point of this file. */ #include "lance.c" +#include "aminetxduo/asm_abi.h" /* lance_tx() links against it; nothing here calls it. Same stub as test_netdev_lance.c keeps, and honest for a later transmit fixture. */ -VOID n68k_copy_longs(volatile void *to, const volatile void *from, ULONG longs) +AMIGA_ASM_ARGS VOID n68k_copy_longs(volatile void *to, const volatile void *from, ULONG longs) { volatile ULONG *dst = (volatile ULONG *)to; const volatile ULONG *src = (const volatile ULONG *)from; diff --git a/src/netdev/test/test_netdev_zz9000.c b/src/netdev/test/test_netdev_zz9000.c index fce0a4c06..4e4610c64 100644 --- a/src/netdev/test/test_netdev_zz9000.c +++ b/src/netdev/test/test_netdev_zz9000.c @@ -19,6 +19,10 @@ #include +/* The stubs below carry AMIGA_ASM_ARGS, so the pin has to be visible + before them, not after the driver it is included from. */ +#include "aminetxduo/asm_abi.h" + #include "netdev_nic.h" #include "netdev_clock.h" @@ -44,7 +48,7 @@ static ULONG bulk_calls; static ULONG bulk_misaligned; /* sources not 0 mod 4 */ static ULONG bulk_longs; -VOID n68k_copy_longs(volatile void *to, const volatile void *from, ULONG longs) +AMIGA_ASM_ARGS VOID n68k_copy_longs(volatile void *to, const volatile void *from, ULONG longs) { bulk_calls++; bulk_longs += longs; @@ -53,7 +57,7 @@ VOID n68k_copy_longs(volatile void *to, const volatile void *from, ULONG longs) memcpy((void *)to, (const void *)from, longs << 2); } -ULONG n68k_copy_longs_sum(void *to, const volatile void *from, ULONG longs) +AMIGA_ASM_ARGS ULONG n68k_copy_longs_sum(void *to, const volatile void *from, ULONG longs) { const UBYTE *s = (const UBYTE *)from; ULONG sum = 0; diff --git a/src/tools/tool_startup.S b/src/tools/tool_startup.S index cba118955..1da44f325 100644 --- a/src/tools/tool_startup.S +++ b/src/tools/tool_startup.S @@ -120,6 +120,13 @@ _____start: | Which form of src/common/ami_udivdi3.c's helpers this machine wants, and the | reference that pulls that archive ahead of libgcc's. The 68060 traps the | 64-bit MULU.L/DIVU.L forms, so it gets the 68020 arm without them. +| +| The two flags are loaded into d0/d1 AND then pushed, and both are wanted. +| ami_rt_cpu_select() is the one call into C in this tree that follows the +| build's convention instead of pinning it (see the argument-passing note in +| cmake/toolchain-m68k-amigaos.cmake), so under -mregparm=0 the callee reads +| the stack and under -mregparm=3 it reads d0/d1. Doing both is what makes +| one file serve either build. Do not "simplify" the push away. moveq #0,d0 moveq #0,d1 btst #AFFB_68020,a6@(AttnFlags+1) diff --git a/tests/crypto68k/CMakeLists.txt b/tests/crypto68k/CMakeLists.txt index 7b7612072..27dc06291 100644 --- a/tests/crypto68k/CMakeLists.txt +++ b/tests/crypto68k/CMakeLists.txt @@ -274,6 +274,7 @@ if(NOT CMAKE_CROSSCOMPILING) target_include_directories(test_crypto68k PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/host/shim" # nx_crypto_port.h, first "${CMAKE_CURRENT_SOURCE_DIR}" # c68k_vectors.h + "${CMAKE_SOURCE_DIR}/include" # aminetxduo/asm_abi.h "${CMAKE_SOURCE_DIR}/src/crypto68k" "${AMINETXDUO_NETXDUO}/crypto_libraries/inc") @@ -297,6 +298,7 @@ if(NOT CMAKE_CROSSCOMPILING) target_include_directories(test_crypto68k_25519 PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/host/shim" # nx_crypto_port.h, first + "${CMAKE_SOURCE_DIR}/include" # aminetxduo/asm_abi.h "${CMAKE_SOURCE_DIR}/src/crypto68k" "${AMINETXDUO_NETXDUO}/crypto_libraries/inc") diff --git a/tests/fuzz/CMakeLists.txt b/tests/fuzz/CMakeLists.txt index e0f48cfdc..906ff65a4 100644 --- a/tests/fuzz/CMakeLists.txt +++ b/tests/fuzz/CMakeLists.txt @@ -550,8 +550,15 @@ if(CMAKE_SIZEOF_VOID_P EQUAL 4) # src/config/test/shim carries the minimal that src/tls/tls.h # needs; tests/tls carries the sample certificate and its private key. + # + # include/ is here for src/crypto68k/crypto68k.h, which includes + # for the AMIGA_ASM_ARGS pin on the c68k_* names. + # That pin is what makes an assembly routine read its arguments where the + # callers leave them, so the header carries it in every build, this one + # included, whether or not that build has the assembly. set(_fuzz_tls_crypto_inc ${_fuzz_tls_inc} + "${CMAKE_SOURCE_DIR}/include" "${CMAKE_SOURCE_DIR}/src/config/test/shim" "${CMAKE_SOURCE_DIR}/src/tls" "${CMAKE_SOURCE_DIR}/src/crypto68k" diff --git a/tests/perf/CMakeLists.txt b/tests/perf/CMakeLists.txt index bf881ed80..6265667ae 100644 --- a/tests/perf/CMakeLists.txt +++ b/tests/perf/CMakeLists.txt @@ -347,6 +347,7 @@ if(NOT CMAKE_CROSSCOMPILING) "${AMINETXDUO_NETXDUO}/ports/linux/gnu/inc" # nx_port.h "${AMINETXDUO_NETXDUO}/common/inc" "${AMINETXDUO_THREADX}/common/inc" + "${CMAKE_SOURCE_DIR}/include" "${CMAKE_SOURCE_DIR}/src/net68k") target_compile_definitions(test_net68k_checksum PRIVATE @@ -374,6 +375,7 @@ if(NOT CMAKE_CROSSCOMPILING) "${AMINETXDUO_NETXDUO}/ports/linux/gnu/inc" "${AMINETXDUO_NETXDUO}/common/inc" "${AMINETXDUO_THREADX}/common/inc" + "${CMAKE_SOURCE_DIR}/include" "${CMAKE_SOURCE_DIR}/src/net68k") # FEATURE_NX_IPV6 is what selects the 128-bit pseudo header in the checksum diff --git a/tests/perf/n68kmv.c b/tests/perf/n68kmv.c index a49e4190f..59d01da2c 100644 --- a/tests/perf/n68kmv.c +++ b/tests/perf/n68kmv.c @@ -22,31 +22,32 @@ #include #include "aminetxduo/compat.h" +#include "aminetxduo/asm_abi.h" /* Declared here rather than from net68k.h, which reaches nx_api.h and its own typedef of VOID. */ -extern ULONG n68k_sum_longwords(const ULONG *p, ULONG count); -extern VOID n68k_copy_bytes(UBYTE *to, const UBYTE *from, ULONG len); +extern AMIGA_ASM_ARGS ULONG n68k_sum_longwords(const ULONG *p, ULONG count); +extern AMIGA_ASM_ARGS VOID n68k_copy_bytes(UBYTE *to, const UBYTE *from, ULONG len); extern VOID n68k_cpu_select(ULONG attnflags); -extern ULONG n68k_sum_longwords_mv0(const ULONG *p, ULONG count); -extern ULONG n68k_sum_longwords_mv20(const ULONG *p, ULONG count); -extern ULONG n68k_sum_longwords_mv40(const ULONG *p, ULONG count); -extern ULONG n68k_sum_longwords_mv60(const ULONG *p, ULONG count); +extern AMIGA_ASM_ARGS ULONG n68k_sum_longwords_mv0(const ULONG *p, ULONG count); +extern AMIGA_ASM_ARGS ULONG n68k_sum_longwords_mv20(const ULONG *p, ULONG count); +extern AMIGA_ASM_ARGS ULONG n68k_sum_longwords_mv40(const ULONG *p, ULONG count); +extern AMIGA_ASM_ARGS ULONG n68k_sum_longwords_mv60(const ULONG *p, ULONG count); -extern ULONG n68k_copy_sum_longwords_mv0(ULONG *to, const ULONG *from, +extern AMIGA_ASM_ARGS ULONG n68k_copy_sum_longwords_mv0(ULONG *to, const ULONG *from, ULONG count); -extern ULONG n68k_copy_sum_longwords_mv20(ULONG *to, const ULONG *from, +extern AMIGA_ASM_ARGS ULONG n68k_copy_sum_longwords_mv20(ULONG *to, const ULONG *from, ULONG count); -extern ULONG n68k_copy_sum_longwords_mv40(ULONG *to, const ULONG *from, +extern AMIGA_ASM_ARGS ULONG n68k_copy_sum_longwords_mv40(ULONG *to, const ULONG *from, ULONG count); -extern ULONG n68k_copy_sum_longwords_mv60(ULONG *to, const ULONG *from, +extern AMIGA_ASM_ARGS ULONG n68k_copy_sum_longwords_mv60(ULONG *to, const ULONG *from, ULONG count); -extern VOID n68k_copy_bytes_mv0(UBYTE *to, const UBYTE *from, ULONG len); -extern VOID n68k_copy_bytes_mv20(UBYTE *to, const UBYTE *from, ULONG len); -extern VOID n68k_copy_bytes_mv40(UBYTE *to, const UBYTE *from, ULONG len); -extern VOID n68k_copy_bytes_mv60(UBYTE *to, const UBYTE *from, ULONG len); +extern AMIGA_ASM_ARGS VOID n68k_copy_bytes_mv0(UBYTE *to, const UBYTE *from, ULONG len); +extern AMIGA_ASM_ARGS VOID n68k_copy_bytes_mv20(UBYTE *to, const UBYTE *from, ULONG len); +extern AMIGA_ASM_ARGS VOID n68k_copy_bytes_mv40(UBYTE *to, const UBYTE *from, ULONG len); +extern AMIGA_ASM_ARGS VOID n68k_copy_bytes_mv60(UBYTE *to, const UBYTE *from, ULONG len); extern VOID (*n68k_vec_copy)(UBYTE *, const UBYTE *, ULONG); @@ -179,7 +180,7 @@ static ULONG copy_sum_reference(ULONG *to, const ULONG *from, ULONG count) /* ---------------------------------------------------------- the checks --- */ -static VOID check_sum(const char *name, ULONG (*fn)(const ULONG *, ULONG)) +static VOID check_sum(const char *name, AMIGA_ASM_ARGS ULONG (*fn)(const ULONG *, ULONG)) { ULONG n; @@ -200,7 +201,7 @@ static VOID check_sum(const char *name, ULONG (*fn)(const ULONG *, ULONG)) } static VOID check_copy_sum(const char *name, - ULONG (*fn)(ULONG *, const ULONG *, ULONG)) + AMIGA_ASM_ARGS ULONG (*fn)(ULONG *, const ULONG *, ULONG)) { ULONG n; @@ -242,7 +243,7 @@ static VOID check_copy_sum(const char *name, /* `all_offsets` must be false on a 68000: the unguarded forms read a longword from an odd address, which is an address error there. */ static VOID check_copy(const char *name, - VOID (*fn)(UBYTE *, const UBYTE *, ULONG), + AMIGA_ASM_ARGS VOID (*fn)(UBYTE *, const UBYTE *, ULONG), int all_offsets) { static const ULONG extra[] = { 96, 127, 128, 129, 160, 255, 256, 300 }; @@ -295,7 +296,12 @@ static VOID check_copy(const char *name, #define ROUNDS 3 -static VOID bench_sum(const char *name, ULONG (*fn)(const ULONG *, ULONG), +/* The pointer TYPE carries the pin as well as the declarations above: an + indirect call is compiled from the type it is made through, and every one + of these targets is n68k_checksum.S or n68k_copy.S, which read the stack. + Without it the bench would call them with the arguments in d0/d1/d2 and + time whatever they made of the registers -- see aminetxduo/asm_abi.h. */ +static VOID bench_sum(const char *name, AMIGA_ASM_ARGS ULONG (*fn)(const ULONG *, ULONG), ULONG words, ULONG reps) { ULONG best = 0xFFFFFFFFUL; @@ -319,7 +325,7 @@ static VOID bench_sum(const char *name, ULONG (*fn)(const ULONG *, ULONG), } static VOID bench_copy(const char *name, - VOID (*fn)(UBYTE *, const UBYTE *, ULONG), + AMIGA_ASM_ARGS VOID (*fn)(UBYTE *, const UBYTE *, ULONG), ULONG len, ULONG reps) { ULONG best = 0xFFFFFFFFUL; diff --git a/tests/perf/prof/prof.c b/tests/perf/prof/prof.c index 1018998e2..a778dcbef 100644 --- a/tests/perf/prof/prof.c +++ b/tests/perf/prof/prof.c @@ -24,6 +24,7 @@ #include #include #include +#include "aminetxduo/asm_abi.h" #include "aminetxduo/compat.h" /* ami_millis(): the probe needs to poke timer.device */ @@ -108,10 +109,10 @@ ULONG prof_chain; /* the vector we displaced */ ULONG prof_taskptr; /* &SysBase->ThisTask */ ULONG prof_ciaticks; /* interrupts from OUR CIA timer specifically */ -extern VOID prof_vector(VOID); -extern VOID prof_cia_stub(VOID); -extern VOID prof_audio_stub(VOID); -extern ULONG prof_read_vbr(VOID); +extern AMIGA_ASM_ARGS VOID prof_vector(VOID); +extern AMIGA_ASM_ARGS VOID prof_cia_stub(VOID); +extern AMIGA_ASM_ARGS VOID prof_audio_stub(VOID); +extern AMIGA_ASM_ARGS ULONG prof_read_vbr(VOID); #define PROF_MAX_LIBS 192 #define PROF_MAX_LVOS 8192 diff --git a/tests/perf/prof/profverify.c b/tests/perf/prof/profverify.c index 0804bfe3f..9a114fb6b 100644 --- a/tests/perf/prof/profverify.c +++ b/tests/perf/prof/profverify.c @@ -12,9 +12,13 @@ #include -extern VOID pv_spin_a(ULONG reps); -extern VOID pv_spin_b(ULONG reps); -extern VOID pv_spin_c(ULONG reps); +/* profverify.S, and it reads its argument at 4(sp). See + aminetxduo/asm_abi.h. */ +#include "aminetxduo/asm_abi.h" + +extern AMIGA_ASM_ARGS VOID pv_spin_a(ULONG reps); +extern AMIGA_ASM_ARGS VOID pv_spin_b(ULONG reps); +extern AMIGA_ASM_ARGS VOID pv_spin_c(ULONG reps); extern UBYTE pv_spin_a_end, pv_spin_b_end, pv_spin_c_end; #define PV_RATE 1000UL diff --git a/tests/sana2/host/test_sana2_copy_host.c b/tests/sana2/host/test_sana2_copy_host.c index 0ec7ea5ed..47c4663a9 100644 --- a/tests/sana2/host/test_sana2_copy_host.c +++ b/tests/sana2/host/test_sana2_copy_host.c @@ -8,6 +8,7 @@ #include #include +#include "aminetxduo/asm_abi.h" static unsigned long h_checks; @@ -27,7 +28,7 @@ static void h_check(int ok, const char *what) /* The real one is src/net68k/n68k_copy.S. Everything below is about which bytes are asked for, not how they are moved. */ -VOID n68k_copy_bytes(UCHAR *to, const UCHAR *from, ULONG len) +AMIGA_ASM_ARGS VOID n68k_copy_bytes(UCHAR *to, const UCHAR *from, ULONG len) { h_copy_bytes_calls++; if (len != 0) @@ -37,7 +38,7 @@ VOID n68k_copy_bytes(UCHAR *to, const UCHAR *from, ULONG len) /* The contract from src/net68k/n68k_checksum.c, not a memcpy: the transmit checksum is only correct if this is, and a stub that copied without summing would make every checksum assertion in this file pass for free. */ -ULONG n68k_copy_sum_longwords(ULONG *to, const ULONG *from, ULONG count) +AMIGA_ASM_ARGS ULONG n68k_copy_sum_longwords(ULONG *to, const ULONG *from, ULONG count) { ULONG acc = 0; diff --git a/tests/sana2/host/test_sana2_copy_netx_host.c b/tests/sana2/host/test_sana2_copy_netx_host.c index 6c5e27c2a..0611dca03 100644 --- a/tests/sana2/host/test_sana2_copy_netx_host.c +++ b/tests/sana2/host/test_sana2_copy_netx_host.c @@ -13,6 +13,7 @@ #include #include +#include "aminetxduo/asm_abi.h" static unsigned long h_checks; @@ -31,13 +32,13 @@ static void h_check(int ok, const char *what) /* src/net68k/n68k_copy.S and n68k_checksum.c on the target; the same contracts here, as in test_sana2_copy_host.c. */ -VOID n68k_copy_bytes(UCHAR *to, const UCHAR *from, ULONG len) +AMIGA_ASM_ARGS VOID n68k_copy_bytes(UCHAR *to, const UCHAR *from, ULONG len) { if (len != 0) memcpy(to, from, (size_t)len); } -ULONG n68k_copy_sum_longwords(ULONG *to, const ULONG *from, ULONG count) +AMIGA_ASM_ARGS ULONG n68k_copy_sum_longwords(ULONG *to, const ULONG *from, ULONG count) { ULONG acc = 0; diff --git a/tests/sana2/host/test_sana2_driver_host.c b/tests/sana2/host/test_sana2_driver_host.c index dc0fe405d..5ebc9d870 100644 --- a/tests/sana2/host/test_sana2_driver_host.c +++ b/tests/sana2/host/test_sana2_driver_host.c @@ -10,6 +10,7 @@ #include #include +#include "aminetxduo/asm_abi.h" static unsigned long h_checks; static unsigned long h_failures; @@ -215,13 +216,13 @@ VOID ami_log(int level, const char *fmt, ...) (VOID)fmt; } -VOID n68k_copy_bytes(UCHAR *to, const UCHAR *from, ULONG len) +AMIGA_ASM_ARGS VOID n68k_copy_bytes(UCHAR *to, const UCHAR *from, ULONG len) { if (len != 0) memcpy(to, from, (size_t)len); } -ULONG n68k_copy_sum_longwords(ULONG *to, const ULONG *from, ULONG count) +AMIGA_ASM_ARGS ULONG n68k_copy_sum_longwords(ULONG *to, const ULONG *from, ULONG count) { ULONG acc = 0; diff --git a/tests/sana2/host/test_sana2_tx_host.c b/tests/sana2/host/test_sana2_tx_host.c index 52618370f..05addc5ba 100644 --- a/tests/sana2/host/test_sana2_tx_host.c +++ b/tests/sana2/host/test_sana2_tx_host.c @@ -13,6 +13,7 @@ #include #include +#include "aminetxduo/asm_abi.h" static unsigned long h_checks; static unsigned long h_failures; @@ -152,13 +153,13 @@ static ULONG h_sleeps; static ULONG h_reply_on_sleep; static ULONG h_flush_reply_on_sleep; -VOID n68k_copy_bytes(UCHAR *to, const UCHAR *from, ULONG len) +AMIGA_ASM_ARGS VOID n68k_copy_bytes(UCHAR *to, const UCHAR *from, ULONG len) { if (len != 0) memcpy(to, from, (size_t)len); } -ULONG n68k_copy_sum_longwords(ULONG *to, const ULONG *from, ULONG count) +AMIGA_ASM_ARGS ULONG n68k_copy_sum_longwords(ULONG *to, const ULONG *from, ULONG count) { ULONG acc = 0; diff --git a/tools/profiler/profspin.c b/tools/profiler/profspin.c index 0ecbf0842..f710e547d 100644 --- a/tools/profiler/profspin.c +++ b/tools/profiler/profspin.c @@ -29,9 +29,26 @@ #include #include -extern VOID spin_a(ULONG reps); -extern VOID spin_b(ULONG reps); -extern VOID spin_c(ULONG reps); +/* + * The three kernels are tools/profiler/profspin.S and read their argument at + * 4(sp). This is built with the stack's own flags, -mregparm=3 among them, so + * without a pin here `reps` would go to d0 and the kernel would count whatever + * the stack happened to hold. + * + * The attribute is spelled out rather than taken from + * : this directory is self-contained on purpose (see + * CMakeLists.txt) and is meant to be lifted out into its own repository, so it + * does not grow an include path into the stack for five lines. + */ +#if defined(__m68k__) && (defined(__GNUC__) || defined(__clang__)) +# define SPIN_ASM_ARGS __attribute__((__stkparm__)) +#else +# define SPIN_ASM_ARGS +#endif + +extern SPIN_ASM_ARGS VOID spin_a(ULONG reps); +extern SPIN_ASM_ARGS VOID spin_b(ULONG reps); +extern SPIN_ASM_ARGS VOID spin_c(ULONG reps); extern UBYTE spin_a_end, spin_b_end, spin_c_end; #define TEMPLATE "RANGES/K,SCALE/K/N,AUDIO/K/N"