diff --git a/tslang/include/TypeScript/gcwrapper.h b/tslang/include/TypeScript/gcwrapper.h index 4f31f969e..46bbbef1a 100644 --- a/tslang/include/TypeScript/gcwrapper.h +++ b/tslang/include/TypeScript/gcwrapper.h @@ -13,6 +13,8 @@ void *_mlir__GC_malloc_atomic(size_t size); void *_mlir__GC_memalign(size_t align, size_t size); +void *_mlir__GC_malloc_uncollectable(size_t size); + void *_mlir__GC_realloc(void *ptr, size_t size); void _mlir__GC_free(void *ptr); diff --git a/tslang/lib/AsyncGCThreadsCommon.inc b/tslang/lib/AsyncGCThreadsCommon.inc index 5164bbe45..f43817b6e 100644 --- a/tslang/lib/AsyncGCThreadsCommon.inc +++ b/tslang/lib/AsyncGCThreadsCommon.inc @@ -11,15 +11,41 @@ #include "TypeScript/AsyncGCThreads.h" +#include #include -#ifndef _WIN32 +#ifdef _WIN32 +#include +#include +#else #include #endif namespace { +// Whether this thread is inside a DllMain - loading or unloading some DLL - where it holds the +// loader lock, and a thread it starts cannot run until the lock is released. The lock is the +// PEB's LoaderLock, which winternl.h leaves out of its PEB; its offset has stayed the same since +// Windows NT 4. Off Windows there is no such lock: a shared object's constructors run under +// dlopen's lock, which a new thread does not need to start. +bool holdsLoaderLock() +{ +#ifdef _WIN32 +#ifdef _WIN64 + constexpr size_t loaderLockOffset = 0x110; +#else + constexpr size_t loaderLockOffset = 0xA0; +#endif + auto peb = reinterpret_cast(NtCurrentTeb()->ProcessEnvironmentBlock); + auto loaderLock = *reinterpret_cast(peb + loaderLockOffset); + return loaderLock != nullptr && + reinterpret_cast(loaderLock->OwningThread) == static_cast(GetCurrentThreadId()); +#else + return false; +#endif +} + bool registerThisThread() { struct GC_stack_base sb; @@ -32,8 +58,8 @@ void unregisterThisThread() } // A thread that came into a gc shared library through one of its exported functions, from a host -// that is not a tslang program (an Android app's JNI threads, a C program's): checked once, and -// registered then if the collector did not know it, until the thread ends. +// that is not a tslang program (an Android app's JNI threads, a C program's): checked once threads +// are enabled, and registered then if the collector did not know it, until the thread ends. struct CallingThread { bool checked = false; @@ -52,12 +78,47 @@ struct CallingThread thread_local CallingThread callingThread; std::once_flag collectorInitialized; -std::once_flag threadsEnabled; +std::once_flag threadsEnabledOnce; +std::atomic threadsEnabled{false}; } // namespace extern "C" void GC_enable_threads(); +namespace +{ + +// Runs GC_enable_threads once, unless now is no time for it: false then, and a later call does. +// +// Not inside a DllMain: GC_allow_register_threads starts the collector's markers and waits for +// them to come up, which they cannot do until the loader lock is released, so the load never +// finished. An exported function runs there when the library calls one while it loads - a static +// constructor of an exported class, top-level code calling an export (the default library hung +// every JIT run). Nor does such a call wait for another thread that is enabling them: that one +// waits for the markers, and they for this thread's loader lock. The markers running already +// (GC_get_parallel) would not tell it either: the count is set before they are waited for. +bool enableThreads() +{ + if (threadsEnabled.load(std::memory_order_acquire)) + { + return true; + } + + if (holdsLoaderLock()) + { + return false; + } + + std::call_once(threadsEnabledOnce, [] { + GC_enable_threads(); + threadsEnabled.store(true, std::memory_order_release); + }); + + return true; +} + +} // namespace + // Called first thing by every exported function of a `-mm=gc` shared library (the GC pass injects // it). The collector stops and scans only the threads registered with it, and aborts a collection // that a thread it does not know started ("Collecting from unknown thread"): a tslang program @@ -67,15 +128,18 @@ extern "C" void GC_enable_threads(); // The collector may well be initialized already: it initializes itself on first use, so a library // with no top-level code (no GC_init injected at load) can find it running. // -// GC_enable_threads runs only for a thread that has to be registered: GC_register_my_thread -// refuses every thread until GC_allow_register_threads was called ("Threads explicit registering -// is not previously enabled"). A registered thread must not run it here. An exported function can -// be called while the library is being loaded - a static constructor of an exported class, or -// top-level code calling an export - and on Windows that is inside DllMain, under the loader lock. -// GC_allow_register_threads starts the parallel marker threads and waits for them to come up, and -// a new thread cannot start until the loader lock is released: the load never finished (the -// default library hung every JIT run). The thread a library is loaded on is the one GC_init -// registered, or one its host registered. +// The first call that may enables threads (enableThreads), whichever thread makes it - before it +// goes on, and before any other thread can come in: +// - GC_register_my_thread refuses every thread until GC_allow_register_threads was called +// ("Threads explicit registering is not previously enabled"); +// - GC_allow_register_threads makes the allocator take its lock only after it started the markers, +// so a newcomer enabling threads raced a thread that was already allocating (#523); +// - the library's own async pool registers its threads through hooks only the library's +// GC_enable_threads sets - each module has its own copy of the scheduler - so they must be set +// even when every caller is registered already (#522). +// While the library loads, the call cannot enable them (see enableThreads) and the thread goes on +// as it is: the one loading is the one GC_init registered, or one its host registered, and no other +// can call in before the load finishes. Its next call, after the load, enables them. extern "C" void __tslang_gc_enter() { auto &thread = callingThread; @@ -84,8 +148,6 @@ extern "C" void __tslang_gc_enter() return; } - thread.checked = true; - std::call_once(collectorInitialized, [] { if (!GC_is_init_called()) { @@ -93,14 +155,16 @@ extern "C" void __tslang_gc_enter() } }); - if (GC_thread_is_registered()) + if (!enableThreads()) { return; } - std::call_once(threadsEnabled, [] { GC_enable_threads(); }); - - thread.registeredHere = registerThisThread(); + thread.checked = true; + if (!GC_thread_is_registered()) + { + thread.registeredHere = registerThisThread(); + } } // Called once from the entry point: the GC pass injects the call beside GC_init, so it happens diff --git a/tslang/lib/TypeScript/AsyncTargetWidthPass.cpp b/tslang/lib/TypeScript/AsyncTargetWidthPass.cpp index ee1ed1396..5899ca9c7 100644 --- a/tslang/lib/TypeScript/AsyncTargetWidthPass.cpp +++ b/tslang/lib/TypeScript/AsyncTargetWidthPass.cpp @@ -46,8 +46,8 @@ class AsyncTargetWidthPass : public mlir::PassWrapper { @@ -46,6 +48,8 @@ class GCPass : public mlir::PassWrapper llvm::SmallVector redundantMemSets; llvm::SmallVector callocCalls; llvm::SmallVector callocDecls; + llvm::SmallVector frameAllocCalls; + llvm::SmallVector frameAllocDecls; m.walk([&](mlir::Operation *op) { // process gctors first if (auto funcOp = dyn_cast_or_null(op)) @@ -63,6 +67,12 @@ class GCPass : public mlir::PassWrapper return; } + if (name == ALIGNED_ALLOC_NAME) + { + frameAllocDecls.push_back(funcOp); + return; + } + if (!funcOp.getBody().empty()) { if (!added) @@ -106,6 +116,12 @@ class GCPass : public mlir::PassWrapper return; } + if (name == ALIGNED_ALLOC_NAME) + { + frameAllocCalls.push_back(callOp); + return; + } + renameCall(name, callOp); } }); @@ -116,6 +132,7 @@ class GCPass : public mlir::PassWrapper } replaceCallocWithGCMalloc(m, callocCalls, callocDecls); + replaceFrameAllocWithUncollectable(m, frameAllocCalls, frameAllocDecls); if (!added) { @@ -138,26 +155,23 @@ class GCPass : public mlir::PassWrapper LLVM_DEBUG(llvm::dbgs() << "\n!! GCPass: AFTER DUMP: \n" << m << "\n";); } - // `calloc` is deliberately absent here: it takes two arguments where GC_malloc takes one, so - // a rename in place would leave a call whose arity disagrees with its callee. It goes through - // replaceCallocWithGCMalloc instead. + // `calloc` and `aligned_alloc` are deliberately absent here: they take two arguments where + // GC_malloc and GC_malloc_uncollectable take one, so a rename in place would leave a call + // whose arity disagrees with its callee. They go through replaceCallocWithGCMalloc and + // replaceFrameAllocWithUncollectable instead. bool mapName(StringRef name, StringRef modeName, StringRef &newName) { if (name == "malloc") { if (modeName == "atomic") { - newName = "GC_malloc_atomic"; + newName = "GC_malloc_atomic"; } else { newName = "GC_malloc"; } } - else if (name == "aligned_alloc") - { - newName = "GC_memalign"; - } else if (name == "realloc") { newName = "GC_realloc"; @@ -206,7 +220,8 @@ class GCPass : public mlir::PassWrapper // each call is treated as returning a distinct, non-aliasing pointer. void markAsAllocatorIfNeeded(StringRef newName, LLVM::LLVMFuncOp funcOp) { - if (newName != "GC_malloc" && newName != "GC_malloc_atomic" && newName != "GC_memalign") + if (newName != "GC_malloc" && newName != "GC_malloc_atomic" && newName != "GC_memalign" && + newName != "GC_malloc_uncollectable") { return; } @@ -222,7 +237,7 @@ class GCPass : public mlir::PassWrapper LLVM::ModRefInfo::NoModRef, LLVM::ModRefInfo::NoModRef); funcOp.setMemoryEffectsAttr(memoryEffects); - // AllocFnKind::Alloc = 1<<0, Zeroed = 1<<4. GC_malloc/GC_malloc_atomic zero-fill; + // AllocFnKind::Alloc = 1<<0, Zeroed = 1<<4. GC_malloc/GC_malloc_atomic/GC_malloc_uncollectable zero-fill; // GC_memalign (GC_memalign) does not guarantee zeroing, so only mark Alloc for it. uint64_t allocKind = newName == "GC_memalign" ? /*Alloc*/ 1 : /*Alloc|Zeroed*/ 1 | (1 << 4); auto kindEntry = mlir::ArrayAttr::get( @@ -318,6 +333,47 @@ class GCPass : public mlir::PassWrapper } } + // An async function's frame is the coroutine lowering's `aligned_alloc(align, size)`, given + // back with `free` (GC_free) when the coroutine finishes - its lifetime is managed by hand. It + // must not be collectable: a frame waiting to run is held only by the runtime's pool queue or + // a token's awaiter list, ordinary heap the collector does not scan. A collection then freed + // it, and the pool resumed whatever reused the block - a crash, or another call's result - + // as soon as several threads awaited at once. Uncollectable memory lives until it is freed + // and is still scanned, so what the frame holds across a suspension stays alive too. + // + // The alignment is dropped: the frame asks for 8 (see the runtime's aligned_alloc shim), and + // the collector aligns every object to its granule, two words. + void replaceFrameAllocWithUncollectable(mlir::ModuleOp module, llvm::SmallVector &calls, + llvm::SmallVector &decls) + { + if (calls.empty() && decls.empty()) + { + return; + } + + PatternRewriter rewriter(module.getContext()); + TypeHelper th(module.getContext()); + + for (auto callOp : calls) + { + LLVMCodeHelper ch(callOp, rewriter, nullptr, tsContext.compileOptions); + auto sizeValue = callOp.getOperand(1); + auto allocFuncOp = ch.getOrInsertFunction( + "GC_malloc_uncollectable", + th.getFunctionType(th.getPtrType(), mlir::ArrayRef{sizeValue.getType()})); + markAsAllocatorIfNeeded("GC_malloc_uncollectable", allocFuncOp); + + rewriter.setInsertionPoint(callOp); + auto allocCall = rewriter.create(callOp->getLoc(), allocFuncOp, ValueRange{sizeValue}); + rewriter.replaceOp(callOp, allocCall.getResults()); + } + + for (auto funcOp : decls) + { + funcOp.erase(); + } + } + static bool isExported(LLVM::LLVMFuncOp funcOp) { if (auto passthrough = funcOp->getAttrOfType("passthrough")) diff --git a/tslang/lib/TypeScriptAsyncRuntime/AsyncRuntime.cpp b/tslang/lib/TypeScriptAsyncRuntime/AsyncRuntime.cpp index 3adce0ad8..83aa48f79 100644 --- a/tslang/lib/TypeScriptAsyncRuntime/AsyncRuntime.cpp +++ b/tslang/lib/TypeScriptAsyncRuntime/AsyncRuntime.cpp @@ -32,7 +32,7 @@ // The coroutine lowering allocates an async function's frame by calling `aligned_alloc` by name // and releases it with plain `free`. MSVC has no `aligned_alloc`, so an executable that awaits // anything does not link at all unless the model rewrites the call (which only `-mm=gc` does, -// to GC_memalign). `_aligned_malloc` is not a stand-in: its memory may only go back through +// to GC_malloc_uncollectable). `_aligned_malloc` is not a stand-in: its memory may only go back through // `_aligned_free`, and pairing it with `free` corrupts the CRT heap. Windows' `malloc` is // already aligned enough for every request anything makes - the frame asks for 8 - so it serves // the request and keeps the pairing honest. diff --git a/tslang/lib/TypeScriptRuntime/MemRuntime.cpp b/tslang/lib/TypeScriptRuntime/MemRuntime.cpp index e39c911e4..915b26620 100644 --- a/tslang/lib/TypeScriptRuntime/MemRuntime.cpp +++ b/tslang/lib/TypeScriptRuntime/MemRuntime.cpp @@ -4,7 +4,7 @@ // `_aligned_free` may release, so that pairing corrupts the CRT heap. Windows' own `malloc` is // already aligned enough for everything that asks, so the request is served from the ordinary // heap and `free` stays honest. (Under `-mm=gc` this never showed, because GCPass rewrites the -// whole pair to GC_memalign/GC_free.) +// whole pair to GC_malloc_uncollectable/GC_free.) #ifndef _WIN32 #if defined(__FreeBSD__) || defined(__NetBSD__) || defined(__OpenBSD__) diff --git a/tslang/lib/TypeScriptRuntime/TypeScriptGC.cpp b/tslang/lib/TypeScriptRuntime/TypeScriptGC.cpp index b1aba1402..77adc3a9f 100644 --- a/tslang/lib/TypeScriptRuntime/TypeScriptGC.cpp +++ b/tslang/lib/TypeScriptRuntime/TypeScriptGC.cpp @@ -15,6 +15,7 @@ void init_gcruntime(llvm::StringMap &exportSymbols) exportSymbol("GC_malloc", &_mlir__GC_malloc); exportSymbol("GC_malloc_atomic", &_mlir__GC_malloc_atomic); exportSymbol("GC_memalign", &_mlir__GC_memalign); + exportSymbol("GC_malloc_uncollectable", &_mlir__GC_malloc_uncollectable); exportSymbol("GC_realloc", &_mlir__GC_realloc); exportSymbol("GC_free", &_mlir__GC_free); exportSymbol("GC_get_heap_size", &_mlir__GC_get_heap_size); diff --git a/tslang/lib/TypeScriptRuntime/TypeScriptRuntime.def b/tslang/lib/TypeScriptRuntime/TypeScriptRuntime.def index 71311de5e..9fc69620e 100644 --- a/tslang/lib/TypeScriptRuntime/TypeScriptRuntime.def +++ b/tslang/lib/TypeScriptRuntime/TypeScriptRuntime.def @@ -12,6 +12,7 @@ EXPORTS GC_malloc=_mlir__GC_malloc GC_malloc_atomic=_mlir__GC_malloc_atomic GC_memalign=_mlir__GC_memalign + GC_malloc_uncollectable=_mlir__GC_malloc_uncollectable GC_realloc=_mlir__GC_realloc GC_free=_mlir__GC_free GC_get_heap_size=_mlir__GC_get_heap_size diff --git a/tslang/lib/TypeScriptRuntime/gc.cpp b/tslang/lib/TypeScriptRuntime/gc.cpp index 1f28e3e9b..cdf83e061 100644 --- a/tslang/lib/TypeScriptRuntime/gc.cpp +++ b/tslang/lib/TypeScriptRuntime/gc.cpp @@ -45,6 +45,11 @@ void *_mlir__GC_memalign(size_t align, size_t size) return GC_memalign(align, size); } +void *_mlir__GC_malloc_uncollectable(size_t size) +{ + return GC_MALLOC_UNCOLLECTABLE(size); +} + void *_mlir__GC_realloc(void *ptr, size_t size) { return GC_REALLOC(ptr, size); diff --git a/tslang/test/check-x86-run.sh b/tslang/test/check-x86-run.sh index 7076c95ce..2a07e76a0 100755 --- a/tslang/test/check-x86-run.sh +++ b/tslang/test/check-x86-run.sh @@ -209,8 +209,8 @@ done # Upstream ConvertAsyncToLLVM hardcodes the coroutine frame allocator's declaration to # aligned_alloc(i64, i64), whatever the target. aligned_alloc takes size_t, which is pointer # width, so at i686 the callee reads (alignment, size=0) and the frame overflows its block. -# The repair pass (task 2/3) must retype the declaration to the pointer width: GC_memalign is -# GCPass's rename of aligned_alloc when a collector is linked in. +# The repair pass (task 2/3) must retype the declaration to the pointer width. When a collector +# is linked in, GCPass rewrites each call to GC_malloc_uncollectable(size), which keeps that width. emit_await_order_ir() { local mm="$1" local out="$work/await_order.$mm.ll" err status @@ -232,12 +232,16 @@ declared_with_i32_params() { grep -Eq "^declare .*ptr @$1\(i32( [^,]*)?, i32( [^)]*)?\)" "$2" } +declared_with_one_i32_param() { + grep -Eq "^declare .*ptr @$1\(i32( [^,)]*)?\)" "$2" +} + if out="$(emit_await_order_ir gc)"; then - if declared_with_i32_params GC_memalign "$out"; then - echo "ok x86 IR -mm=gc: frame allocator is GC_memalign(i32, i32)" + if declared_with_one_i32_param GC_malloc_uncollectable "$out"; then + echo "ok x86 IR -mm=gc: frame allocator is GC_malloc_uncollectable(i32)" else - echo "FAIL x86 IR -mm=gc: frame allocator is GC_memalign(i32, i32)" - grep -m1 -E '@GC_memalign\(|@aligned_alloc\(' "$out" | sed 's/^/ got: /' + echo "FAIL x86 IR -mm=gc: frame allocator is GC_malloc_uncollectable(i32)" + grep -m1 -E '@GC_malloc_uncollectable\(|@GC_memalign\(|@aligned_alloc\(' "$out" | sed 's/^/ got: /' fail=1 fi fi diff --git a/tslang/test/tester/foreign-thread-gc-host.cpp b/tslang/test/tester/foreign-thread-gc-host.cpp index d6877066f..54e53a689 100644 --- a/tslang/test/tester/foreign-thread-gc-host.cpp +++ b/tslang/test/tester/foreign-thread-gc-host.cpp @@ -6,6 +6,12 @@ // // Calls work(seed) times on the loading thread, then on each of new threads, // and checks every result. Exits 0 when all are right. +// +// It also checks that the library enabled the collector's threads on the loading thread's calls, +// before any other thread came in (#523): enabling them starts the parallel markers and only then +// makes the allocator take its lock, so a thread that enables them races any thread already +// allocating. The markers running (GC_get_parallel) is what shows it. A collector with no parallel +// marking never starts them, and then there is nothing to check. #include #include @@ -19,6 +25,7 @@ #endif using WorkFn = double (*)(double); +using GetParallelFn = int (*)(); static WorkFn work; static int calls; @@ -65,6 +72,10 @@ int main(int argc, char **argv) } work = reinterpret_cast(GetProcAddress(library, "work")); + // the collector is gc.dll, which the library loaded + auto collector = GetModuleHandleA("gc.dll"); + auto getParallel = + collector ? reinterpret_cast(GetProcAddress(collector, "GC_get_parallel")) : nullptr; #else auto library = dlopen(argv[1], RTLD_NOW | RTLD_LOCAL); if (!library) @@ -74,6 +85,12 @@ int main(int argc, char **argv) } work = reinterpret_cast(dlsym(library, "work")); + // the collector is linked into the library, or loaded beside it + auto getParallel = reinterpret_cast(dlsym(library, "GC_get_parallel")); + if (!getParallel) + { + getParallel = reinterpret_cast(dlsym(RTLD_DEFAULT, "GC_get_parallel")); + } #endif if (!work) { @@ -83,6 +100,7 @@ int main(int argc, char **argv) run(0); std::printf("loading thread: %d calls\n", calls); + auto markersAfterLoadingThread = getParallel ? getParallel() : 0; std::vector pool; for (long id = 1; id <= threads; id++) @@ -96,5 +114,18 @@ int main(int argc, char **argv) } std::printf("%d other threads: %d calls each\n", threads, calls); + + if (!getParallel) + { + std::printf("no GC_get_parallel: not checked when the collector's threads were enabled\n"); + return 0; + } + + if (getParallel() > 0 && markersAfterLoadingThread == 0) + { + std::printf("the collector's threads were enabled only when another thread called in\n"); + return 3; + } + return 0; } diff --git a/tslang/test/tester/foreign-thread-gc.cmake b/tslang/test/tester/foreign-thread-gc.cmake index 5743747ad..c699f2126 100644 --- a/tslang/test/tester/foreign-thread-gc.cmake +++ b/tslang/test/tester/foreign-thread-gc.cmake @@ -16,6 +16,19 @@ # marker threads and waited for them; they could not start until the load finished, and the load # never did (the default library hung every JIT run). The allocations escape into globals, so the # optimizer cannot fold the calls away. +# +# The host also checks that the library enabled the collector's threads on its loading thread's +# calls, before another thread came in (#523): enabling them starts the markers before the +# allocator takes its lock, so a newcomer enabling them raced a registered thread's allocation. +# +# And with the work done on the library's own async pool (#522), which collects there. The pool's +# threads register themselves through hooks that only the library's GC_enable_threads sets (each +# module has its own copy of the scheduler); when only a registered thread called in, they never +# were, and the collection a pool thread started aborted ("Collecting from unknown thread"). With +# them registered, eight host threads awaiting at once crashed on Linux, or got another call's +# result: an async function's frame was collectable, and one waiting in the pool's queue - ordinary +# heap the collector does not scan - was freed by a collection (GCPass now allocates frames +# uncollectable; they are freed by hand when the coroutine finishes). cmake_minimum_required(VERSION 3.17.3) @@ -67,6 +80,36 @@ export function makeNode(value: number): Node { export let kept: Node = makeNode(1); ") +file(WRITE "${WORK_DIR}/async-pool.ts" [=[ +declare function GC_gcollect(): void; + +class Node { + constructor(public value: number, public next: Node | undefined) {} +} + +// runs on one of the library's pool threads, and now and then collects there +async function sumOnThePool(seed: number): number { + let head: Node | undefined = undefined; + for (let i = 0; i < 200; i++) { + head = new Node(seed + i, head); + } + + if (seed % 500 == 0) { + GC_gcollect(); + } + + let sum: number = 0; + for (let node = head; node !== undefined; node = node.next) { + sum += node.value; + } + + return sum; +} + +export function work(seed: number): number { + return await sumOnThePool(seed); +} +]=]) set(libs "--gc-lib-path=${GC_LIB}" "--tslang-lib-path=${TSLANG_LIB}" "--llvm-lib-path=${LLVM_LIB}") if(DEFINED GC_SHARED_LIB) @@ -84,7 +127,7 @@ else() set(pic "-relocation-model=pic") endif() -foreach(name with-top-level no-top-level static-constructor-at-load export-called-at-load) +foreach(name with-top-level no-top-level static-constructor-at-load export-called-at-load async-pool) set(library "${WORK_DIR}/${prefix}${name}${suffix}") execute_process(COMMAND "${TSLANG}" --emit=dll --opt -mm=gc --no-default-lib ${pic} ${libs} ${name}.ts -o "${library}" WORKING_DIRECTORY "${WORK_DIR}" diff --git a/tslang/tslang/transform.cpp b/tslang/tslang/transform.cpp index 2d6195061..3a8c1f8fb 100644 --- a/tslang/tslang/transform.cpp +++ b/tslang/tslang/transform.cpp @@ -182,7 +182,7 @@ int runMLIRPasses(mlir::MLIRContext &context, llvm::SourceMgr &sourceMgr, mlir:: { #ifdef ENABLE_ASYNC pm.addPass(mlir::createConvertAsyncToLLVMPass()); - // Before GCPass, which renames aligned_alloc to GC_memalign and so keeps the corrected + // Before GCPass, which rewrites aligned_alloc to GC_malloc_uncollectable and so keeps the corrected // signature; before LowerToLLVM, whose memref alloc lowering (AlignedAlloc) looks up // aligned_alloc with a pointer-width index and would reject the (i64, i64) declaration as a // redefinition.