Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions tslang/include/TypeScript/gcwrapper.h
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,8 @@ void *_mlir__GC_malloc_atomic(size_t size);

void *_mlir__GC_memalign(size_t align, size_t size);

void *_mlir__GC_malloc_uncollectable(size_t size);

void *_mlir__GC_realloc(void *ptr, size_t size);

void _mlir__GC_free(void *ptr);
Expand Down
102 changes: 83 additions & 19 deletions tslang/lib/AsyncGCThreadsCommon.inc
Original file line number Diff line number Diff line change
Expand Up @@ -11,15 +11,41 @@

#include "TypeScript/AsyncGCThreads.h"

#include <atomic>
#include <mutex>

#ifndef _WIN32
#ifdef _WIN32
#include <windows.h>
#include <winternl.h>
#else
#include <pthread.h>
#endif

namespace
{

// Whether this thread is inside a DllMain - loading or unloading some DLL - where it holds the
// loader lock, and a thread it starts cannot run until the lock is released. The lock is the
// PEB's LoaderLock, which winternl.h leaves out of its PEB; its offset has stayed the same since
// Windows NT 4. Off Windows there is no such lock: a shared object's constructors run under
// dlopen's lock, which a new thread does not need to start.
bool holdsLoaderLock()
{
#ifdef _WIN32
#ifdef _WIN64
constexpr size_t loaderLockOffset = 0x110;
#else
constexpr size_t loaderLockOffset = 0xA0;
#endif
auto peb = reinterpret_cast<const char *>(NtCurrentTeb()->ProcessEnvironmentBlock);
auto loaderLock = *reinterpret_cast<RTL_CRITICAL_SECTION *const *>(peb + loaderLockOffset);
return loaderLock != nullptr &&
reinterpret_cast<ULONG_PTR>(loaderLock->OwningThread) == static_cast<ULONG_PTR>(GetCurrentThreadId());
#else
return false;
#endif
}

bool registerThisThread()
{
struct GC_stack_base sb;
Expand All @@ -32,8 +58,8 @@ void unregisterThisThread()
}

// A thread that came into a gc shared library through one of its exported functions, from a host
// that is not a tslang program (an Android app's JNI threads, a C program's): checked once, and
// registered then if the collector did not know it, until the thread ends.
// that is not a tslang program (an Android app's JNI threads, a C program's): checked once threads
// are enabled, and registered then if the collector did not know it, until the thread ends.
struct CallingThread
{
bool checked = false;
Expand All @@ -52,12 +78,47 @@ struct CallingThread
thread_local CallingThread callingThread;

std::once_flag collectorInitialized;
std::once_flag threadsEnabled;
std::once_flag threadsEnabledOnce;
std::atomic<bool> threadsEnabled{false};

} // namespace

extern "C" void GC_enable_threads();

namespace
{

// Runs GC_enable_threads once, unless now is no time for it: false then, and a later call does.
//
// Not inside a DllMain: GC_allow_register_threads starts the collector's markers and waits for
// them to come up, which they cannot do until the loader lock is released, so the load never
// finished. An exported function runs there when the library calls one while it loads - a static
// constructor of an exported class, top-level code calling an export (the default library hung
// every JIT run). Nor does such a call wait for another thread that is enabling them: that one
// waits for the markers, and they for this thread's loader lock. The markers running already
// (GC_get_parallel) would not tell it either: the count is set before they are waited for.
bool enableThreads()
{
if (threadsEnabled.load(std::memory_order_acquire))
{
return true;
}

if (holdsLoaderLock())
{
return false;
}

std::call_once(threadsEnabledOnce, [] {
GC_enable_threads();
threadsEnabled.store(true, std::memory_order_release);
});

return true;
}

} // namespace

// Called first thing by every exported function of a `-mm=gc` shared library (the GC pass injects
// it). The collector stops and scans only the threads registered with it, and aborts a collection
// that a thread it does not know started ("Collecting from unknown thread"): a tslang program
Expand All @@ -67,15 +128,18 @@ extern "C" void GC_enable_threads();
// The collector may well be initialized already: it initializes itself on first use, so a library
// with no top-level code (no GC_init injected at load) can find it running.
//
// GC_enable_threads runs only for a thread that has to be registered: GC_register_my_thread
// refuses every thread until GC_allow_register_threads was called ("Threads explicit registering
// is not previously enabled"). A registered thread must not run it here. An exported function can
// be called while the library is being loaded - a static constructor of an exported class, or
// top-level code calling an export - and on Windows that is inside DllMain, under the loader lock.
// GC_allow_register_threads starts the parallel marker threads and waits for them to come up, and
// a new thread cannot start until the loader lock is released: the load never finished (the
// default library hung every JIT run). The thread a library is loaded on is the one GC_init
// registered, or one its host registered.
// The first call that may enables threads (enableThreads), whichever thread makes it - before it
// goes on, and before any other thread can come in:
// - GC_register_my_thread refuses every thread until GC_allow_register_threads was called
// ("Threads explicit registering is not previously enabled");
// - GC_allow_register_threads makes the allocator take its lock only after it started the markers,
// so a newcomer enabling threads raced a thread that was already allocating (#523);
// - the library's own async pool registers its threads through hooks only the library's
// GC_enable_threads sets - each module has its own copy of the scheduler - so they must be set
// even when every caller is registered already (#522).
// While the library loads, the call cannot enable them (see enableThreads) and the thread goes on
// as it is: the one loading is the one GC_init registered, or one its host registered, and no other
// can call in before the load finishes. Its next call, after the load, enables them.
extern "C" void __tslang_gc_enter()
{
auto &thread = callingThread;
Expand All @@ -84,23 +148,23 @@ extern "C" void __tslang_gc_enter()
return;
}

thread.checked = true;

std::call_once(collectorInitialized, [] {
if (!GC_is_init_called())
{
GC_init();
}
});

if (GC_thread_is_registered())
if (!enableThreads())
{
return;
}

std::call_once(threadsEnabled, [] { GC_enable_threads(); });

thread.registeredHere = registerThisThread();
thread.checked = true;
if (!GC_thread_is_registered())
{
thread.registeredHere = registerThisThread();
}
}

// Called once from the entry point: the GC pass injects the call beside GC_init, so it happens
Expand Down
4 changes: 2 additions & 2 deletions tslang/lib/TypeScript/AsyncTargetWidthPass.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -46,8 +46,8 @@ class AsyncTargetWidthPass : public mlir::PassWrapper<AsyncTargetWidthPass, Modu
// overflows its block; at wasm32 the call does not match the definition's signature. Retype the declaration to the pointer width and truncate the arguments at each call.
// The values are a coroutine frame's size and alignment, which fit in 32 bits.
//
// Runs directly after ConvertAsyncToLLVM and before GCPass, which renames the declaration
// (and its calls) to GC_memalign and so keeps the corrected signature. Under the other memory
// Runs directly after ConvertAsyncToLLVM and before GCPass, which rewrites the calls to
// GC_malloc_uncollectable(size) and so keeps the corrected size type. Under the other memory
// models the name resolves to the runtime's aligned_alloc shim, which is (size_t, size_t) too.
mlir::LogicalResult fixFrameAllocator(mlir::ModuleOp m)
{
Expand Down
76 changes: 66 additions & 10 deletions tslang/lib/TypeScript/GCPass.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,8 @@ namespace

// what LLVMCodeHelperBase::_MemoryAlloc asks for when it wants a zeroed block
constexpr auto CALLOC_NAME = "calloc";
// what the coroutine lowering (ConvertAsyncToLLVM) allocates an async function's frame with
constexpr auto ALIGNED_ALLOC_NAME = "aligned_alloc";

class GCPass : public mlir::PassWrapper<GCPass, ModulePass>
{
Expand All @@ -46,6 +48,8 @@ class GCPass : public mlir::PassWrapper<GCPass, ModulePass>
llvm::SmallVector<LLVM::MemsetOp> redundantMemSets;
llvm::SmallVector<LLVM::CallOp> callocCalls;
llvm::SmallVector<LLVM::LLVMFuncOp> callocDecls;
llvm::SmallVector<LLVM::CallOp> frameAllocCalls;
llvm::SmallVector<LLVM::LLVMFuncOp> frameAllocDecls;
m.walk([&](mlir::Operation *op) {
// process gctors first
if (auto funcOp = dyn_cast_or_null<LLVM::LLVMFuncOp>(op))
Expand All @@ -63,6 +67,12 @@ class GCPass : public mlir::PassWrapper<GCPass, ModulePass>
return;
}

if (name == ALIGNED_ALLOC_NAME)
{
frameAllocDecls.push_back(funcOp);
return;
}

if (!funcOp.getBody().empty())
{
if (!added)
Expand Down Expand Up @@ -106,6 +116,12 @@ class GCPass : public mlir::PassWrapper<GCPass, ModulePass>
return;
}

if (name == ALIGNED_ALLOC_NAME)
{
frameAllocCalls.push_back(callOp);
return;
}

renameCall(name, callOp);
}
});
Expand All @@ -116,6 +132,7 @@ class GCPass : public mlir::PassWrapper<GCPass, ModulePass>
}

replaceCallocWithGCMalloc(m, callocCalls, callocDecls);
replaceFrameAllocWithUncollectable(m, frameAllocCalls, frameAllocDecls);

if (!added)
{
Expand All @@ -138,26 +155,23 @@ class GCPass : public mlir::PassWrapper<GCPass, ModulePass>
LLVM_DEBUG(llvm::dbgs() << "\n!! GCPass: AFTER DUMP: \n" << m << "\n";);
}

// `calloc` is deliberately absent here: it takes two arguments where GC_malloc takes one, so
// a rename in place would leave a call whose arity disagrees with its callee. It goes through
// replaceCallocWithGCMalloc instead.
// `calloc` and `aligned_alloc` are deliberately absent here: they take two arguments where
// GC_malloc and GC_malloc_uncollectable take one, so a rename in place would leave a call
// whose arity disagrees with its callee. They go through replaceCallocWithGCMalloc and
// replaceFrameAllocWithUncollectable instead.
bool mapName(StringRef name, StringRef modeName, StringRef &newName)
{
if (name == "malloc")
{
if (modeName == "atomic")
{
newName = "GC_malloc_atomic";
newName = "GC_malloc_atomic";
}
else
{
newName = "GC_malloc";
}
}
else if (name == "aligned_alloc")
{
newName = "GC_memalign";
}
else if (name == "realloc")
{
newName = "GC_realloc";
Expand Down Expand Up @@ -206,7 +220,8 @@ class GCPass : public mlir::PassWrapper<GCPass, ModulePass>
// each call is treated as returning a distinct, non-aliasing pointer.
void markAsAllocatorIfNeeded(StringRef newName, LLVM::LLVMFuncOp funcOp)
{
if (newName != "GC_malloc" && newName != "GC_malloc_atomic" && newName != "GC_memalign")
if (newName != "GC_malloc" && newName != "GC_malloc_atomic" && newName != "GC_memalign" &&
newName != "GC_malloc_uncollectable")
{
return;
}
Expand All @@ -222,7 +237,7 @@ class GCPass : public mlir::PassWrapper<GCPass, ModulePass>
LLVM::ModRefInfo::NoModRef, LLVM::ModRefInfo::NoModRef);
funcOp.setMemoryEffectsAttr(memoryEffects);

// AllocFnKind::Alloc = 1<<0, Zeroed = 1<<4. GC_malloc/GC_malloc_atomic zero-fill;
// AllocFnKind::Alloc = 1<<0, Zeroed = 1<<4. GC_malloc/GC_malloc_atomic/GC_malloc_uncollectable zero-fill;
// GC_memalign (GC_memalign) does not guarantee zeroing, so only mark Alloc for it.
uint64_t allocKind = newName == "GC_memalign" ? /*Alloc*/ 1 : /*Alloc|Zeroed*/ 1 | (1 << 4);
auto kindEntry = mlir::ArrayAttr::get(
Expand Down Expand Up @@ -318,6 +333,47 @@ class GCPass : public mlir::PassWrapper<GCPass, ModulePass>
}
}

// An async function's frame is the coroutine lowering's `aligned_alloc(align, size)`, given
// back with `free` (GC_free) when the coroutine finishes - its lifetime is managed by hand. It
// must not be collectable: a frame waiting to run is held only by the runtime's pool queue or
// a token's awaiter list, ordinary heap the collector does not scan. A collection then freed
// it, and the pool resumed whatever reused the block - a crash, or another call's result -
// as soon as several threads awaited at once. Uncollectable memory lives until it is freed
// and is still scanned, so what the frame holds across a suspension stays alive too.
//
// The alignment is dropped: the frame asks for 8 (see the runtime's aligned_alloc shim), and
// the collector aligns every object to its granule, two words.
void replaceFrameAllocWithUncollectable(mlir::ModuleOp module, llvm::SmallVector<LLVM::CallOp> &calls,
llvm::SmallVector<LLVM::LLVMFuncOp> &decls)
{
if (calls.empty() && decls.empty())
{
return;
}

PatternRewriter rewriter(module.getContext());
TypeHelper th(module.getContext());

for (auto callOp : calls)
{
LLVMCodeHelper ch(callOp, rewriter, nullptr, tsContext.compileOptions);
auto sizeValue = callOp.getOperand(1);
auto allocFuncOp = ch.getOrInsertFunction(
"GC_malloc_uncollectable",
th.getFunctionType(th.getPtrType(), mlir::ArrayRef<mlir::Type>{sizeValue.getType()}));
markAsAllocatorIfNeeded("GC_malloc_uncollectable", allocFuncOp);

rewriter.setInsertionPoint(callOp);
auto allocCall = rewriter.create<LLVM::CallOp>(callOp->getLoc(), allocFuncOp, ValueRange{sizeValue});
rewriter.replaceOp(callOp, allocCall.getResults());
}

for (auto funcOp : decls)
{
funcOp.erase();
}
}

static bool isExported(LLVM::LLVMFuncOp funcOp)
{
if (auto passthrough = funcOp->getAttrOfType<mlir::ArrayAttr>("passthrough"))
Expand Down
2 changes: 1 addition & 1 deletion tslang/lib/TypeScriptAsyncRuntime/AsyncRuntime.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,7 @@
// The coroutine lowering allocates an async function's frame by calling `aligned_alloc` by name
// and releases it with plain `free`. MSVC has no `aligned_alloc`, so an executable that awaits
// anything does not link at all unless the model rewrites the call (which only `-mm=gc` does,
// to GC_memalign). `_aligned_malloc` is not a stand-in: its memory may only go back through
// to GC_malloc_uncollectable). `_aligned_malloc` is not a stand-in: its memory may only go back through
// `_aligned_free`, and pairing it with `free` corrupts the CRT heap. Windows' `malloc` is
// already aligned enough for every request anything makes - the frame asks for 8 - so it serves
// the request and keeps the pairing honest.
Expand Down
2 changes: 1 addition & 1 deletion tslang/lib/TypeScriptRuntime/MemRuntime.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@
// `_aligned_free` may release, so that pairing corrupts the CRT heap. Windows' own `malloc` is
// already aligned enough for everything that asks, so the request is served from the ordinary
// heap and `free` stays honest. (Under `-mm=gc` this never showed, because GCPass rewrites the
// whole pair to GC_memalign/GC_free.)
// whole pair to GC_malloc_uncollectable/GC_free.)

#ifndef _WIN32
#if defined(__FreeBSD__) || defined(__NetBSD__) || defined(__OpenBSD__)
Expand Down
1 change: 1 addition & 0 deletions tslang/lib/TypeScriptRuntime/TypeScriptGC.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,7 @@ void init_gcruntime(llvm::StringMap<void *> &exportSymbols)
exportSymbol("GC_malloc", &_mlir__GC_malloc);
exportSymbol("GC_malloc_atomic", &_mlir__GC_malloc_atomic);
exportSymbol("GC_memalign", &_mlir__GC_memalign);
exportSymbol("GC_malloc_uncollectable", &_mlir__GC_malloc_uncollectable);
exportSymbol("GC_realloc", &_mlir__GC_realloc);
exportSymbol("GC_free", &_mlir__GC_free);
exportSymbol("GC_get_heap_size", &_mlir__GC_get_heap_size);
Expand Down
1 change: 1 addition & 0 deletions tslang/lib/TypeScriptRuntime/TypeScriptRuntime.def
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@ EXPORTS
GC_malloc=_mlir__GC_malloc
GC_malloc_atomic=_mlir__GC_malloc_atomic
GC_memalign=_mlir__GC_memalign
GC_malloc_uncollectable=_mlir__GC_malloc_uncollectable
GC_realloc=_mlir__GC_realloc
GC_free=_mlir__GC_free
GC_get_heap_size=_mlir__GC_get_heap_size
Expand Down
5 changes: 5 additions & 0 deletions tslang/lib/TypeScriptRuntime/gc.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,11 @@ void *_mlir__GC_memalign(size_t align, size_t size)
return GC_memalign(align, size);
}

void *_mlir__GC_malloc_uncollectable(size_t size)
{
return GC_MALLOC_UNCOLLECTABLE(size);
}

void *_mlir__GC_realloc(void *ptr, size_t size)
{
return GC_REALLOC(ptr, size);
Expand Down
Loading
Loading