This uses an SGThreadExclusive controlled by Emesary notifications that are received from the main loop. When active at the end of a frame the garbage collection thread will be released; if it is already running this will do nothing. Optionally at the start of the mainloop we can wait for the previous GC to finish. The actions of the background GC is controlled by notifications - again received from the main loop which in turn uses properties. I initially thought that the wait at the start of the frame would be necessary; however in 100 or so hours of flight without the await for completion at the start of frame no threading problems (or any other problems) were shown; so nasal-gc-threaded-wait is defaulted to false which gives a slight boost in performance. So what this does is to it removes the GC pause of 10-20ms every 4 seconds (test using the F-15). This change doesn't really give much extra performance per frame because normally GC is only performed when needed.
477 lines
12 KiB
C
477 lines
12 KiB
C
#include "nasal.h"
|
|
#include "data.h"
|
|
#include "code.h"
|
|
#define MIN_BLOCK_SIZE 32
|
|
|
|
static void reap(struct naPool* p);
|
|
static void mark(naRef r);
|
|
|
|
struct Block {
|
|
int size;
|
|
char* block;
|
|
struct Block* next;
|
|
};
|
|
// Must be called with the giant exclusive lock!
|
|
extern void global_stamp();
|
|
extern int global_elapsedUSec();
|
|
|
|
static int freeDead()
|
|
{
|
|
int i;
|
|
for(i=0; i<globals->ndead; i++)
|
|
naFree(globals->deadBlocks[i]);
|
|
globals->ndead = 0;
|
|
return i;
|
|
}
|
|
|
|
static void marktemps(struct Context* c)
|
|
{
|
|
int i;
|
|
naRef r = naNil();
|
|
for(i=0; i<c->ntemps; i++) {
|
|
SETPTR(r, c->temps[i]);
|
|
mark(r);
|
|
}
|
|
}
|
|
//#define GC_DETAIL_DEBUG
|
|
static int __elements_visited = 0;
|
|
static int gc_busy=0;
|
|
// Must be called with the big lock!
|
|
static void garbageCollect()
|
|
{
|
|
if (gc_busy)
|
|
return;
|
|
gc_busy = 1;
|
|
int i;
|
|
struct Context* c;
|
|
globals->allocCount = 0;
|
|
c = globals->allContexts;
|
|
|
|
#if GC_DETAIL_DEBUG
|
|
int ctxc = 0;
|
|
__elements_visited = 0;
|
|
int st = global_elapsedUSec();
|
|
int et = 0;
|
|
int stel = __elements_visited;
|
|
int eel = 0;
|
|
#endif
|
|
|
|
c = globals->allContexts;
|
|
while (c) {
|
|
#if GC_DETAIL_DEBUG
|
|
ctxc++;
|
|
#endif
|
|
for (i = 0; i < NUM_NASAL_TYPES; i++)
|
|
c->nfree[i] = 0;
|
|
for (i = 0; i < c->fTop; i++) {
|
|
mark(c->fStack[i].func);
|
|
mark(c->fStack[i].locals);
|
|
}
|
|
for (i = 0; i < c->opTop; i++)
|
|
mark(c->opStack[i]);
|
|
mark(c->dieArg);
|
|
marktemps(c);
|
|
c = c->nextAll;
|
|
}
|
|
#if GC_DETAIL_DEBUG
|
|
et = global_elapsedUSec() - st;
|
|
st = global_elapsedUSec();
|
|
eel = __elements_visited - stel; stel = __elements_visited;
|
|
printf("--> garbageCollect(#e%-5d): %-4d ", eel, et);
|
|
#endif
|
|
|
|
mark(globals->save);
|
|
#if GC_DETAIL_DEBUG
|
|
et = global_elapsedUSec() - st;
|
|
st = global_elapsedUSec();
|
|
eel = __elements_visited - stel; stel = __elements_visited;
|
|
printf("s(%5d) %-5d ", eel, et);
|
|
#endif
|
|
|
|
mark(globals->save_hash);
|
|
#if GC_DETAIL_DEBUG
|
|
et = global_elapsedUSec() - st;
|
|
st = global_elapsedUSec();
|
|
eel = __elements_visited - stel; stel = __elements_visited;
|
|
printf("h(%5d) %-5d ", eel, et);
|
|
#endif
|
|
|
|
|
|
mark(globals->symbols);
|
|
#if GC_DETAIL_DEBUG
|
|
et = global_elapsedUSec() - st;
|
|
st = global_elapsedUSec();
|
|
eel = __elements_visited - stel; stel = __elements_visited;
|
|
//printf("sy(%5d) %-4d ", eel, et);
|
|
#endif
|
|
|
|
mark(globals->meRef);
|
|
#if GC_DETAIL_DEBUG
|
|
et = global_elapsedUSec() - st;
|
|
st = global_elapsedUSec();
|
|
eel = __elements_visited - stel; stel = __elements_visited;
|
|
//printf("me(%5d) %-5d ", eel, et);
|
|
#endif
|
|
|
|
mark(globals->argRef);
|
|
#if GC_DETAIL_DEBUG
|
|
et = global_elapsedUSec() - st;
|
|
st = global_elapsedUSec();
|
|
eel = __elements_visited - stel; stel = __elements_visited;
|
|
//printf("ar(%5d) %-5d ", eel, et);
|
|
#endif
|
|
|
|
mark(globals->parentsRef);
|
|
#if GC_DETAIL_DEBUG
|
|
et = global_elapsedUSec() - st;
|
|
st = global_elapsedUSec();
|
|
eel = __elements_visited - stel; stel = __elements_visited;
|
|
#endif
|
|
//printf(" ev[%3d] %-5d", eel, et);
|
|
// Finally collect all the freed objects
|
|
for (i = 0; i < NUM_NASAL_TYPES; i++) {
|
|
reap(&(globals->pools[i]));
|
|
}
|
|
#if GC_DETAIL_DEBUG
|
|
et = global_elapsedUSec() - st;
|
|
st = global_elapsedUSec();
|
|
printf(" >> reap %-5d", et);
|
|
#endif
|
|
// Make enough space for the dead blocks we need to free during
|
|
// execution. This works out to 1 spot for every 2 live objects,
|
|
// which should be limit the number of bottleneck operations
|
|
// without imposing an undue burden of extra "freeable" memory.
|
|
if(globals->deadsz < globals->allocCount) {
|
|
globals->deadsz = globals->allocCount;
|
|
if(globals->deadsz < 256000) globals->deadsz = 256000;
|
|
naFree(globals->deadBlocks);
|
|
globals->deadBlocks = naAlloc(sizeof(void*) * globals->deadsz);
|
|
}
|
|
globals->needGC = 0;
|
|
#if GC_DETAIL_DEBUG
|
|
et = global_elapsedUSec() - st;
|
|
st = global_elapsedUSec();
|
|
printf(">> %-5d ", et);
|
|
#endif
|
|
gc_busy = 0;
|
|
}
|
|
|
|
void naModLock()
|
|
{
|
|
LOCK();
|
|
globals->nThreads++;
|
|
UNLOCK();
|
|
naCheckBottleneck();
|
|
}
|
|
|
|
void naModUnlock()
|
|
{
|
|
LOCK();
|
|
globals->nThreads--;
|
|
// We might be the "last" thread needed for collection. Since
|
|
// we're releasing our modlock to do something else for a while,
|
|
// wake someone else up to do it.
|
|
if(globals->waitCount == globals->nThreads)
|
|
naSemUp(globals->sem, 1);
|
|
UNLOCK();
|
|
}
|
|
|
|
// Must be called with the main lock. Engages the "bottleneck", where
|
|
// all threads will block so that one (the last one to call this
|
|
// function) can run alone. This is done for GC, and also to free the
|
|
// list of "dead" blocks when it gets full (which is part of GC, if
|
|
// you think about it).
|
|
static void bottleneck()
|
|
{
|
|
global_stamp();
|
|
struct Globals* g = globals;
|
|
g->bottleneck = 1;
|
|
while(g->bottleneck && g->waitCount < g->nThreads - 1) {
|
|
g->waitCount++;
|
|
UNLOCK(); naSemDown(g->sem); LOCK();
|
|
g->waitCount--;
|
|
}
|
|
#if GC_DETAIL_DEBUG
|
|
printf("GC: wait %2d ", global_elapsedUSec());
|
|
#endif
|
|
if(g->waitCount >= g->nThreads - 1) {
|
|
int fd = freeDead();
|
|
#if GC_DETAIL_DEBUG
|
|
printf("--> freedead (%5d) : %5d", fd, global_elapsedUSec());
|
|
#endif
|
|
if(g->needGC)
|
|
garbageCollect();
|
|
if(g->waitCount) naSemUp(g->sem, g->waitCount);
|
|
g->bottleneck = 0;
|
|
}
|
|
#if GC_DETAIL_DEBUG
|
|
printf(" :: finished: %5d\n", global_elapsedUSec());
|
|
#endif
|
|
}
|
|
|
|
static void bottleneckFreeDead()
|
|
{
|
|
global_stamp();
|
|
struct Globals* g = globals;
|
|
g->bottleneck = 1;
|
|
while (g->bottleneck && g->waitCount < g->nThreads - 1) {
|
|
g->waitCount++;
|
|
UNLOCK(); naSemDown(g->sem); LOCK();
|
|
g->waitCount--;
|
|
}
|
|
if (g->waitCount >= g->nThreads - 1) {
|
|
freeDead();
|
|
if (g->waitCount) naSemUp(g->sem, g->waitCount);
|
|
g->bottleneck = 0;
|
|
}
|
|
}
|
|
|
|
void naGC()
|
|
{
|
|
LOCK();
|
|
globals->needGC = 1;
|
|
bottleneck();
|
|
UNLOCK();
|
|
naCheckBottleneck();
|
|
}
|
|
int naGarbageCollect()
|
|
{
|
|
int rv = 1;
|
|
LOCK();
|
|
//
|
|
// The number here is again based on observation - if this is too low then the inline GC will be used
|
|
// which is fine occasionally.
|
|
// So what we're doing by checking the global alloc is to see if GC is likely required during the next frame and if
|
|
// so we pre-empt this by doing it now.
|
|
// GC can typically take between 5ms and 50ms (F-15, FG1000 PFD & MFD, Advanced weather) - but usually it is completed
|
|
// prior to the start of the next frame.
|
|
|
|
globals->needGC = nasal_globals->allocCount < 23000;
|
|
if (globals->needGC)
|
|
bottleneck();
|
|
else {
|
|
bottleneckFreeDead();
|
|
rv = 0;
|
|
}
|
|
UNLOCK();
|
|
naCheckBottleneck();
|
|
return rv;
|
|
}
|
|
|
|
void naCheckBottleneck()
|
|
{
|
|
if(globals->bottleneck) { LOCK(); bottleneck(); UNLOCK(); }
|
|
}
|
|
|
|
static void naCode_gcclean(struct naCode* o)
|
|
{
|
|
naFree(o->constants); o->constants = 0;
|
|
}
|
|
|
|
static void naCCode_gcclean(struct naCCode* c)
|
|
{
|
|
if(c->fptru && c->user_data && c->destroy) c->destroy(c->user_data);
|
|
c->user_data = 0;
|
|
}
|
|
|
|
static void naGhost_gcclean(struct naGhost* g)
|
|
{
|
|
if(g->ptr && g->gtype->destroy) g->gtype->destroy(g->ptr);
|
|
g->ptr = 0;
|
|
}
|
|
|
|
static void freeelem(struct naPool* p, struct naObj* o)
|
|
{
|
|
// Clean up any intrinsic storage the object might have...
|
|
switch(p->type) {
|
|
case T_STR: naStr_gcclean ((struct naStr*) o); break;
|
|
case T_VEC: naVec_gcclean ((struct naVec*) o); break;
|
|
case T_HASH: naiGCHashClean ((struct naHash*) o); break;
|
|
case T_CODE: naCode_gcclean ((struct naCode*) o); break;
|
|
case T_CCODE: naCCode_gcclean((struct naCCode*)o); break;
|
|
case T_GHOST: naGhost_gcclean((struct naGhost*)o); break;
|
|
}
|
|
p->free[p->nfree++] = o; // ...and add it to the free list
|
|
}
|
|
|
|
static void newBlock(struct naPool* p, int need)
|
|
{
|
|
int i;
|
|
struct Block* newb;
|
|
|
|
if(need < MIN_BLOCK_SIZE) need = MIN_BLOCK_SIZE;
|
|
|
|
newb = naAlloc(sizeof(struct Block));
|
|
newb->block = naAlloc(need * p->elemsz);
|
|
newb->size = need;
|
|
newb->next = p->blocks;
|
|
p->blocks = newb;
|
|
naBZero(newb->block, need * p->elemsz);
|
|
|
|
if(need > p->freesz - p->freetop) need = p->freesz - p->freetop;
|
|
p->nfree = 0;
|
|
p->free = p->free0 + p->freetop;
|
|
for(i=0; i < need; i++) {
|
|
struct naObj* o = (struct naObj*)(newb->block + i*p->elemsz);
|
|
o->mark = 0;
|
|
p->free[p->nfree++] = o;
|
|
}
|
|
p->freetop += need;
|
|
}
|
|
|
|
void naGC_init(struct naPool* p, int type)
|
|
{
|
|
p->type = type;
|
|
p->elemsz = naTypeSize(type);
|
|
p->blocks = 0;
|
|
|
|
p->free0 = p->free = 0;
|
|
p->nfree = p->freesz = p->freetop = 0;
|
|
reap(p);
|
|
}
|
|
|
|
static int poolsize(struct naPool* p)
|
|
{
|
|
int total = 0;
|
|
struct Block* b = p->blocks;
|
|
while(b) { total += b->size; b = b->next; }
|
|
return total;
|
|
}
|
|
int GCglobalAlloc() {
|
|
return globals->allocCount;
|
|
}
|
|
struct naObj** naGC_get(struct naPool* p, int n, int* nout)
|
|
{
|
|
struct naObj** result;
|
|
naCheckBottleneck();
|
|
LOCK();
|
|
while(globals->allocCount < 0 || (p->nfree == 0 && p->freetop >= p->freesz)) {
|
|
globals->needGC = 1;
|
|
#if GC_DETAIL_DEBUG
|
|
printf("++");
|
|
#endif
|
|
bottleneck();
|
|
}
|
|
if(p->nfree == 0)
|
|
newBlock(p, poolsize(p)/8);
|
|
n = p->nfree < n ? p->nfree : n;
|
|
*nout = n;
|
|
p->nfree -= n;
|
|
globals->allocCount -= n;
|
|
result = (struct naObj**)(p->free + p->nfree);
|
|
UNLOCK();
|
|
return result;
|
|
}
|
|
|
|
static void markvec(naRef r)
|
|
{
|
|
int i;
|
|
struct VecRec* vr = PTR(r).vec->rec;
|
|
if(!vr) return;
|
|
for(i=0; i<vr->size; i++)
|
|
mark(vr->array[i]);
|
|
}
|
|
|
|
// Sets the reference bit on the object, and recursively on all
|
|
// objects reachable from it. Uses the processor stack for recursion...
|
|
static void mark(naRef r)
|
|
{
|
|
int i;
|
|
|
|
if(IS_NUM(r) || IS_NIL(r))
|
|
return;
|
|
|
|
if(PTR(r).obj->mark == 1)
|
|
return;
|
|
__elements_visited++;
|
|
PTR(r).obj->mark = 1;
|
|
switch(PTR(r).obj->type) {
|
|
case T_VEC: markvec(r); break;
|
|
case T_HASH: naiGCMarkHash(r); break;
|
|
case T_CODE:
|
|
mark(PTR(r).code->srcFile);
|
|
for(i=0; i<PTR(r).code->nConstants; i++)
|
|
mark(PTR(r).code->constants[i]);
|
|
break;
|
|
case T_FUNC:
|
|
mark(PTR(r).func->code);
|
|
mark(PTR(r).func->namespace);
|
|
mark(PTR(r).func->next);
|
|
break;
|
|
case T_GHOST:
|
|
mark(PTR(r).ghost->data);
|
|
break;
|
|
}
|
|
}
|
|
|
|
void naiGCMark(naRef r)
|
|
{
|
|
mark(r);
|
|
}
|
|
|
|
// Collects all the unreachable objects into a free list, and
|
|
// allocates more space if needed.
|
|
static void reap(struct naPool* p)
|
|
{
|
|
struct Block* b;
|
|
int elem, freesz, total = poolsize(p);
|
|
freesz = total < MIN_BLOCK_SIZE ? MIN_BLOCK_SIZE : total;
|
|
freesz = (3 * freesz / 2) + (globals->nThreads * OBJ_CACHE_SZ);
|
|
if(p->freesz < freesz) {
|
|
naFree(p->free0);
|
|
p->freesz = freesz;
|
|
p->free = p->free0 = naAlloc(sizeof(void*) * p->freesz);
|
|
}
|
|
|
|
p->nfree = 0;
|
|
p->free = p->free0;
|
|
|
|
for(b = p->blocks; b; b = b->next)
|
|
for(elem=0; elem < b->size; elem++) {
|
|
struct naObj* o = (struct naObj*)(b->block + elem * p->elemsz);
|
|
if(o->mark == 0)
|
|
freeelem(p, o);
|
|
o->mark = 0;
|
|
}
|
|
|
|
p->freetop = p->nfree;
|
|
|
|
// allocs of this type until the next collection
|
|
globals->allocCount += total/2;
|
|
|
|
// Allocate more if necessary (try to keep 25-50% of the objects
|
|
// available)
|
|
// This was changed (2019.2) to allocate in larger blocks
|
|
// previously it used total/4 and used/2 now we
|
|
// use total/2 and used / 1
|
|
if (p->nfree < total / 2) {
|
|
int used = total - p->nfree;
|
|
int avail = total - used;
|
|
int need = used / 1 - avail;
|
|
if (need > 0)
|
|
newBlock(p, need);
|
|
}
|
|
}
|
|
|
|
// Does the swap, returning the old value
|
|
static void* doswap(void** target, void* val)
|
|
{
|
|
void* old = *target;
|
|
*target = val;
|
|
return old;
|
|
}
|
|
|
|
// Atomically replaces target with a new pointer, and adds the old one
|
|
// to the list of blocks to free the next time something holds the
|
|
// giant lock.
|
|
void naGC_swapfree(void** target, void* val)
|
|
{
|
|
void* old;
|
|
LOCK();
|
|
old = doswap(target, val);
|
|
while(globals->ndead >= globals->deadsz)
|
|
bottleneck();
|
|
globals->deadBlocks[globals->ndead++] = old;
|
|
UNLOCK();
|
|
}
|