diff --git a/lib/system/yrc.nim b/lib/system/yrc.nim index 681fb1db94..38fbc18b51 100644 --- a/lib/system/yrc.nim +++ b/lib/system/yrc.nim @@ -25,7 +25,7 @@ # collection anywhere. A cell sits in at most one buffer, guarded by an # atomic test-and-set of inRootsFlag; buffered cells are forced live. # -# Up to MaxPar collections run CONCURRENTLY — with the mutators and with +# Up to one collection PER THREAD runs CONCURRENTLY — with the mutators and with # each other — each capturing a disjoint partition of the heap by CAS-ing # a claim tag into the rootIdx header word. gMergeLock covers only the # tag-slot claim and orphan adoption. All waiting (for a free slot, for a @@ -255,10 +255,6 @@ proc setLenZeroed[T](s: var RawSeq[T]; n: int) = s.len = n zeroMem(s.d, n *% sizeof(T)) -proc setLenUninit[T](s: var RawSeq[T]; n: int) = - if s.cap < n: resize(s, n) - s.len = n - type TraceEntry = object ## (slot, value) snapshot taken at trace time. The value is read exactly @@ -270,6 +266,7 @@ type TarjanFrame = object u: int32 # dense index of the cell this frame belongs to + ebase: int32 # edges.len when this cell's frame was pushed base: int # traceStack.len before this cell's trace ran CaptureRec = object @@ -281,7 +278,7 @@ type desc: PNimTypeV2 rcWord: int # rc word as captured lowlink: int32 - sccOf: int32 # -1 while the cell is on the Tarjan stack + selfRefs: int32 # self edges, folded out of the edge array SccRec = object ## per SCC of the condensation; the deadness pass reads all of these @@ -293,7 +290,6 @@ type deadIn: int # number of edges from dead SCCs memStart: int32 # offset into sccMembers crossOff: int32 # offset into crossTgt (cross edges by source) - crossCursor: int32 flags: uint8 CaptureBufs = object @@ -301,9 +297,19 @@ type ## threadvar) and persistent across collections, so that frequent ## small collections don't pay per-collection allocations recs: RawSeq[CaptureRec] + sccIdx: RawSeq[int32] # per captured cell: its SCC, -1 while the cell + # is on the Tarjan stack. Deliberately NOT a + # CaptureRec field: it is read once per captured + # EDGE, and a 4-byte stride keeps that array + # L1-resident where a 32-byte record stride would + # miss on almost every edge. tstack: RawSeq[int32] frames: RawSeq[TarjanFrame] - edges: RawSeq[int64] # (u shl 32) or v, dense indices + edges: RawSeq[int32] # PENDING out-edge targets (dense indices) of the + # SCCs still being built. An SCC's edges are + # classified and dropped the moment it is emitted + # (see `capture`), so this stays proportional to + # the DFS frontier, not to the captured graph. sccs: RawSeq[SccRec] sccMembers: RawSeq[int32] crossTgt: RawSeq[int32] @@ -336,10 +342,11 @@ proc trace(s: Cell; desc: PNimTypeV2; j: var GcEnv) {.inline.} = # The spare rootIdx header word (unused by YRC's root registration, which # relies on inRootsFlag) doubles as the capture claim: it packs the owning -# collection's tag with the cell's dense discovery index. Up to MaxPar -# collections run CONCURRENTLY, each capturing a disjoint partition of the -# heap: the first collection to CAS its tag into a cell owns it; everyone -# else treats the cell as an opaque survivor. Stale tags (from retired +# collection's tag with the cell's dense discovery index. Collections run +# CONCURRENTLY — one slot per collecting thread, see gParSlots — each +# capturing a disjoint partition of the heap: the first collection to CAS +# its tag into a cell owns it; everyone else treats the cell as an opaque +# survivor. Stale tags (from retired # collections) never need clearing, they are simply reclaimable. # # The word does double duty a second time: commit re-stamps proven-live @@ -350,13 +357,29 @@ proc trace(s: Cell; desc: PNimTypeV2; j: var GcEnv) {.inline.} = # collection. Roots always bypass the stamp: every death has a dec-witness # that gets registered, and registered cells are always scanned as roots. const - MaxPar {.intdefine.} = 8 # max concurrent collections + MaxPar {.intdefine.} = 256 + ## CAPACITY of the slot table, not a tuning knob: a hard ceiling on + ## concurrent collections, sized to stay out of the way (16 bytes of BSS + ## per entry, and scans only ever run over `gParSlots`, below). Raise it + ## only for programs with more than this many threads collecting AT ONCE. ParSlots = MaxPar var gMergeLock: Lock # protects the tag slots + orphaned roots gActiveTags: array[ParSlots, int64] # 0 = free slot gSlotPhase: array[ParSlots, int] # 0 idle, 1 capturing, 2 committing + gParSlots: int = 1 + ## How many slots are IN PLAY. Every scan below runs over this prefix + ## rather than over the whole table, and it is raised (under gMergeLock, + ## never lowered) exactly when a thread wants to collect and every slot + ## already in play is busy. It therefore converges on the number of + ## threads that actually collect CONCURRENTLY — one slot each — and then + ## stops: the parallelism is auto-tuned to the workload instead of being + ## a compile-time guess. A fixed bound throttled hard whenever the thread + ## count exceeded it, because the surplus threads did not just collect + ## later, they PARKED in `parkUntil(anySlotFree())` waiting for a slot + ## (32 threads against 8 slots: 6.3s vs 2.6s with a slot each). + ## Starting at 1 also means single-threaded programs scan one entry. gSoloCapture: int # a solo collection is in its capture phase gTagCounter: int64 gMyTag {.threadvar.}: int64 @@ -426,9 +449,17 @@ proc collectorEvent() {.inline.} = broadcast gWaitCond release gWaitLock +proc slotsInPlay(): int {.inline.} = + ## Must be loaded AFTER whatever claim word the caller is validating: a + ## slot is put in play before the collection owning it can tag any cell, + ## so a load ordered after reading a tagged word is guaranteed to cover + ## the slot that wrote the tag. `gParSlots` only ever grows, so a scan + ## over this prefix can never shrink under a reader. + atomicLoadN(addr gParSlots, ATOMIC_ACQUIRE) + proc anySlotFree(): bool {.inline.} = result = false - for sl in 0 ..< ParSlots: + for sl in 0 ..< slotsInPlay(): if atomicLoadN(addr gActiveTags[sl], ATOMIC_ACQUIRE) == 0: return true @@ -443,7 +474,7 @@ template denseIdx(c: Cell): int32 = proc isActiveTag(t: int64): bool {.inline.} = result = false if t != 0: - for s in 0 ..< ParSlots: + for s in 0 ..< slotsInPlay(): if atomicLoadN(addr gActiveTags[s], ATOMIC_ACQUIRE) == t: return true @@ -485,6 +516,25 @@ var ## draining this thread's stripe queue registers candidates here, and ## this thread's collections steal it as their slice — no lock, and ## collections keep the cache locality of thread-local data. + gPendingCells {.threadvar.}: CellSeq[Cell] + ## Cells this thread committed dead — slots nil'ed, references already + ## decremented — but has not handed back to the allocator yet, because a + ## capture that was in flight at commit time may still hold a stale + ## (slot, value) snapshot pointing at them. Released at the start of this + ## thread's next collection; see `releasePending`. + gPendingWatch {.threadvar.}: RawSeq[int64] + ## The captures that batch must outlive, packed as (slot shl 32) or tag. + gPendingSlot {.threadvar.}: int + gPendingActive {.threadvar.}: bool + gSpareRoots {.threadvar.}: CellSeq[Cell] + ## The buffer a finished collection hands back, so that stealing + ## `gLocalRoots` costs no allocation in steady state. + gTraceBuf {.threadvar.}: CellSeq[TraceEntry] + gFreeBuf {.threadvar.}: CellSeq[Cell] + ## Storage for `GcEnv.traceStack` / `GcEnv.toFree`, kept across + ## collections like the `gCap` arrays: both grow to the size of the + ## captured graph, so re-allocating and re-growing them per collection + ## was a memcpy of the whole trace frontier every time. stripes: array[NumStripes, Stripe] rootsThreshold: int = 128 # shared adaptive heuristic; races are benign defaultThreshold = when defined(nimFixedOrc): 10_000 else: 128 @@ -677,6 +727,62 @@ template orcAssert(cond, msg) = cfprintf(cstderr, "[Bug!] %s\n", msg) rawQuit 1 +proc graceSatisfied(): bool = + ## Has every capture recorded in `gPendingWatch` finished? A capture that + ## started later cannot hold a stale snapshot of the batch: it reads every + ## slot fresh, and the batch's cells are unreachable by then. + result = true + for i in 0 ..< gPendingWatch.len: + let w = gPendingWatch.d[i] + let s = int(w shr 32) + let tg = w and 0xFFFFFFFF'i64 + if atomicLoadN(addr gActiveTags[s], ATOMIC_ACQUIRE) == tg and + atomicLoadN(addr gSlotPhase[s], ATOMIC_ACQUIRE) == 1: + return false + +proc buildPendingWatch(): bool = + ## Record the collections that are in their CAPTURE phase right now: they + ## are the only ones that can hold a stale (slot, value) snapshot of the + ## cells we are about to release. `false` (nothing capturing) is the common + ## case below a handful of threads, and means the batch can be freed on the + ## spot with no deferral and no destructor-timing change at all. + if gPendingWatch.d == nil: init gPendingWatch + gPendingWatch.len = 0 + for s in 0 ..< slotsInPlay(): + if s != gMySlot: + let tg = atomicLoadN(addr gActiveTags[s], ATOMIC_ACQUIRE) + if tg != 0 and atomicLoadN(addr gSlotPhase[s], ATOMIC_ACQUIRE) == 1: + gPendingWatch.add((int64(s) shl 32) or tg) + result = gPendingWatch.len > 0 + +proc releasePending() = + ## Hand this thread's parked batch back to the allocator and give its slot + ## up. Called at the start of every collection, so by the time it runs the + ## watched captures have had a whole collection's worth of time to finish + ## and the wait below is virtually always already satisfied — that is the + ## whole point: the wait moved off the commit path, where it blocked BOTH + ## the committing collector and (because commit runs inside the GC fence) + ## every mutator doing a seq operation. + if not gPendingActive: return + parkUntil(graceSatisfied()) + # Give the slot up FIRST: the batch is unreachable and no capture can hold + # a snapshot of it any more, so the tag has nothing left to protect. + gPendingActive = false + gPendingWatch.len = 0 + atomicStoreN(addr gActiveTags[gPendingSlot], 0, ATOMIC_SEQ_CST) + collectorEvent() + # Destructors run here. `Collecting` is the existing re-entrancy guard: a + # destructor-driven dec that overflows a stripe must drain it, not start a + # nested collection that would write into the batch we are walking. + let prev = lockState + lockState = Collecting + for i in 0 ..< gPendingCells.len: + when orcLeakDetector: + writeCell("CYCLIC OBJECT FREED", gPendingCells.d[i][0], gPendingCells.d[i][1]) + free(gPendingCells.d[i][0], gPendingCells.d[i][1]) + gPendingCells.len = 0 + lockState = prev + proc nimTraceRef(q: pointer; desc: PNimTypeV2; env: pointer) {.compilerRtl, inl.} = let p = cast[ptr pointer](q) # read the slot exactly once: mutators may exchange it concurrently. @@ -698,6 +804,7 @@ proc nimTraceRefDyn(q: pointer; env: pointer) {.compilerRtl, inl.} = proc prepareCapture() = if gCap.recs.d == nil: init gCap.recs + init gCap.sccIdx init gCap.tstack init gCap.frames init gCap.edges @@ -709,11 +816,13 @@ proc prepareCapture() = init gCap.ages else: gCap.recs.len = 0 + gCap.sccIdx.len = 0 gCap.tstack.len = 0 gCap.frames.len = 0 gCap.edges.len = 0 gCap.sccs.len = 0 gCap.sccMembers.len = 0 + gCap.crossTgt.len = 0 gCap.crossPend.len = 0 gCap.prunedSrc.len = 0 gCap.ages.len = 0 @@ -722,16 +831,20 @@ proc prepareCapture() = # inRootsFlag between capture and commit, which must not look like a # mutation to the commit-time rc validation. proc claimCell(c: Cell; desc: PNimTypeV2; cap: ptr CaptureBufs; - pruneLive: bool): int32 = + pruneLive: bool; old0: int64): int32 = ## Dense index if this collection owns `c` (claiming and registering it ## if it was unclaimed), -1 if another ACTIVE collection owns it, or ## -2 if `pruneLive` and the cell was proven live in the current epoch ## (treat as an opaque live external, don't descend). + ## + ## `old0` is `c`'s claim word as the caller already read it (acquire): the + ## DFS reads it to test ownership, and re-reading it here would be a second + ## dependent load of the same cold header word on every traversed edge. if gAmSolo: # no other collection is (or can start) capturing: plain stores. # This recovers the sequential capture speed of the single-collector # design whenever collections do not actually overlap. - let old = c.rootIdx + let old = old0 if (old shr 32) == gMyTag: return int32(old and 0xFFFFFFFF) if pruneLive and (old shr 32) == (gMyEpochStamp shr 32) and @@ -744,18 +857,22 @@ proc claimCell(c: Cell; desc: PNimTypeV2; cap: ptr CaptureBufs; if old != 0: bumpStat gStatCapRepeat cap.recs.add CaptureRec(cell: c, desc: desc, rcWord: loadRc(c) and not rcMask, - lowlink: int32(idx), sccOf: -1'i32) + lowlink: int32(idx), selfRefs: 0'i32) + cap.sccIdx.add -1'i32 cap.ages.add int32(if isEpochStamp(old): min(stampAge(old), 1000) else: 0) cap.tstack.add int32(idx) return int32(idx) + var old = old0 while true: - var old = atomicLoadN(addr c.rootIdx, ATOMIC_ACQUIRE) if (old shr 32) == gMyTag: return int32(old and 0xFFFFFFFF) if pruneLive and (old shr 32) == (gMyEpochStamp shr 32) and stampAge(old) >= YrcPromoteAge: return -2 - if isActiveTag(old shr 32): + # an epoch stamp is never a tag (tags are allocated below `epochBase`), + # so the scan over the active-tag slots is skipped for the common + # "cell survived an earlier collection" word + if not isEpochStamp(old) and isActiveTag(old shr 32): return -1 let idx = cap.recs.len if atomicCompareExchangeN(addr c.rootIdx, addr old, @@ -766,38 +883,69 @@ proc claimCell(c: Cell; desc: PNimTypeV2; cap: ptr CaptureBufs; if old != 0: bumpStat gStatCapRepeat cap.recs.add CaptureRec(cell: c, desc: desc, rcWord: loadRc(c) and not rcMask, - lowlink: int32(idx), sccOf: -1'i32) + lowlink: int32(idx), selfRefs: 0'i32) + cap.sccIdx.add -1'i32 cap.ages.add int32(if isEpochStamp(old): min(stampAge(old), 1000) else: 0) cap.tstack.add int32(idx) return int32(idx) + # a failed CAS leaves the fresh claim word in `old`; loop with it proc capture(s: Cell; desc: PNimTypeV2; j: var GcEnv; cap: ptr CaptureBufs) = ## Iterative Tarjan SCC over everything reachable from `s`. A frame's ## pending out-edges are the traceStack entries above frame.base; a child ## pushes and drains its own segment above ours, so when the child's frame ## pops, the stack is back at our segment and we resume popping our edges. - if isStamped(s): return + ## + ## An SCC's out-edges are classified into internal/cross the moment the SCC + ## is emitted, not in a later pass over a whole-graph edge list. That works + ## because `cap.edges` is truncated back to a frame's `ebase` whenever that + ## frame's cell turns out to be an SCC root: edges of already-emitted + ## sub-SCCs are gone, so `edges[ebase(u) ..< len]` at u's emission holds + ## exactly the out-edges of u's SCC. Every one of them targets either a + ## member (u would not be an SCC root if a member pointed at a cell still + ## on the stack below u) or an SCC emitted earlier, so `sccIdx` is final + ## for all of them and the cross targets can be appended to `crossTgt` + ## contiguously — which makes `crossOff` a prefix offset for free. + let rootWord = atomicLoadN(addr s.rootIdx, ATOMIC_ACQUIRE) + if (rootWord shr 32) == gMyTag: return orcAssert(j.traceStack.len == 0, "capture: trace stack not empty") # roots never prune: a dec-witnessed suspicion overrides any epoch stamp - let root = claimCell(s, desc, cap, pruneLive = false) + let root = claimCell(s, desc, cap, pruneLive = false, old0 = rootWord) if root < 0: return # another active collection owns this candidate; it handles it trace(s, desc, j) - cap.frames.add TarjanFrame(u: root, base: 0) - while cap.frames.len > 0: - let u = cap.frames.d[cap.frames.len -% 1].u - let base = cap.frames.d[cap.frames.len -% 1].base + # The innermost frame is kept in `u`/`base` instead of being re-read from + # `cap.frames` on every iteration: the loop body runs once per captured + # EDGE, so reloading the frame there costs more than the frame stack itself. + # `cap.frames` therefore only holds the ANCESTORS of `u`. + var u = root + var base = 0 + var ebase = 0'i32 + while true: if j.traceStack.len > base: - let (entry, tdesc) = j.traceStack.pop() - let t = head(entry.val) - if isStamped(t): - let v = denseIdx(t) - cap.edges.add (int64(u) shl 32) or int64(v) - if cap.recs.d[v].sccOf < 0 and v < cap.recs.d[u].lowlink: - cap.recs.d[u].lowlink = v + # inlined pop: the stamped path never needs the entry's descriptor + let last = j.traceStack.len -% 1 + j.traceStack.len = last + let t = head(j.traceStack.d[last][0].val) + # one load of the target's claim word serves both the ownership test + # and the dense-index extraction + let cw = atomicLoadN(addr t.rootIdx, ATOMIC_ACQUIRE) + if (cw shr 32) == gMyTag: + let v = int32(cw and 0xFFFFFFFF) + if v == u: + # A self edge is internal by construction and its reference is + # already in `rcWord`, so it cancels out of `sumRefs - internal` + # exactly. Counting it here keeps it out of the edge array and out + # of the classification pass below. + inc cap.recs.d[u].selfRefs + else: + cap.edges.add v + if cap.sccIdx.d[v] < 0 and v < cap.recs.d[u].lowlink: + cap.recs.d[u].lowlink = v else: + let tdesc = j.traceStack.d[last][1] let childBase = j.traceStack.len - let v = claimCell(t, tdesc, cap, pruneLive = true) + let v = claimCell(t, tdesc, cap, pruneLive = true, old0 = cw) if v == -1: # cross-collection edge: the owner sees our reference in the rc # word and classifies the target live; we re-register it as a @@ -810,75 +958,72 @@ proc capture(s: Cell; desc: PNimTypeV2; j: var GcEnv; cap: ptr CaptureBufs) = when defined(nimOrcStats): bumpStat gStatCapPruned else: - cap.edges.add (int64(u) shl 32) or int64(v) + cap.edges.add v trace(t, tdesc, j) - cap.frames.add TarjanFrame(u: v, base: childBase) + cap.frames.add TarjanFrame(u: u, ebase: ebase, base: base) + u = v + ebase = int32(cap.edges.len) + base = childBase else: - cap.frames.len = cap.frames.len -% 1 - if cap.frames.len > 0: - let pu = cap.frames.d[cap.frames.len -% 1].u - if cap.recs.d[u].lowlink < cap.recs.d[pu].lowlink: - cap.recs.d[pu].lowlink = cap.recs.d[u].lowlink - if cap.recs.d[u].lowlink == u: + let lowU = cap.recs.d[u].lowlink + if lowU == u: # u is the root of an SCC: pop the members off the Tarjan stack + let sid = int32(j.nScc) let memStart = int32(cap.sccMembers.len) var sum = 0 while true: - let w = cap.tstack.pop() - cap.recs.d[w].sccOf = int32(j.nScc) - cap.sccMembers.add w - sum = sum +% (cap.recs.d[w].rcWord shr rcShift) +% 1 - if w == u: break - cap.sccs.add SccRec(sumRefs: sum, memStart: memStart) + let m = cap.tstack.pop() + cap.sccIdx.d[m] = sid + cap.sccMembers.add m + # `- selfRefs`: a self edge counts in both `sumRefs` and `internal` + # and was folded out of the edge array in the loop above + sum = sum +% (cap.recs.d[m].rcWord shr rcShift) +% 1 -% + cap.recs.d[m].selfRefs + if m == u: break + # classify this SCC's out-edges now that every target's SCC is final + let crossOff = int32(cap.crossTgt.len) + var internal = 0 + for i in ebase ..< int32(cap.edges.len): + let sv = cap.sccIdx.d[cap.edges.d[i]] + if sv == sid: inc internal + else: cap.crossTgt.add sv + cap.edges.len = ebase + cap.sccs.add SccRec(sumRefs: sum, internal: internal, + memStart: memStart, crossOff: crossOff) inc j.nScc + if cap.frames.len == 0: break + let pi = cap.frames.len -% 1 + cap.frames.len = pi + let pu = cap.frames.d[pi].u + if lowU < cap.recs.d[pu].lowlink: + cap.recs.d[pu].lowlink = lowU + u = pu + ebase = cap.frames.d[pi].ebase + base = cap.frames.d[pi].base # ---------------- phase 2: deadness, side arrays only ---------------- proc computeDeadness(j: var GcEnv; cap: ptr CaptureBufs) = let nScc = j.nScc - # append the sentinel record ([nScc]); capture left internal/deadIn/flags - # zero-initialized and memStart valid. crossOff is filled below as a prefix - # sum, so the sentinel closes the last SCC's member and cross-edge slices. - cap.sccs.add SccRec(memStart: int32(cap.sccMembers.len)) - # classify captured edges: internal to an SCC vs condensation cross edges - var nCross = 0 - for i in 0 ..< cap.edges.len: - let e = cap.edges.d[i] - let su = cap.recs.d[int32(e shr 32)].sccOf - let sv = cap.recs.d[int32(e and 0xFFFFFFFF'i64)].sccOf - if su == sv: - inc cap.sccs.d[su].internal - else: - inc cap.sccs.d[su].crossOff - inc nCross - var total = 0'i32 - for s in 0 ..< nScc: - let c = cap.sccs.d[s].crossOff - cap.sccs.d[s].crossOff = total - cap.sccs.d[s].crossCursor = total - total = total +% c - cap.sccs.d[nScc].crossOff = total - setLenUninit cap.crossTgt, nCross - for i in 0 ..< cap.edges.len: - let e = cap.edges.d[i] - let su = cap.recs.d[int32(e shr 32)].sccOf - let sv = cap.recs.d[int32(e and 0xFFFFFFFF'i64)].sccOf - if su != sv: - cap.crossTgt.d[cap.sccs.d[su].crossCursor] = sv - inc cap.sccs.d[su].crossCursor + # append the sentinel record ([nScc]) that closes the last SCC's member and + # cross-edge slices; capture left deadIn/flags zero-initialized and filled + # sumRefs/internal/memStart/crossOff in already, so the condensation is + # complete the moment the DFS ends. + cap.sccs.add SccRec(memStart: int32(cap.sccMembers.len), + crossOff: int32(cap.crossTgt.len)) # pruned out-edges taint the source SCC: pruning cannot cause a false # "dead" (an untraced target only ever ADDS unexplained external refs), # but a "live" verdict may lean on a stamp that went stale within the # epoch, so validate re-registers surviving pruned SCCs for i in 0 ..< cap.prunedSrc.len: - let s = cap.recs.d[cap.prunedSrc.d[i]].sccOf + let s = cap.sccIdx.d[cap.prunedSrc.d[i]] cap.sccs.d[s].flags = cap.sccs.d[s].flags or flagPruned # cells that stay registered as roots (partial collection) count as # externally referenced: the roots buffer itself points at them for mi in 0 ..< cap.sccMembers.len: let m = cap.sccMembers.d[mi] if (loadRc(cap.recs.d[m].cell) and inRootsFlag) != 0: - let s = cap.recs.d[m].sccOf + let s = cap.sccIdx.d[m] cap.sccs.d[s].flags = cap.sccs.d[s].flags or flagForcedLive # deadness over the condensation. Tarjan emits sinks first, so higher SCC # ids are sources and every cross edge goes from a higher id to a lower @@ -910,7 +1055,7 @@ proc markDirtyFromQueues(j: var GcEnv; cap: ptr CaptureBufs) = template taint(cp: Cell) = let c = cp if isStamped(c): - let s = cap.recs.d[denseIdx(c)].sccOf + let s = cap.sccIdx.d[denseIdx(c)] cap.sccs.d[s].flags = cap.sccs.d[s].flags or flagDirty for i in 0.. 0: let (entry, _) = j.traceStack.pop() entry.slot[] = nil - when orcLeakDetector: - writeCell("CYCLIC OBJECT FREED", cell, desc) - free(cell, desc) + holdOrFree(cell, desc, deferred) j.freed = cap.recs.len + if deferred: + gPendingSlot = gMySlot + gPendingActive = true else: - init j.toFree + if gFreeBuf.d == nil: init gFreeBuf + gFreeBuf.len = 0 + j.toFree = gFreeBuf for s in 0 ..< j.nScc: if (cap.sccs.d[s].flags and flagDead) != 0: for mi in cap.sccs.d[s].memStart ..< cap.sccs.d[s+1].memStart: @@ -1067,32 +1227,60 @@ proc commitDead(j: var GcEnv; cap: ptr CaptureBufs) = let (entry, tdesc) = j.traceStack.pop() let t = head(entry.val) entry.slot[] = nil - if not deadCell(t): + let tw = atomicLoadN(addr t.rootIdx, ATOMIC_RELAXED) + if not deadCell(tw): trialDec(t) # a stamped target was not analyzed by THIS collection, so # this dec may be the death blow: keep the cell examinable - if isEpochStamp(atomicLoadN(addr t.rootIdx, ATOMIC_RELAXED)): + if isEpochStamp(tw): registerLocal(t, tdesc) - # epoch-stamp what this collection PROVED live, carrying the cell's - # survival age: only cells that keep surviving get promoted to ages - # where captures prune them, so die-young data is never deferred. - # Demoted (dirty) and pruned SCCs stay unproven — leave their stale - # tags claimable. Our tag is still active, so no foreign claim can - # race these stores. + # epoch-stamp what this collection PROVED live, carrying the survival + # age: only cells that keep surviving get promoted to ages where + # captures prune them, so die-young data is never deferred. Demoted + # (dirty) and pruned SCCs stay unproven — leave their stale tags + # claimable. Our tag is still active, so no foreign claim can race + # these stores. + # + # The age is the SCC's, not the cell's: the whole SCC gets the age of + # its YOUNGEST member, so promotion is all-or-nothing. A per-cell age + # lets one member of an SCC promote ahead of its own SCC-mates; the + # next capture then prunes that INTERNAL edge, which taints the SCC as + # flagPruned, which stops it from ever being stamped again — freezing + # every member's age at its current value and re-tracing the whole + # structure on every collection from then on. Members age at different + # rates whenever a structure is built incrementally (a list appended to + # across several collections), so this is the common case, not a corner + # one. Taking the minimum can only delay a promotion, never hasten one, + # so it cannot widen the floating-garbage bound. for s in 0 ..< j.nScc: if (cap.sccs.d[s].flags and (flagDead or flagDirty or flagPruned)) == 0: - for mi in cap.sccs.d[s].memStart ..< cap.sccs.d[s+1].memStart: - let m = cap.sccMembers.d[mi] - atomicStoreN(addr cap.recs.d[m].cell.rootIdx, - gMyEpochStamp or int64(cap.ages.d[m] +% 1), - ATOMIC_RELAXED) - graceWait() - for i in 0 ..< j.toFree.len: - when orcLeakDetector: - writeCell("CYCLIC OBJECT FREED", j.toFree.d[i][0], j.toFree.d[i][1]) - free(j.toFree.d[i][0], j.toFree.d[i][1]) + let memStart = cap.sccs.d[s].memStart + let memEnd = cap.sccs.d[s+1].memStart + var age = high(int32) + for mi in memStart ..< memEnd: + let a = cap.ages.d[cap.sccMembers.d[mi]] + if a < age: age = a + let stamp = gMyEpochStamp or int64(age +% 1) + for mi in memStart ..< memEnd: + atomicStoreN(addr cap.recs.d[cap.sccMembers.d[mi]].cell.rootIdx, + stamp, ATOMIC_RELAXED) j.freed = j.toFree.len - deinit j.toFree + if j.toFree.len > 0 and buildPendingWatch(): + # park the whole batch by swapping buffers: the collection keeps the + # (now empty) buffer the previous batch used, so neither side allocates + let spare = gPendingCells + gPendingCells = j.toFree + j.toFree = spare + j.toFree.len = 0 + gPendingSlot = gMySlot + gPendingActive = true + else: + for i in 0 ..< j.toFree.len: + when orcLeakDetector: + writeCell("CYCLIC OBJECT FREED", j.toFree.d[i][0], j.toFree.d[i][1]) + free(j.toFree.d[i][0], j.toFree.d[i][1]) + j.toFree.len = 0 + gFreeBuf = j.toFree proc startCollection(minRoots, keepBelow: int; slice: var CellSeq[Cell]; wait: bool; drainAll = false): bool = @@ -1100,11 +1288,12 @@ proc startCollection(minRoots, keepBelow: int; slice: var CellSeq[Cell]; ## collect) and try to become a collector over THIS THREAD's candidates: ## claim a tag slot — the only step still under gMergeLock — and steal ## the thread-local buffer as this collection's slice, lock-free. When - ## there is enough work but all ParSlots collections are running, `wait` - ## decides between parking until a slot frees (backpressure for - ## overflowing mutators) and giving up. Either way the drain happened, + ## there is enough work but the slot table is full (more than MaxPar + ## threads collecting at once), `wait` decides between parking until a + ## slot frees and giving up. Either way the drain happened, ## so the caller's overflowing queue has room again. result = false + releasePending() # last collection's batch: its watch list is long clear if drainAll: drainAllStripes() else: drainStripe(getStripeIdx()) adoptOrphans() @@ -1112,15 +1301,23 @@ proc startCollection(minRoots, keepBelow: int; slice: var CellSeq[Cell]; mayRunCycleCollect(): acquire gMergeLock var slot = -1 - for sl in 0 ..< ParSlots: + let inPlay = gParSlots + for sl in 0 ..< inPlay: if atomicLoadN(addr gActiveTags[sl], ATOMIC_RELAXED) == 0: slot = sl break + if slot < 0 and inPlay < ParSlots: + # Every slot in play is busy and the table has room: widen the pool + # instead of throttling this thread. Publishing the wider bound before + # the tag lands in the new slot is what makes `slotsInPlay` safe. + slot = inPlay + atomicStoreN(addr gParSlots, inPlay +% 1, ATOMIC_SEQ_CST) if slot < 0: release gMergeLock if not wait: break - # backpressure: all ParSlots collections are running; park until one - # finishes (finishCollection broadcasts) instead of burning a core + # the table itself is full (more than MaxPar threads collecting at + # once): park until one finishes (finishCollection broadcasts) + # instead of burning a core parkUntil(anySlotFree()) drainStripe(getStripeIdx()) # the world moved while we waited adoptOrphans() @@ -1131,7 +1328,7 @@ proc startCollection(minRoots, keepBelow: int; slice: var CellSeq[Cell]; gMySlot = slot gMyEpochStamp = epochStamp(atomicLoadN(addr gEpoch, ATOMIC_RELAXED)) var othersActive = false - for sl in 0 ..< ParSlots: + for sl in 0 ..< gParSlots: if sl != slot and atomicLoadN(addr gActiveTags[sl], ATOMIC_RELAXED) != 0: othersActive = true gAmSolo = not othersActive @@ -1143,7 +1340,12 @@ proc startCollection(minRoots, keepBelow: int; slice: var CellSeq[Cell]; # our buffer, our slice: no lock needed if keepBelow == 0: slice = gLocalRoots # steal the whole buffer - init(gLocalRoots) + if gSpareRoots.d != nil: + gLocalRoots = gSpareRoots + gLocalRoots.len = 0 + gSpareRoots = default(CellSeq[Cell]) + else: + init(gLocalRoots) else: init(slice, max(gLocalRoots.len - keepBelow, 8)) for i in keepBelow ..< gLocalRoots.len: @@ -1155,8 +1357,16 @@ proc startCollection(minRoots, keepBelow: int; slice: var CellSeq[Cell]; proc finishCollection() = if atomicAddFetch(addr gCollectionCounter, 1, ATOMIC_RELAXED) mod YrcEpochLen == 0: discard atomicAddFetch(addr gEpoch, 1, ATOMIC_RELAXED) - atomicStoreN(addr gActiveTags[gMySlot], 0, ATOMIC_SEQ_CST) - atomicStoreN(addr gSlotPhase[gMySlot], 0, ATOMIC_RELEASE) + if gPendingActive: + # A batch is parked under this collection's tag. Clear only the PHASE — + # so nobody's grace check waits on us — and leave the tag in + # `gActiveTags`: it is what stops a foreign capture from claiming, and + # then freeing, a cell that is sitting in the batch. `releasePending` + # gives the slot back. + atomicStoreN(addr gSlotPhase[gMySlot], 0, ATOMIC_RELEASE) + else: + atomicStoreN(addr gActiveTags[gMySlot], 0, ATOMIC_SEQ_CST) + atomicStoreN(addr gSlotPhase[gMySlot], 0, ATOMIC_RELEASE) gMyTag = 0 gAmSolo = false collectorEvent() # wake backpressure and grace waiters @@ -1170,7 +1380,9 @@ proc collectCyclesImpl(j: var GcEnv; slice: var CellSeq[Cell]) = for i in countdown(last, 0): writeCell("root", slice.d[i][0], slice.d[i][1]) - init j.traceStack + if gTraceBuf.d == nil: init gTraceBuf + gTraceBuf.len = 0 + j.traceStack = gTraceBuf prepareCapture() let cap = addr gCap # hoist the TLS lookup out of the hot loops j.nScc = 0 @@ -1194,11 +1406,11 @@ proc collectCyclesImpl(j: var GcEnv; slice: var CellSeq[Cell]) = commitDead(j, cap) j.keepThreshold = j.freed == j.touched and j.touched > 0 - deinit j.traceStack + gTraceBuf = j.traceStack # hand the (possibly grown) buffer back proc runCollection(j: var GcEnv; slice: var CellSeq[Cell]) = ## Runs one collection over the stolen slice, concurrently with mutators - ## AND with up to ParSlots-1 other collections over disjoint partitions. + ## AND with the other collecting threads over disjoint partitions. yrcGcFenceEnter() # freeze seq structure mutations, not ref writes if not gAmSolo: # a solo collection claims with plain stores; nobody else may claim @@ -1210,7 +1422,10 @@ proc runCollection(j: var GcEnv; slice: var CellSeq[Cell]) = lockState = prev yrcGcFenceExit() finishCollection() - deinit slice + if gSpareRoots.d == nil and slice.d != nil: + gSpareRoots = slice # recycle it as the next steal's replacement + else: + deinit slice when defined(nimOrcStats): var freedCyclicObjects {.threadvar.}: int @@ -1283,6 +1498,7 @@ proc GC_runOrc* = if startCollection(1, 0, slice, wait = true, drainAll = true): var j: GcEnv runCollection(j, slice) + releasePending() # GC_fullCollect must not leave a batch parked # note: aborted SCCs and cross-collection targets legitimately leave # re-registered roots behind; other RUNNING threads' local candidates # are theirs to collect (exiting threads spill to the orphan buffer) @@ -1315,6 +1531,7 @@ proc nimYrcThreadTeardown() = ## of ours is stranded in a queue no other thread hashes to, then spill ## our candidate buffer to the global orphan buffer, where the next ## collection on any thread adopts it. + releasePending() # nobody else can release this thread's parked batch drainStripe(getStripeIdx()) if gLocalRoots.len > 0: acquire gMergeLock @@ -1326,6 +1543,11 @@ proc nimYrcThreadTeardown() = deinit(gLocalRoots) gLocalRoots.d = nil gLocalRoots.len = 0 + deinit(gSpareRoots) + deinit(gTraceBuf) + deinit(gFreeBuf) + deinit(gPendingCells) + deinit(gPendingWatch) proc GC_enableMarkAndSweep*() = GC_enableOrc() proc GC_disableMarkAndSweep*() = GC_disableOrc()