diff --git a/Packages/ContinuityKit/Sources/Ingest/PreparationQueue+Stems.swift b/Packages/ContinuityKit/Sources/Ingest/PreparationQueue+Stems.swift index 8226163..c51a8a9 100644 --- a/Packages/ContinuityKit/Sources/Ingest/PreparationQueue+Stems.swift +++ b/Packages/ContinuityKit/Sources/Ingest/PreparationQueue+Stems.swift @@ -73,7 +73,12 @@ extension PreparationQueue { // don't spend CPU-minutes redoing them (reconcileStemLinks links them up next pass). if StemCache.hasStems(key: key) { await limiter.release() - await MainActor.run { _ = self?.stemsInFlight.remove(key) } + let queueDrained = await MainActor.run { () -> Bool in + guard let self else { return true } + self.stemsInFlight.remove(key) + return self.stemsInFlight.isEmpty + } + if queueDrained { OnnxStemSeparator.releaseSession() } return } do { @@ -98,7 +103,15 @@ extension PreparationQueue { // key we just wrote (it may not be in the protected set if the queue moved on). let protected = await MainActor.run { self?.protectedStemKeys ?? [] } StemCache.enforceBudget(protecting: protected.union([key])) - await MainActor.run { _ = self?.stemsInFlight.remove(key) } + let queueDrained = await MainActor.run { () -> Bool in + guard let self else { return true } + self.stemsInFlight.remove(key) + return self.stemsInFlight.isEmpty + } + // Last separation done: free the ORT session's weights instead of holding them + // under live playback for the rest of the process (jetsam margin on device). + // The next batch re-pays one model load — an offline job can afford that. + if queueDrained { OnnxStemSeparator.releaseSession() } } } } diff --git a/Packages/ContinuityKit/Sources/Ingest/StemSeparator.swift b/Packages/ContinuityKit/Sources/Ingest/StemSeparator.swift index 372aa50..f9eea9d 100644 --- a/Packages/ContinuityKit/Sources/Ingest/StemSeparator.swift +++ b/Packages/ContinuityKit/Sources/Ingest/StemSeparator.swift @@ -74,6 +74,16 @@ final class OnnxStemSeparator: StemSeparating { private static let sessionLock = NSLock() nonisolated(unsafe) private static var cachedSession: (path: String, session: ORTSession)? + /// Drops the cached session, returning its weights/arenas to the OS. Called when the + /// separation queue drains: the session is worth caching *between back-to-back tracks*, + /// but holding hundreds of MB of transformer weights under live playback + UI for the + /// rest of the process lifetime is exactly the jetsam margin long sessions die on. + static func releaseSession() { + sessionLock.lock() + defer { sessionLock.unlock() } + cachedSession = nil + } + private static func sharedSession(modelURL: URL) throws -> ORTSession { sessionLock.lock() defer { sessionLock.unlock() } @@ -86,6 +96,10 @@ final class OnnxStemSeparator: StemSeparating { // plenty for an offline cache-once job. try options.setIntraOpNumThreads(1) try options.addConfigEntry(withKey: "session.intra_op.allow_spinning", value: "0") + // Prepacking rewrites weights into kernel-friendly layouts as ADDITIONAL copies — + // for an 85M-param transformer that's hundreds of MB of duplicated RSS for a + // marginal speedup on an offline cache-once job. Keeping the originals only. + try options.addConfigEntry(withKey: "session.disable_prepacking", value: "1") let session = try ORTSession(env: env, modelPath: modelURL.path, sessionOptions: options) cachedSession = (modelURL.path, session) return session