diff --git a/Cotabby.xcodeproj/project.pbxproj b/Cotabby.xcodeproj/project.pbxproj index ddd90ac8..51457023 100644 --- a/Cotabby.xcodeproj/project.pbxproj +++ b/Cotabby.xcodeproj/project.pbxproj @@ -365,6 +365,7 @@ 5D6BF62973E33472E20E82B5 /* SuggestionCoordinator+Lifecycle.swift in Sources */ = {isa = PBXBuildFile; fileRef = 7814C85DA992C8124474FD81 /* SuggestionCoordinator+Lifecycle.swift */; }; 5D888AF36EE5B86B99F99987 /* ModelDownloadTransferControls.swift in Sources */ = {isa = PBXBuildFile; fileRef = A25B3A7C488A6D863FE56113 /* ModelDownloadTransferControls.swift */; }; 5E1CE9C9062D2DEB4B96FD5C /* PerDomainDisableSettings.swift in Sources */ = {isa = PBXBuildFile; fileRef = 6F49C55FFF4352A6763BAE1A /* PerDomainDisableSettings.swift */; }; + 5E2DB725654F411FB808AB01 /* LocalRuntimeResidencyPolicy.swift in Sources */ = {isa = PBXBuildFile; fileRef = 4BB5841A73249EA3C1A4FA7E /* LocalRuntimeResidencyPolicy.swift */; }; 5E307FD6E4846B8A0B03BF59 /* SuggestionCaretLayoutRepairTests.swift in Sources */ = {isa = PBXBuildFile; fileRef = B228E9A83E5F6D879C384DA9 /* SuggestionCaretLayoutRepairTests.swift */; }; 5E43AD2C53EA1813A6A138E3 /* SuggestionSettingsModel.swift in Sources */ = {isa = PBXBuildFile; fileRef = DFFEDED24AED136A9C83F730 /* SuggestionSettingsModel.swift */; }; 5E506F0CD6FA82B323264C88 /* RandomMacroEvaluatorTests.swift in Sources */ = {isa = PBXBuildFile; fileRef = 4B0DF8F53EDD401B6A9EE06E /* RandomMacroEvaluatorTests.swift */; }; @@ -696,6 +697,7 @@ B4A2F4CF9253A927961A8E08 /* SuggestionPresentationTimingTests.swift in Sources */ = {isa = PBXBuildFile; fileRef = D89680F74D4E1097F81FC02A /* SuggestionPresentationTimingTests.swift */; }; B4D36F5D03E3143CE74582F9 /* AppearancePaneView.swift in Sources */ = {isa = PBXBuildFile; fileRef = AFBE491B3CA04FE9069B7B0F /* AppearancePaneView.swift */; }; B50D696D40BB500C5CD39F00 /* Aria2DownloadService.swift in Sources */ = {isa = PBXBuildFile; fileRef = DE6E2A19BF6B87A0C6FA9D49 /* Aria2DownloadService.swift */; }; + B5241C6026A756A7FDE4F041 /* LocalRuntimeResidencyPolicy.swift in Sources */ = {isa = PBXBuildFile; fileRef = 4BB5841A73249EA3C1A4FA7E /* LocalRuntimeResidencyPolicy.swift */; }; B55B160E0534AE23BAC1C3DA /* CotabbyApp.swift in Sources */ = {isa = PBXBuildFile; fileRef = CC1EDFB535AAA2EE0D67828A /* CotabbyApp.swift */; }; B582B229486BFA041A4641BA /* DateMacroEvaluatorTests.swift in Sources */ = {isa = PBXBuildFile; fileRef = E34A14D9D852BB9C6829DDC7 /* DateMacroEvaluatorTests.swift */; }; B5C255E58C128C4042FA7F66 /* PhrasePredictionScoringTests.swift in Sources */ = {isa = PBXBuildFile; fileRef = 8E542E57459488F3D39A9053 /* PhrasePredictionScoringTests.swift */; }; @@ -888,6 +890,7 @@ E6C99AB3E1A56E71F3A518FB /* ModelDownloadManager.swift in Sources */ = {isa = PBXBuildFile; fileRef = 050DC9D858A124FD1CFA4D99 /* ModelDownloadManager.swift */; }; E6DD9EAAFF4001E2922B5F55 /* OnboardingStyle.swift in Sources */ = {isa = PBXBuildFile; fileRef = 9ABC808DE01AF9AD04E38943 /* OnboardingStyle.swift */; }; E72C52902EC84711A80C6D25 /* KeyRecorderView.swift in Sources */ = {isa = PBXBuildFile; fileRef = A1D44E50290A78AFAE7753DB /* KeyRecorderView.swift */; }; + E79834E768C2001098CE6132 /* LocalRuntimeResidencyPolicyTests.swift in Sources */ = {isa = PBXBuildFile; fileRef = C0CF171FCE0BD5D84E555F1E /* LocalRuntimeResidencyPolicyTests.swift */; }; E83E989C5132F2A8C8FED8EB /* SuggestionSettingsDomainTests.swift in Sources */ = {isa = PBXBuildFile; fileRef = 2E7C883990AE384602BA9AF6 /* SuggestionSettingsDomainTests.swift */; }; E846FC1E29526242117F126F /* MenuBarPresentationObserver.swift in Sources */ = {isa = PBXBuildFile; fileRef = 9DE86A49339BB507D197629D /* MenuBarPresentationObserver.swift */; }; E8768B22E683F35098DF4CCF /* InlinePreviewPanelController.swift in Sources */ = {isa = PBXBuildFile; fileRef = 924CAA5E25C596A9FAB7602B /* InlinePreviewPanelController.swift */; }; @@ -1170,6 +1173,7 @@ 4B0DF8F53EDD401B6A9EE06E /* RandomMacroEvaluatorTests.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = RandomMacroEvaluatorTests.swift; sourceTree = ""; }; 4B8665A5495891F9E3DDA48B /* de-100k.txt */ = {isa = PBXFileReference; lastKnownFileType = text; path = "de-100k.txt"; sourceTree = ""; }; 4BA1D28712B262ABD96F8399 /* ModelAndPresentationValueTests.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = ModelAndPresentationValueTests.swift; sourceTree = ""; }; + 4BB5841A73249EA3C1A4FA7E /* LocalRuntimeResidencyPolicy.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = LocalRuntimeResidencyPolicy.swift; sourceTree = ""; }; 4BC92317837813ACA5051177 /* Cotabby Dev.app */ = {isa = PBXFileReference; explicitFileType = wrapper.application; includeInIndex = 0; path = "Cotabby Dev.app"; sourceTree = BUILT_PRODUCTS_DIR; }; 4BF9A081715185297A31EB8F /* GhostCaretRefinementTests.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = GhostCaretRefinementTests.swift; sourceTree = ""; }; 4CE9156494BFC0A1E12E0B6C /* OpenAICompatibleSuggestionEngineTests.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = OpenAICompatibleSuggestionEngineTests.swift; sourceTree = ""; }; @@ -1465,6 +1469,7 @@ C01F11F7F29EFC3A27A1EE5D /* FocusSnapshotResolverSelectionTests.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = FocusSnapshotResolverSelectionTests.swift; sourceTree = ""; }; C0894DA7EA3E157596DA0E66 /* SuggestionStaleAcceptanceEchoTests.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = SuggestionStaleAcceptanceEchoTests.swift; sourceTree = ""; }; C08EC70547439E840417F5A5 /* GhostBaselinePolicy.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = GhostBaselinePolicy.swift; sourceTree = ""; }; + C0CF171FCE0BD5D84E555F1E /* LocalRuntimeResidencyPolicyTests.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = LocalRuntimeResidencyPolicyTests.swift; sourceTree = ""; }; C1337F5C94FA9A6388841805 /* FocusPollBackoffTests.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = FocusPollBackoffTests.swift; sourceTree = ""; }; C1A8962FE08E1C76F0EEF423 /* CaretRunPlacementTests.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = CaretRunPlacementTests.swift; sourceTree = ""; }; C1D68B5ABA427D3E87000E78 /* InlineCommandCoordinator.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = InlineCommandCoordinator.swift; sourceTree = ""; }; @@ -2843,6 +2848,7 @@ D1AB5EBB44EE64EB3EB1426F /* BundledRuntimeLocator.swift */, C319B4FF652512A69377DD5C /* DecodeStopPolicy.swift */, 6F324C645A4E53FB6E90061A /* DownloadOutcomeClassifier.swift */, + 4BB5841A73249EA3C1A4FA7E /* LocalRuntimeResidencyPolicy.swift */, 50AECDF57800BBB6617620D5 /* TokenHealingBuffer.swift */, FDAA9ADF6C0BD9D16016AA9B /* TokenHealingPlan.swift */, ); @@ -3560,6 +3566,7 @@ EBD7B94FF42D8C538C328E80 /* BundledRuntimeLocatorTests.swift */, 81D7F428477976398CFC5CCF /* DecodeStopPolicyTests.swift */, 7B3AD6D4125ED99669F4EC32 /* DownloadOutcomeClassifierTests.swift */, + C0CF171FCE0BD5D84E555F1E /* LocalRuntimeResidencyPolicyTests.swift */, 80588CA72AB588807BE1C499 /* TokenHealingBufferTests.swift */, CEDD364CC7C995916BD1E4B0 /* TokenHealingPlanTests.swift */, ); @@ -3996,6 +4003,7 @@ 163735A1B906AE8B009CC547 /* LlamaRuntimeManager.swift in Sources */, F50F54FB1B3B2B2E464B8683 /* LlamaRuntimeModels.swift in Sources */, 3A5BA316620E502DAED81EE1 /* LlamaSuggestionEngine.swift in Sources */, + B5241C6026A756A7FDE4F041 /* LocalRuntimeResidencyPolicy.swift in Sources */, C6867B979A03C7928BC83966 /* LowPowerModeMonitor.swift in Sources */, 647F30C40CE57EDFDA95958F /* MacroController.swift in Sources */, 1D54C941ED086C6427CFD773 /* MacroEngine.swift in Sources */, @@ -4323,6 +4331,7 @@ 7BEA76E69707BC760B0D2394 /* LlamaRuntimeManager.swift in Sources */, D42BBBD704B21898047615E0 /* LlamaRuntimeModels.swift in Sources */, 3CEF8F66CBBD5D828135FF9E /* LlamaSuggestionEngine.swift in Sources */, + 5E2DB725654F411FB808AB01 /* LocalRuntimeResidencyPolicy.swift in Sources */, 01D6FF389B0E14C5BFDE4FBB /* LowPowerModeMonitor.swift in Sources */, E04950B45EF6C6CE47EE5081 /* MacroController.swift in Sources */, 2EDFE6F33018D2D43D5CB813 /* MacroEngine.swift in Sources */, @@ -4616,6 +4625,7 @@ EA91121F23229B9BA8474680 /* LlamaSuggestionEngineTests.swift in Sources */, BE16C2851265AB659A322835 /* LlamaSuggestionEvalTests.swift in Sources */, 67C36958132793C76843F875 /* LlamaTypingSessionEvalTests.swift in Sources */, + E79834E768C2001098CE6132 /* LocalRuntimeResidencyPolicyTests.swift in Sources */, C4ED7301D4DC86DC558B0C04 /* LowPowerModeMonitorTests.swift in Sources */, 332F2AE6277FB6E871589E4D /* MacroControllerTests.swift in Sources */, AB8F186A7CB13779DAF729A0 /* MacroEngineTests.swift in Sources */, diff --git a/Cotabby/App/Core/AppDelegate.swift b/Cotabby/App/Core/AppDelegate.swift index e29c9e63..f22465dd 100644 --- a/Cotabby/App/Core/AppDelegate.swift +++ b/Cotabby/App/Core/AppDelegate.swift @@ -83,11 +83,14 @@ final class AppDelegate: NSObject, NSApplicationDelegate { } .store(in: &cancellables) - suggestionSettings.$selectedEngine + // The engine and the two fallback switches decide whether the local model stays loaded. + // The publisher carries the decision computed from the emitted values: `@Published` emits + // before the property changes, so reading `suggestionSettings` here would act on the old + // settings. `dropFirst` skips the replayed current value; launch applies it explicitly. + suggestionSettings.localRuntimeResidencyPublisher .dropFirst() - .removeDuplicates() - .sink { [weak self] _ in - self?.startRuntimeIfPreferredEngineRequiresIt() + .sink { [weak self] keepsModelLoaded in + self?.applyRuntimeResidency(keepsModelLoaded: keepsModelLoaded) } .store(in: &cancellables) @@ -278,15 +281,21 @@ final class AppDelegate: NSObject, NSApplicationDelegate { ) } - /// Warm the local runtime only when the user is actually on a local engine path. - /// This avoids noisy startup failures and wasted work for Apple Intelligence users. + /// Warm the local runtime only when the user is actually on a local engine path, or chose to keep + /// the Apple Intelligence fallback model ready. This avoids noisy startup failures and wasted + /// work for everyone else. Safe to call outside a `@Published` sink, where settings are current. private func startRuntimeIfPreferredEngineRequiresIt() { - switch suggestionSettings.selectedEngine { - case .llamaOpenSource: + applyRuntimeResidency(keepsModelLoaded: suggestionSettings.keepsLocalRuntimeLoaded) + } + + /// Starts or stops the runtime for a decision from `LocalRuntimeResidencyPolicy`. + private func applyRuntimeResidency(keepsModelLoaded: Bool) { + if keepsModelLoaded { runtimeModel.startIfNeeded() - case .appleIntelligence, .openAICompatible: + } else { // Switching away must release Metal buffers and the mapped GGUF. Otherwise an Ollama - // user still pays the duplicate memory cost the external endpoint is meant to avoid. + // user still pays the duplicate memory cost the external endpoint is meant to avoid, and + // an Apple Intelligence user keeps a model they asked not to keep loaded. runtimeModel.stop() } } diff --git a/Cotabby/Models/Runtime/RuntimeBootstrapModel.swift b/Cotabby/Models/Runtime/RuntimeBootstrapModel.swift index bd9d6906..26f5d8ce 100644 --- a/Cotabby/Models/Runtime/RuntimeBootstrapModel.swift +++ b/Cotabby/Models/Runtime/RuntimeBootstrapModel.swift @@ -131,6 +131,27 @@ final class RuntimeBootstrapModel: ObservableObject { await runtimeTask?.value } + /// Persists the user's chosen model without loading it. `selectModel` would load the GGUF right + /// away, which is wrong while the runtime is meant to stay unloaded, as with the Apple + /// Intelligence fallback picker when "Keep Fallback Model Loaded" is off. The runtime manager + /// resolves the selection on every prepare, so the next on-demand fallback or engine switch + /// loads this model, replacing any model a previous fallback left loaded. + func selectModelWithoutLoading(_ filename: String) { + guard availableModels.contains(where: { $0.filename == filename }), + selectedModelFilename != filename else { + return + } + + selectedModelFilename = filename + persistSelectedModelFilename(filename) + // Signal the switch even though nothing loads yet: a model an earlier fallback loaded may + // still be generating or showing a suggestion, and the next request runs on the new model. + // Clearing that state now matches `selectModel`, so no completion from the old model is + // left on screen to accept. + onWillReloadModel?() + runtimeManager.configureSelectedModel(filename: filename) + } + /// Cancels pending startup work and forwards shutdown to the underlying runtime manager. func stop() { runtimeTask?.cancel() diff --git a/Cotabby/Models/Settings/SuggestionSettingsData.swift b/Cotabby/Models/Settings/SuggestionSettingsData.swift index d8241b99..19aa4416 100644 --- a/Cotabby/Models/Settings/SuggestionSettingsData.swift +++ b/Cotabby/Models/Settings/SuggestionSettingsData.swift @@ -28,6 +28,11 @@ struct SuggestionEngineSettings: Equatable { var pluggedInEngine: SuggestionEngineKind var pluggedInModelFilename: String var pluggedInEndpointModelName: String + /// When Apple Intelligence rejects a language, retry with the selected Open Source model. + var isAppleLanguageFallbackEnabled: Bool + /// Keep that fallback model loaded while Apple Intelligence is selected, so the first + /// fallback suggestion does not wait for the model to load. Costs the model's memory. + var keepsFallbackModelLoaded: Bool } /// Completion length, timing, streaming, and acceptance behavior. @@ -185,6 +190,16 @@ extension SuggestionSettingsData { set { engine.isPowerBasedModelSwitchingEnabled = newValue } } + var isAppleLanguageFallbackEnabled: Bool { + get { engine.isAppleLanguageFallbackEnabled } + set { engine.isAppleLanguageFallbackEnabled = newValue } + } + + var keepsFallbackModelLoaded: Bool { + get { engine.keepsFallbackModelLoaded } + set { engine.keepsFallbackModelLoaded = newValue } + } + var batteryEngine: SuggestionEngineKind { get { engine.batteryEngine } set { engine.batteryEngine = newValue } diff --git a/Cotabby/Models/Settings/SuggestionSettingsModel.swift b/Cotabby/Models/Settings/SuggestionSettingsModel.swift index 61644320..25c35191 100644 --- a/Cotabby/Models/Settings/SuggestionSettingsModel.swift +++ b/Cotabby/Models/Settings/SuggestionSettingsModel.swift @@ -158,6 +158,11 @@ final class SuggestionSettingsModel: ObservableObject { @Published private(set) var perAppShortcutOverrides: [PerAppShortcutOverride] @Published private(set) var acceptanceGranularity: AcceptanceGranularity @Published private(set) var isPowerBasedModelSwitchingEnabled: Bool + /// Retry with the Open Source model when Apple Intelligence rejects the text's language. + /// Read live by `SuggestionEngineRouter` at request time. + @Published private(set) var isAppleLanguageFallbackEnabled: Bool + /// Keep the fallback model loaded while Apple Intelligence is selected (see `AppDelegate`). + @Published private(set) var keepsFallbackModelLoaded: Bool @Published private(set) var batteryEngine: SuggestionEngineKind @Published private(set) var batteryModelFilename: String @Published private(set) var batteryEndpointModelName: String @@ -286,6 +291,8 @@ final class SuggestionSettingsModel: ObservableObject { perAppShortcutOverrides = data.perAppShortcutOverrides acceptanceGranularity = data.acceptanceGranularity isPowerBasedModelSwitchingEnabled = data.isPowerBasedModelSwitchingEnabled + isAppleLanguageFallbackEnabled = data.isAppleLanguageFallbackEnabled + keepsFallbackModelLoaded = data.keepsFallbackModelLoaded batteryEngine = data.batteryEngine batteryModelFilename = data.batteryModelFilename batteryEndpointModelName = data.batteryEndpointModelName @@ -369,6 +376,8 @@ final class SuggestionSettingsModel: ObservableObject { perAppShortcutOverrides = data.perAppShortcutOverrides acceptanceGranularity = data.acceptanceGranularity isPowerBasedModelSwitchingEnabled = data.isPowerBasedModelSwitchingEnabled + isAppleLanguageFallbackEnabled = data.isAppleLanguageFallbackEnabled + keepsFallbackModelLoaded = data.keepsFallbackModelLoaded batteryEngine = data.batteryEngine batteryModelFilename = data.batteryModelFilename batteryEndpointModelName = data.batteryEndpointModelName @@ -410,7 +419,9 @@ final class SuggestionSettingsModel: ObservableObject { batteryEndpointModelName: batteryEndpointModelName, pluggedInEngine: pluggedInEngine, pluggedInModelFilename: pluggedInModelFilename, - pluggedInEndpointModelName: pluggedInEndpointModelName + pluggedInEndpointModelName: pluggedInEndpointModelName, + isAppleLanguageFallbackEnabled: isAppleLanguageFallbackEnabled, + keepsFallbackModelLoaded: keepsFallbackModelLoaded ), completion: SuggestionCompletionSettings( selectedWordCountPreset: selectedWordCountPreset, @@ -571,6 +582,50 @@ final class SuggestionSettingsModel: ObservableObject { endpointCredentialRevision &+= 1 } + func setAppleLanguageFallbackEnabled(_ enabled: Bool) { + guard isAppleLanguageFallbackEnabled != enabled else { return } + isAppleLanguageFallbackEnabled = enabled + store.saveAppleLanguageFallbackEnabled(enabled) + } + + func setKeepsFallbackModelLoaded(_ enabled: Bool) { + guard keepsFallbackModelLoaded != enabled else { return } + keepsFallbackModelLoaded = enabled + store.saveKeepsFallbackModelLoaded(enabled) + } + + /// Whether the local runtime should keep its model loaded under the current settings. Read + /// this outside a `@Published` sink; inside one, use `localRuntimeResidencyPublisher`. + var keepsLocalRuntimeLoaded: Bool { + LocalRuntimeResidencyPolicy.keepsModelLoaded( + engine: selectedEngine, + isAppleLanguageFallbackEnabled: isAppleLanguageFallbackEnabled, + keepsFallbackModelLoaded: keepsFallbackModelLoaded + ) + } + + /// Emits the residency decision whenever the engine or either fallback switch changes, starting + /// with the current value. The decision is computed from the *emitted* values on purpose: + /// `@Published` publishes from the property's `willSet`, so a subscriber that read + /// `selectedEngine` back would still see the previous engine and start or stop the wrong way. + /// Equal decisions are not collapsed: switching from Apple Intelligence to the endpoint emits + /// `false` again, and that repeat is what releases a model the fallback loaded on demand. + var localRuntimeResidencyPublisher: AnyPublisher { + Publishers.CombineLatest3( + $selectedEngine.removeDuplicates(), + $isAppleLanguageFallbackEnabled.removeDuplicates(), + $keepsFallbackModelLoaded.removeDuplicates() + ) + .map { engine, isFallbackEnabled, keepsFallbackLoaded in + LocalRuntimeResidencyPolicy.keepsModelLoaded( + engine: engine, + isAppleLanguageFallbackEnabled: isFallbackEnabled, + keepsFallbackModelLoaded: keepsFallbackLoaded + ) + } + .eraseToAnyPublisher() + } + func setPowerBasedModelSwitchingEnabled(_ enabled: Bool) { guard isPowerBasedModelSwitchingEnabled != enabled else { return diff --git a/Cotabby/Services/Runtime/SuggestionEngineRouter.swift b/Cotabby/Services/Runtime/SuggestionEngineRouter.swift index 0c184347..fb107bd7 100644 --- a/Cotabby/Services/Runtime/SuggestionEngineRouter.swift +++ b/Cotabby/Services/Runtime/SuggestionEngineRouter.swift @@ -62,6 +62,22 @@ final class SuggestionEngineRouter { recordQualityOutcome(result) return result } catch SuggestionClientError.unsupportedLanguageOrLocale(let message) { + // The user can turn the fallback off (Engine & Model → Apple Intelligence). Then an + // unsupported language simply gets no suggestion, and the local model never loads. + guard suggestionSettings.isAppleLanguageFallbackEnabled else { + CotabbyLogger.suggestion.info( + "Apple Intelligence unsupported for locale; fallback is turned off", + metadata: metadata.merging(["reason": .string(message)]) { _, new in new } + ) + let result = SuggestionResult( + generation: request.generation, rawText: "", text: "", latency: 0, + suppressionReason: "appleLanguageUnsupported" + ) + // The coordinator leaves results that carry a suppression reason to the router, + // so this is the only place the withheld request can be counted. + recordQualityOutcome(result) + return result + } CotabbyLogger.suggestion.info( "Apple Intelligence unsupported for locale, falling back to open-source: \(message)", metadata: metadata.merging([ diff --git a/Cotabby/Support/Runtime/BundledRuntimeLocator.swift b/Cotabby/Support/Runtime/BundledRuntimeLocator.swift index 165e6ca1..ab8b6e68 100644 --- a/Cotabby/Support/Runtime/BundledRuntimeLocator.swift +++ b/Cotabby/Support/Runtime/BundledRuntimeLocator.swift @@ -103,8 +103,12 @@ struct BundledRuntimeLocator { /// them out of the model picker. static func discoverGGUFModelURLs(in directoryURL: URL, maxDepth: Int = 4) -> [URL] { let fileManager = FileManager.default + // `FileManager.enumerator` does not descend into a root that is itself a symbolic link, so a + // models folder linked to another location (an external drive, another app's folder) would + // list nothing. Resolve the root first; links deeper inside stay unfollowed, which keeps + // the bounded walk from escaping into arbitrary trees. guard let enumerator = fileManager.enumerator( - at: directoryURL, + at: directoryURL.resolvingSymlinksInPath(), includingPropertiesForKeys: [.isRegularFileKey], options: [.skipsHiddenFiles, .skipsPackageDescendants] ) else { diff --git a/Cotabby/Support/Runtime/LocalRuntimeResidencyPolicy.swift b/Cotabby/Support/Runtime/LocalRuntimeResidencyPolicy.swift new file mode 100644 index 00000000..39d7cc79 --- /dev/null +++ b/Cotabby/Support/Runtime/LocalRuntimeResidencyPolicy.swift @@ -0,0 +1,33 @@ +import Foundation + +/// Decides whether the in-process llama runtime should keep its model loaded in memory for a given +/// engine configuration. +/// +/// Why this exists as its own type: the answer depends on three settings (the engine plus the two +/// Apple Intelligence fallback switches), and two different places act on it. `AppDelegate` starts +/// or stops the runtime when any of them changes, and the fallback model picker in Settings decides +/// whether choosing a model should load it now or only record it. One pure rule keeps those callers +/// from drifting apart and lets tests pin the table without AppKit or a real model. +/// +/// "Not resident" does not forbid loading. Under Apple Intelligence the language fallback still +/// loads the selected model on demand (`LlamaRuntimeManager.preparedRuntime()`); this policy only +/// answers whether Cotabby should load it ahead of time and keep it there. +enum LocalRuntimeResidencyPolicy { + static func keepsModelLoaded( + engine: SuggestionEngineKind, + isAppleLanguageFallbackEnabled: Bool, + keepsFallbackModelLoaded: Bool + ) -> Bool { + switch engine { + case .llamaOpenSource: + return true + case .appleIntelligence: + // Both switches must be on: a kept-loaded model is useless when the fallback that + // would use it is turned off. + return isAppleLanguageFallbackEnabled && keepsFallbackModelLoaded + case .openAICompatible: + // The endpoint runs its own model, so a resident GGUF would only double memory use. + return false + } + } +} diff --git a/Cotabby/Support/Settings/SuggestionSettingsStore.swift b/Cotabby/Support/Settings/SuggestionSettingsStore.swift index 2f724e93..81466e1c 100644 --- a/Cotabby/Support/Settings/SuggestionSettingsStore.swift +++ b/Cotabby/Support/Settings/SuggestionSettingsStore.swift @@ -189,6 +189,8 @@ struct SuggestionSettingsStore { private static let globalToggleKeyLabelDefaultsKey = "cotabbyGlobalToggleKeyLabel" private static let acceptanceGranularityDefaultsKey = "cotabbyAcceptanceGranularity" + private static let appleLanguageFallbackEnabledDefaultsKey = "cotabbyAppleLanguageFallbackEnabled" + private static let keepFallbackModelLoadedDefaultsKey = "cotabbyKeepFallbackModelLoaded" private static let powerModelSwitchingEnabledDefaultsKey = "cotabbyPowerBasedModelSwitchingEnabled" private static let batteryEngineDefaultsKey = "cotabbyBatteryEngine" private static let batteryModelFilenameDefaultsKey = "cotabbyBatteryModelFilename" @@ -270,6 +272,8 @@ struct SuggestionSettingsStore { globalToggleKeyLabelDefaultsKey, acceptanceGranularityDefaultsKey, powerModelSwitchingEnabledDefaultsKey, + appleLanguageFallbackEnabledDefaultsKey, + keepFallbackModelLoadedDefaultsKey, batteryEngineDefaultsKey, batteryModelFilenameDefaultsKey, batteryEndpointModelNameDefaultsKey, @@ -549,6 +553,11 @@ struct SuggestionSettingsStore { let resolvedPowerBasedModelSwitchingEnabled = userDefaults.object(forKey: Self.powerModelSwitchingEnabledDefaultsKey) as? Bool ?? false + // On by default: falling back is what Cotabby has always done for unsupported languages. + let resolvedAppleLanguageFallbackEnabled = + userDefaults.object(forKey: Self.appleLanguageFallbackEnabledDefaultsKey) as? Bool ?? true + let resolvedKeepsFallbackModelLoaded = + userDefaults.object(forKey: Self.keepFallbackModelLoadedDefaultsKey) as? Bool ?? false let resolvedBatteryEngine = userDefaults.string(forKey: Self.batteryEngineDefaultsKey) .flatMap(SuggestionEngineKind.init(rawValue:)) ?? .llamaOpenSource let resolvedBatteryModelFilename = userDefaults.string(forKey: Self.batteryModelFilenameDefaultsKey) ?? "" @@ -580,7 +589,9 @@ struct SuggestionSettingsStore { batteryEndpointModelName: resolvedBatteryEndpointModelName, pluggedInEngine: resolvedPluggedInEngine, pluggedInModelFilename: resolvedPluggedInModelFilename, - pluggedInEndpointModelName: resolvedPluggedInEndpointModelName + pluggedInEndpointModelName: resolvedPluggedInEndpointModelName, + isAppleLanguageFallbackEnabled: resolvedAppleLanguageFallbackEnabled, + keepsFallbackModelLoaded: resolvedKeepsFallbackModelLoaded ), completion: SuggestionCompletionSettings( selectedWordCountPreset: resolvedWordCountPreset, @@ -724,6 +735,8 @@ struct SuggestionSettingsStore { savePerAppShortcutOverrides(data.perAppShortcutOverrides) saveAcceptanceGranularity(data.acceptanceGranularity) savePowerBasedModelSwitchingEnabled(data.isPowerBasedModelSwitchingEnabled) + saveAppleLanguageFallbackEnabled(data.isAppleLanguageFallbackEnabled) + saveKeepsFallbackModelLoaded(data.keepsFallbackModelLoaded) saveBatteryEngine(data.batteryEngine) saveBatteryModelFilename(data.batteryModelFilename) saveBatteryEndpointModelName(data.batteryEndpointModelName) @@ -851,6 +864,14 @@ struct SuggestionSettingsStore { userDefaults.set(mode.rawValue, forKey: Self.openAICompatibleAPIModeDefaultsKey) } + func saveAppleLanguageFallbackEnabled(_ enabled: Bool) { + userDefaults.set(enabled, forKey: Self.appleLanguageFallbackEnabledDefaultsKey) + } + + func saveKeepsFallbackModelLoaded(_ enabled: Bool) { + userDefaults.set(enabled, forKey: Self.keepFallbackModelLoadedDefaultsKey) + } + func savePowerBasedModelSwitchingEnabled(_ enabled: Bool) { userDefaults.set(enabled, forKey: Self.powerModelSwitchingEnabledDefaultsKey) } diff --git a/Cotabby/UI/Settings/Panes/Engine/EngineAndModelPaneView+AppleIntelligence.swift b/Cotabby/UI/Settings/Panes/Engine/EngineAndModelPaneView+AppleIntelligence.swift index c3de29ef..4567ed8b 100644 --- a/Cotabby/UI/Settings/Panes/Engine/EngineAndModelPaneView+AppleIntelligence.swift +++ b/Cotabby/UI/Settings/Panes/Engine/EngineAndModelPaneView+AppleIntelligence.swift @@ -1,6 +1,7 @@ import SwiftUI -/// Apple Intelligence availability presentation. +/// Apple Intelligence availability plus the language-fallback controls (fallback switch, keep-loaded +/// switch, and fallback model picker). /// These members are internal because Swift extensions in separate files cannot share lexical `private` access; /// the owning view itself remains module-internal. extension EngineAndModelPaneView { @@ -23,6 +24,92 @@ extension EngineAndModelPaneView { ) } .settingsItem(.appleIntelligenceAvailability) + + Toggle(isOn: Binding( + get: { suggestionSettings.isAppleLanguageFallbackEnabled }, + set: { suggestionSettings.setAppleLanguageFallbackEnabled($0) } + )) { + SettingsRowLabel( + title: "Fall Back to Open Source Model", + description: "When Apple Intelligence doesn't support the language you're writing in, " + + "suggest with \(fallbackModelName) instead. Turn off to get no suggestion in those languages.", + systemImage: "arrow.triangle.branch" + ) + } + .settingsItem(.appleLanguageFallback) + + Toggle(isOn: Binding( + get: { suggestionSettings.keepsFallbackModelLoaded }, + set: { suggestionSettings.setKeepsFallbackModelLoaded($0) } + )) { + SettingsRowLabel( + title: "Keep Fallback Model Loaded", + description: "Load the fallback model in advance so its first suggestion doesn't wait for it " + + "to load. Uses the model's memory (up to several GB) while Apple Intelligence is selected.", + systemImage: "memorychip" + ) + } + .disabled(!suggestionSettings.isAppleLanguageFallbackEnabled) + .settingsItem(.keepFallbackModelLoaded) + + // The Open Source section is hidden while Apple Intelligence is the engine, so the + // fallback model is chosen here. It is the same selection the Open Source engine uses: + // the local runtime holds one model at a time. + if runtimeModel.availableModels.isEmpty { + // Only worth a warning while the fallback is on; with it off, no model is needed. + if suggestionSettings.isAppleLanguageFallbackEnabled { + Text("No downloaded models were found, so there is nothing to fall back to. " + + "Switch the engine to Open Source to download one.") + .font(.caption) + .foregroundStyle(.orange) + } + } else { + Picker(selection: fallbackModelBinding) { + ForEach(runtimeModel.availableModels) { model in + Text(model.displayName).tag(model.filename) + } + } label: { + SettingsRowLabel( + title: "Fallback Model", + description: suggestionSettings.isPowerBasedModelSwitchingEnabled + ? "Set automatically by power source. Turn off power-based switching in the " + + "Power section to choose it here." + : "The downloaded model used when Apple Intelligence can't handle the " + + "language. It is also your Open Source engine's model.", + systemImage: "shippingbox" + ) + } + // Disabled under power-based switching for the same reason as the Open Source + // picker: the power profiles own the selected model and would revert a pick here. + .disabled(!suggestionSettings.isAppleLanguageFallbackEnabled + || suggestionSettings.isPowerBasedModelSwitchingEnabled) + .settingsItem(.appleLanguageFallbackModel) + } } } + + /// The model the fallback uses: the selected Open Source model. The local runtime holds one + /// model at a time, so the fallback cannot use a different one without swapping it in. Uses the + /// catalog name so the toggle names the model the same way the picker below lists it. + private var fallbackModelName: String { + runtimeModel.selectedModelFilename.map { RuntimeModelCatalog.displayName(for: $0) } ?? "your Open Source model" + } + + /// Shares the Open Source selection, but picking a model here must not load it on its own. + /// While the runtime is meant to stay unloaded ("Keep Fallback Model Loaded" off), the choice is + /// only recorded and the next fallback loads it; `selectedModelBinding` would load several GB + /// the user asked not to keep in memory. When the runtime is kept loaded, swap the model now so + /// the fallback stays ready. + private var fallbackModelBinding: Binding { + Binding( + get: { selectedModelBinding.wrappedValue }, + set: { filename in + if suggestionSettings.keepsLocalRuntimeLoaded { + Task { await runtimeModel.selectModel(filename) } + } else { + runtimeModel.selectModelWithoutLoading(filename) + } + } + ) + } } diff --git a/Cotabby/UI/Settings/SettingsIndex.swift b/Cotabby/UI/Settings/SettingsIndex.swift index 29ce5545..7066a069 100644 --- a/Cotabby/UI/Settings/SettingsIndex.swift +++ b/Cotabby/UI/Settings/SettingsIndex.swift @@ -63,6 +63,9 @@ enum SettingsItem: String, CaseIterable, Identifiable { // Engine & Model case engine case appleIntelligenceAvailability + case appleLanguageFallback + case keepFallbackModelLoaded + case appleLanguageFallbackModel case modelStatus case selectedModel case lowPowerModeAutoDisable @@ -152,6 +155,9 @@ enum SettingsItem: String, CaseIterable, Identifiable { case .contextLivePreview: return "Live Preview" case .engine: return "Engine" case .appleIntelligenceAvailability: return "Apple Intelligence Availability" + case .appleLanguageFallback: return "Fall Back to Open Source Model" + case .keepFallbackModelLoaded: return "Keep Fallback Model Loaded" + case .appleLanguageFallbackModel: return "Fallback Model" case .modelStatus: return "Model Status" case .selectedModel: return "Selected Model" case .lowPowerModeAutoDisable: return "Pause in Low Power Mode" @@ -236,6 +242,9 @@ enum SettingsItem: String, CaseIterable, Identifiable { case .contextLivePreview: return "text.cursor" case .engine: return "cpu" case .appleIntelligenceAvailability: return "apple.logo" + case .appleLanguageFallback: return "arrow.triangle.branch" + case .keepFallbackModelLoaded: return "memorychip" + case .appleLanguageFallbackModel: return "shippingbox" case .modelStatus: return "info.circle" case .selectedModel: return "shippingbox" case .lowPowerModeAutoDisable: return "bolt.slash.circle" @@ -293,8 +302,9 @@ enum SettingsItem: String, CaseIterable, Identifiable { return .writing case .extendedContext, .contextLivePreview: return .context - case .engine, .appleIntelligenceAvailability, .modelStatus, .selectedModel, - .lowPowerModeAutoDisable, .powerBasedModelSwitching, .batteryModel, .pluggedInModel, + case .engine, .appleIntelligenceAvailability, .appleLanguageFallback, .keepFallbackModelLoaded, + .appleLanguageFallbackModel, .modelStatus, .selectedModel, .lowPowerModeAutoDisable, + .powerBasedModelSwitching, .batteryModel, .pluggedInModel, .downloadModels, .huggingFaceBrowser, .modelsFolder, .lmStudio, .endpointBaseURL, .endpointAPIMode, .endpointAPIKey, .endpointStatus, .endpointModel: return .engineAndModel @@ -363,6 +373,9 @@ enum SettingsItem: String, CaseIterable, Identifiable { case .contextLivePreview: return "A real field that exercises the full pipeline." case .engine: return "Apple Intelligence, bundled Open Source, or a local endpoint." case .appleIntelligenceAvailability: return "Whether this Mac can run Apple Intelligence." + case .appleLanguageFallback: return "Use the local model for languages Apple Intelligence doesn't support." + case .keepFallbackModelLoaded: return "Preload the fallback model so its first suggestion is fast." + case .appleLanguageFallbackModel: return "Which downloaded model the language fallback uses." case .modelStatus: return "Whether the local model is loaded and ready." case .selectedModel: return "Which downloaded model generates suggestions." case .lowPowerModeAutoDisable: return "Pause suggestions while Low Power Mode is active." @@ -537,6 +550,14 @@ enum SettingsItem: String, CaseIterable, Identifiable { "provider", "runtime", "foundation models", "oss", "local", "endpoint", "ollama", "openai compatible", "lm studio", "vllm", "on-device", "model engine"] + case .appleLanguageFallbackModel: + return ["fallback model", "fallback", "model", "gemma", "gguf", "local model", "choose model"] + case .appleLanguageFallback: + return ["fallback", "fall back", "unsupported language", "language", "macedonian", + "open source", "local model", "gemma", "llama"] + case .keepFallbackModelLoaded: + return ["keep loaded", "preload", "warm", "memory", "ram", "resident", "fallback", + "first suggestion", "load time"] case .appleIntelligenceAvailability: return ["apple intelligence", "availability", "available", "supported", "compatibility", "status", "macos", "device support"] diff --git a/CotabbyTests/Models/Runtime/RuntimeBootstrapModelTests.swift b/CotabbyTests/Models/Runtime/RuntimeBootstrapModelTests.swift index dd15319a..62a79ff1 100644 --- a/CotabbyTests/Models/Runtime/RuntimeBootstrapModelTests.swift +++ b/CotabbyTests/Models/Runtime/RuntimeBootstrapModelTests.swift @@ -233,6 +233,65 @@ final class RuntimeBootstrapModelTests: XCTestCase { } } + // MARK: - selectModelWithoutLoading + + func test_selectModelWithoutLoading_persistsTheChoiceWithoutStartingTheRuntime() throws { + let directory = try makeModelDirectory(filenames: ["alpha.gguf", "beta.gguf"]) + let userDefaults = makeUserDefaults() + let model = runOnMainActor { makeModel(modelDirectory: directory, userDefaults: userDefaults) } + let reloads = ReloadCounter() + runOnMainActor { model.onWillReloadModel = { reloads.increment() } } + + // With the files gone, any load attempt fails inside the locator and leaves a .failed state, + // so an idle state afterwards proves no load was started. + try removeModelFile("alpha.gguf", in: directory) + try removeModelFile("beta.gguf", in: directory) + + runOnMainActor { + model.selectModelWithoutLoading("beta.gguf") + model.selectModelWithoutLoading("beta.gguf") + model.selectModelWithoutLoading("ghost.gguf") + } + RunLoop.main.run(until: Date(timeIntervalSinceNow: 0.05)) + + runOnMainActor { + XCTAssertEqual(model.selectedModelFilename, "beta.gguf", "An unknown filename must be ignored") + XCTAssertEqual(userDefaults.string(forKey: Self.selectionKey), "beta.gguf") + XCTAssertEqual(model.state, .idle, "Recording a choice must not load the model") + // The next request runs on the new model, so suggestion state from the old one is + // cleared once. The repeated pick and the unknown filename change nothing and must not + // signal again. + XCTAssertEqual(reloads.count, 1, "The model switch must clear suggestion state from the old model") + } + } + + func test_selectModelWithoutLoading_isUsedByTheNextStart() throws { + let directory = try makeModelDirectory(filenames: ["alpha.gguf", "beta.gguf", "charlie.gguf"]) + let userDefaults = makeUserDefaults() + let model = runOnMainActor { makeModel(modelDirectory: directory, userDefaults: userDefaults) } + runOnMainActor { + XCTAssertEqual(model.selectedModelFilename, "alpha.gguf") + model.selectModelWithoutLoading("beta.gguf") + } + // charlie stays on disk so resolution fails by name ("beta.gguf was not found") instead of + // with a generic empty-folder error. Removing alpha too keeps a regression that still loads + // alpha off the native path: it would fail naming alpha instead. + try removeModelFile("alpha.gguf", in: directory) + try removeModelFile("beta.gguf", in: directory) + + let (failed, cancellable) = expectFailureState(of: model) + runOnMainActor { model.startIfNeeded() } + wait(for: [failed], timeout: 10) + cancellable.cancel() + + runOnMainActor { + XCTAssertTrue( + model.diagnostics.lastError?.contains("beta.gguf") == true, + "Expected a beta.gguf load failure, got \(model.diagnostics.lastError ?? "nil")" + ) + } + } + // MARK: - Available-model reconciliation func test_refreshAvailableModels_discoversNewModelsAndKeepsValidSelection() throws { diff --git a/CotabbyTests/Models/Settings/SuggestionSettingsModelTests.swift b/CotabbyTests/Models/Settings/SuggestionSettingsModelTests.swift index c0cd176c..9153a63e 100644 --- a/CotabbyTests/Models/Settings/SuggestionSettingsModelTests.swift +++ b/CotabbyTests/Models/Settings/SuggestionSettingsModelTests.swift @@ -414,6 +414,52 @@ final class SuggestionSettingsModelTests: XCTestCase { XCTAssertEqual(model.pluggedInModelFilename, "") } + // MARK: - Local runtime residency + + /// `AppDelegate` starts or stops the runtime from these emissions. Each one must reflect the + /// setting that was just written, not the value it replaced: `@Published` sends from `willSet`, + /// so a decision built by reading the properties back would turn "Keep Fallback Model Loaded" + /// into its opposite. + func test_localRuntimeResidencyPublisher_emitsTheDecisionForTheNewSettings() { + let model = makeModel() + model.selectEngine(.appleIntelligence) + var decisions: [Bool] = [] + let subscription = model.localRuntimeResidencyPublisher.sink { decisions.append($0) } + defer { subscription.cancel() } + XCTAssertEqual(decisions, [false], "Defaults keep the fallback on but do not preload its model") + + model.setKeepsFallbackModelLoaded(true) + XCTAssertEqual(decisions, [false, true]) + XCTAssertTrue(model.keepsLocalRuntimeLoaded) + + model.setAppleLanguageFallbackEnabled(false) + XCTAssertEqual(decisions, [false, true, false], "Keep-loaded means nothing once the fallback is off") + + model.setAppleLanguageFallbackEnabled(true) + model.selectEngine(.openAICompatible) + XCTAssertEqual(decisions, [false, true, false, true, false]) + XCTAssertFalse(model.keepsLocalRuntimeLoaded) + + model.selectEngine(.llamaOpenSource) + XCTAssertEqual(decisions.last, true) + XCTAssertTrue(model.keepsLocalRuntimeLoaded) + } + + func test_localRuntimeResidencyPublisher_repeatsAnUnchangedDecisionOnEngineSwitch() { + // Apple Intelligence without keep-loaded and the endpoint both answer "not resident". The + // repeat must still arrive: it is what stops a model the fallback loaded on demand. + let model = makeModel() + model.selectEngine(.appleIntelligence) + var decisions: [Bool] = [] + let subscription = model.localRuntimeResidencyPublisher.sink { decisions.append($0) } + defer { subscription.cancel() } + + model.selectEngine(.openAICompatible) + model.selectEngine(.openAICompatible) + + XCTAssertEqual(decisions, [false, false], "A repeated write of the same engine is not a change") + } + // MARK: - Endpoint configuration func test_openAICompatibleConfiguration_validatesTheStoredEndpointFields() throws { diff --git a/CotabbyTests/Services/Runtime/SuggestionEngineRouterTests.swift b/CotabbyTests/Services/Runtime/SuggestionEngineRouterTests.swift index a5677310..9db1bed3 100644 --- a/CotabbyTests/Services/Runtime/SuggestionEngineRouterTests.swift +++ b/CotabbyTests/Services/Runtime/SuggestionEngineRouterTests.swift @@ -164,6 +164,41 @@ final class SuggestionEngineRouterRoutingTests: XCTestCase { XCTAssertEqual(rig.metrics.entries.first?.modelName, "test-model.gguf") } + func test_unsupportedLocale_withFallbackOff_returnsNoSuggestionAndSkipsTheLocalModel() async throws { + let rig = makeRig(engine: .appleIntelligence) + rig.settings.setAppleLanguageFallbackEnabled(false) + rig.foundation.script = { _ in + throw SuggestionClientError.unsupportedLanguageOrLocale("Locale not supported.") + } + + let result = try await rig.router.generateSuggestion(for: CotabbyTestFixtures.suggestionRequest()) + + XCTAssertEqual(result.text, "") + XCTAssertEqual(result.suppressionReason, "appleLanguageUnsupported") + XCTAssertTrue(rig.llama.requests.isEmpty, "With the fallback off the local model must not run") + // The coordinator skips results that carry a suppression reason, so the router must count + // this one or the Performance pane never shows it. + XCTAssertEqual(rig.quality.counters.generated, 1) + XCTAssertEqual(rig.quality.counters.suppressedByReason, ["appleLanguageUnsupported": 1]) + XCTAssertTrue(rig.metrics.entries.isEmpty, "Nothing was generated, so there is no latency to record") + } + + func test_fallbackSettingsDefaultToTodaysBehaviorAndPersist() { + let defaults = makeDefaults() + let settings = SuggestionSettingsModel(configuration: .standard, userDefaults: defaults) + Self.retained.append(settings) + XCTAssertTrue(settings.isAppleLanguageFallbackEnabled) + XCTAssertFalse(settings.keepsFallbackModelLoaded) + + settings.setAppleLanguageFallbackEnabled(false) + settings.setKeepsFallbackModelLoaded(true) + + let reloaded = SuggestionSettingsModel(configuration: .standard, userDefaults: defaults) + Self.retained.append(reloaded) + XCTAssertFalse(reloaded.isAppleLanguageFallbackEnabled) + XCTAssertTrue(reloaded.keepsFallbackModelLoaded) + } + func test_unsupportedLocale_fallbackFailureComposesBothMessages() async { let rig = makeRig(engine: .appleIntelligence) rig.foundation.script = { _ in diff --git a/CotabbyTests/Support/Runtime/BundledRuntimeLocatorTests.swift b/CotabbyTests/Support/Runtime/BundledRuntimeLocatorTests.swift index b64978f4..142ea79d 100644 --- a/CotabbyTests/Support/Runtime/BundledRuntimeLocatorTests.swift +++ b/CotabbyTests/Support/Runtime/BundledRuntimeLocatorTests.swift @@ -467,4 +467,19 @@ final class BundledRuntimeLocatorTests: XCTestCase { gpuLayerCount: -1 ) } + + func test_discoverFindsModelsThroughASymlinkedModelsFolder() throws { + let root = FileManager.default.temporaryDirectory.appendingPathComponent("gguf-link-\(UUID().uuidString)") + let real = root.appendingPathComponent("real") + let link = root.appendingPathComponent("LlamaRuntime") + try FileManager.default.createDirectory(at: real, withIntermediateDirectories: true) + defer { try? FileManager.default.removeItem(at: root) } + try Data("gguf".utf8).write(to: real.appendingPathComponent("model.gguf")) + try FileManager.default.createSymbolicLink(at: link, withDestinationURL: real) + + let found = BundledRuntimeLocator.discoverGGUFModelURLs(in: link) + + XCTAssertEqual(found.map(\.lastPathComponent), ["model.gguf"]) + } + } diff --git a/CotabbyTests/Support/Runtime/LocalRuntimeResidencyPolicyTests.swift b/CotabbyTests/Support/Runtime/LocalRuntimeResidencyPolicyTests.swift new file mode 100644 index 00000000..54075064 --- /dev/null +++ b/CotabbyTests/Support/Runtime/LocalRuntimeResidencyPolicyTests.swift @@ -0,0 +1,60 @@ +import XCTest +@testable import Cotabby + +/// Pins when the local llama runtime stays loaded. `AppDelegate` starts or stops the runtime from +/// this answer, and the Apple Intelligence fallback picker uses it to decide whether choosing a +/// model may load it, so a wrong row here costs either several GB of memory or a slow first +/// fallback suggestion. +final class LocalRuntimeResidencyPolicyTests: XCTestCase { + func test_openSourceAlwaysKeepsTheModelLoaded() { + for fallback in [false, true] { + for keepLoaded in [false, true] { + XCTAssertTrue( + LocalRuntimeResidencyPolicy.keepsModelLoaded( + engine: .llamaOpenSource, + isAppleLanguageFallbackEnabled: fallback, + keepsFallbackModelLoaded: keepLoaded + ), + "Open Source generates with the local model, so the fallback switches must not unload it" + ) + } + } + } + + func test_endpointNeverKeepsTheModelLoaded() { + for fallback in [false, true] { + for keepLoaded in [false, true] { + XCTAssertFalse( + LocalRuntimeResidencyPolicy.keepsModelLoaded( + engine: .openAICompatible, + isAppleLanguageFallbackEnabled: fallback, + keepsFallbackModelLoaded: keepLoaded + ), + "The endpoint runs its own model, so a resident GGUF would only duplicate memory" + ) + } + } + } + + func test_appleIntelligenceKeepsTheModelLoadedOnlyWhenBothFallbackSwitchesAreOn() { + let cases: [(fallback: Bool, keepLoaded: Bool, expected: Bool)] = [ + (false, false, false), + (true, false, false), + // Keep-loaded is disabled in Settings while the fallback is off, but its stored value + // survives. It must not hold memory for a fallback that will never run. + (false, true, false), + (true, true, true) + ] + for testCase in cases { + XCTAssertEqual( + LocalRuntimeResidencyPolicy.keepsModelLoaded( + engine: .appleIntelligence, + isAppleLanguageFallbackEnabled: testCase.fallback, + keepsFallbackModelLoaded: testCase.keepLoaded + ), + testCase.expected, + "fallback=\(testCase.fallback) keepLoaded=\(testCase.keepLoaded)" + ) + } + } +} diff --git a/CotabbyTests/UI/Settings/SettingsIndexTests.swift b/CotabbyTests/UI/Settings/SettingsIndexTests.swift index 25857bb4..2349e8a2 100644 --- a/CotabbyTests/UI/Settings/SettingsIndexTests.swift +++ b/CotabbyTests/UI/Settings/SettingsIndexTests.swift @@ -64,7 +64,11 @@ final class SettingsIndexTests: XCTestCase { ("typo", .automaticallyFixTypos), ("model status", .modelStatus), ("battery", .batteryModel), - ("plugged", .pluggedInModel) + ("plugged", .pluggedInModel), + ("unsupported language", .appleLanguageFallback), + ("fallback model", .appleLanguageFallbackModel), + ("keep loaded", .keepFallbackModelLoaded), + ("preload", .keepFallbackModelLoaded) ] for expectation in expectations { XCTAssertTrue( diff --git a/SOURCE_LAYOUT.md b/SOURCE_LAYOUT.md index 5ff2b576..1b96bbd2 100644 --- a/SOURCE_LAYOUT.md +++ b/SOURCE_LAYOUT.md @@ -92,6 +92,7 @@ Cotabby/ │ │ ├── BundledRuntimeLocator.swift │ │ ├── DecodeStopPolicy.swift │ │ ├── DownloadOutcomeClassifier.swift +│ │ ├── LocalRuntimeResidencyPolicy.swift when the local model stays loaded │ │ ├── TokenHealingPlan.swift bounded word-fragment retokenization at the caret │ │ └── TokenHealingBuffer.swift exact byte replay and lossless streamed UTF-8 │ ├── Settings/ persistence and settings policies