Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
35 commits
Select commit Hold shift + click to select a range
fedb853
Pin Flash native MTP verification corrections
Oct 6, 2026
d08eb33
Pin vmlx-swift perf/claude-swift-port-oct6 for speed proof
Oct 6, 2026
de69800
Speculation on by default for Flash-Next and bundled 27B DFlash2
Oct 7, 2026
f154244
Pin vmlx-swift 0e26bcb5: Qwen3.8-27B-JANGH2 loads
Oct 7, 2026
2fdd211
Integrate current schema support before speculative runtime qualifica…
Oct 7, 2026
97cb72b
Honor AR across loaded drafters and expose bundled speculation before…
Oct 7, 2026
ba87ee9
Describe bundle-aware speculative defaults and Off behavior
Oct 7, 2026
d7287a2
Show compatible DFlash capability consistently before model loading
Oct 7, 2026
5394dc1
Clarify advisory capacity diagnostics without changing explicit limits
Oct 7, 2026
7119be8
Migrate only provenance-tagged defaults to bundle-aware speculation
Oct 7, 2026
0c12a53
Expect bundle defaults only for fresh or provenance-owned settings
Oct 7, 2026
f3be35b
Use complete metadata fixture for bundled drafter detection
Oct 7, 2026
23a4958
Describe bundled DFlash selection accurately in picker help
Oct 7, 2026
136aac4
Restore bundle policy when resetting speculative decoding
Oct 7, 2026
4590c61
Reject non-string speculative option values
Oct 7, 2026
7c9e4a2
Test picker reset persists family speculation default
Oct 7, 2026
d7967fb
Align local model guides with bundle-aware speculation defaults
Oct 7, 2026
94d9bce
Import engine metadata API in preload fixture tests
Oct 7, 2026
0370584
Release unshared DFlash drafters after model teardown
Oct 7, 2026
e50fa3f
Pin reviewed Qwen defaults and complete drafter validation
Oct 7, 2026
ffb03f2
Pin reviewed Qwen runtime portability fixes
Oct 7, 2026
3627d4f
Pin vmlx-swift a7123840: sampled MTP, DFlash 2 trees, media speculation
Oct 7, 2026
492c66f
Keep the admitted scratch floor in MLX's freed-buffer pool
Oct 7, 2026
dde94c0
Pin vmlx-swift e40f9fa4: media conversation restore, direct 8-bit lan…
Oct 7, 2026
53231ce
Pin vmlx-swift a5a0689c (docs-only over the live-proven e40f9fa4)
Oct 7, 2026
5380ba8
Pin vmlx-swift 90eecf3c (JANGH dense split-K verify tile)
Oct 7, 2026
4c9e5e4
Pin vmlx-swift e54e7388 (DFlash 2 drafter cache after prefix restore)
Oct 7, 2026
f41de0a
Reload when leaving or entering the speculative family default
Oct 7, 2026
9167afe
fix: avoid mislabeling DFlash completion telemetry as plain decode
Oct 7, 2026
0a4428f
build: pin audited native MTP cache boundary candidate
Oct 7, 2026
a96dae9
fix: inspect bundle architecture before native MTP exemptions
Oct 7, 2026
d546b3a
Delegation usage reports the runtime's prompt tokens
Oct 7, 2026
f6700d3
Pin vmlx-swift dc139e2c (native MTP head completeness)
Oct 7, 2026
1383948
Pin vmlx-swift 7c5a8a08 (n-gram on SSD, residency on GPU weights)
Oct 7, 2026
15e917f
Pin Qwen speculative runtime to merged engine audit
Oct 7, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

24 changes: 18 additions & 6 deletions Packages/OsaurusCore/Models/Chat/NativeMTPSelectionDefault.swift
Original file line number Diff line number Diff line change
@@ -1,19 +1,30 @@
import Foundation
import MLXLMCommon

/// Native MTP is opt-in. Selecting a model only discovers capability; it never
/// activates speculation. Retain the old provenance keys so an automatic
/// family default can be retired without overwriting an explicit choice.
/// Selecting a model discovers the family-default capability without loading it.
/// Preserve explicit Off/Adaptive and historical provenance; never infer that a
/// persisted Off was unintentional.
enum NativeMTPSelectionDefault {
static let userChoseKey = "nativeMTPSegmentUserChose"
static let familyDefaultKey = "nativeMTPSegmentIsFamilyDefault"

/// A missing value means Reset to default, never an explicit Off choice.
static func mode(for segment: String?) -> VMLXMTPServerMode? {
switch segment {
case nil: return .familyDefault
case "off": return .off
case "auto": return .auto
default: return nil
}
}

/// Product controls expose only Off (AR) and On (Adaptive). Keep the
/// engine's legacy fields decodable, but never persist a native depth cap.
/// External DFlash selection and its block size are independent.
/// Keep external drafter selection stored; Off still disables its execution.
static func adaptiveSelection(_ settings: VMLXServerMTPSettings) -> VMLXServerMTPSettings {
var result = settings
result.mode = settings.mode == .off ? .off : .auto
// Off and the family default are kept as-is; everything else is On (Adaptive).
result.mode = (settings.mode == .off || settings.mode == .familyDefault) ? settings.mode : .auto
result.explicitDepth = nil
result.draftTokenLimit = nil
return result
Expand Down Expand Up @@ -48,7 +59,8 @@ enum NativeMTPSelectionDefault {
defaults.bool(forKey: familyDefaultKey),
settings == .init(mode: .forceOn, explicitDepth: 3)
|| settings == .init(mode: .auto)
|| settings == .init(mode: .off)
else { return settings }
return .init(mode: .off)
return .init(mode: .familyDefault)
}
}
17 changes: 17 additions & 0 deletions Packages/OsaurusCore/Models/Configuration/ModelFamilyNames.swift
Original file line number Diff line number Diff line change
Expand Up @@ -119,6 +119,23 @@ enum ModelFamilyNames {
|| normalized.contains("nex-n2-pro")
}

/// The MTP launch exception must match the actual text architecture as well
/// as the legacy JANG name. A renamed Qwen bundle must still inspect its head.
static func isMiMoOrN2JANGRuntimeFamily(_ modelId: String, configData: Data?) -> Bool {
guard isMiMoOrN2JANGRuntimeFamily(modelId),
let configData,
let config = try? JSONSerialization.jsonObject(with: configData) as? [String: Any],
let modelType = config["model_type"] as? String,
["mimo_v2", "mimo_v2_flash"].contains(modelType)
else { return false }
if let text = config["text_config"] as? [String: Any],
let textType = text["model_type"] as? String
{
return ["mimo_v2", "mimo_v2_flash"].contains(textType)
}
return true
}

/// DeepSeek-V4 / DSV4 Flash bundles (`model_type=deepseek_v4`).
/// Match both public repo forms (`DeepSeek-V4-...`) and shorthand
/// runtime names (`DSV4-...`, `deepseekv4-...`) while avoiding
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -65,13 +65,17 @@ struct ModelOptionsSnapshot: Encodable, Equatable {
/// refuses MTP.
struct NativeMTP: Equatable, Sendable {
let blocked: Bool
/// The bundle is in the family the default turns on (Qwen3.8 Flash-Next): under
/// `mtp.mode == .familyDefault` it runs Adaptive without a user choice.
var familyDefaultOn: Bool = false

/// File-only bundle inspection; run it off the main thread.
static func inspect(model: String) -> NativeMTP? {
guard let status = ModelRuntime.inspectLoadingModelMTP(name: model),
status.bundleHasMTP, status.isTargetMTPFamily
status.speculationAvailable
else { return nil }
return NativeMTP(blocked: status.isBlocked)
return NativeMTP(
blocked: status.speculationBlocked, familyDefaultOn: status.familyDefaultOn)
}
}

Expand Down Expand Up @@ -169,36 +173,41 @@ struct ModelOptionsSnapshot: Encodable, Equatable {
let off = Segment(id: "off", label: L("Off (AR)"), description: nil)
let adaptive = Segment(id: "auto", label: L("On (Adaptive)"), description: nil)
let mode = ServerController.runtimeSettingsForConfigureTool().settings.mtp.mode
let selected = nativeMTP.blocked || mode == .off ? "off" : "auto"
// `.familyDefault` shows what the engine will actually run for THIS bundle.
let effectiveOn = mode == .familyDefault ? nativeMTP.familyDefaultOn : mode != .off
let selected = nativeMTP.blocked || !effectiveOn ? "off" : "auto"
return Option(
id: nativeMTPOptionId,
label: L("Native MTP"),
label: L("Speculative Decoding"),
icon: "hare",
// The phone shows the choice alone, without the composer's copy.
help: nil,
kind: "segmented",
segments: nativeMTP.blocked ? [off] : [off, adaptive],
selected: selected,
on: nil,
// Off is the default, as the composer shows it.
explicit: selected != "off"
// Only an explicit Off / On choice is an override; the family default is not.
explicit: mode != .familyDefault
)
}

/// Stores a Native MTP choice through the composer's and Settings' one
/// route. A nil value resets to the default, Off.
/// route. A nil value restores the bundle-aware default.
@MainActor
static func applyNativeMTP(_ value: ModelOptionValue?, nativeMTP: NativeMTP?) async throws {
guard let nativeMTP else { throw ApplyError.unknownOption }
let segment = value?.stringValue ?? "off"
guard segment == "off" || (segment == "auto" && !nativeMTP.blocked) else {
guard value == nil || value?.stringValue != nil,
let mode = NativeMTPSelectionDefault.mode(for: value?.stringValue),
mode != .auto || !nativeMTP.blocked
else {
throw ApplyError.invalidValue
}
var settings = ServerController.runtimeSettingsForConfigureTool().settings
settings.mtp.mode = segment == "off" ? .off : .auto
settings.mtp.mode = mode
settings.mtp.draftTokenLimit = nil
settings.mtp.explicitDepth = nil
_ = await ServerController.applyRuntimeSettingsFromConfigureTool(settings)
_ = await ServerController.applyRuntimeSettingsFromConfigureTool(
settings, mtpSelectionIsFamilyDefault: mode == .familyDefault)
}

/// Stores one choice the way the Mac composer's picker rows do. A nil
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -388,7 +388,7 @@ public enum ServerRuntimeSettingsStore {
// Run and persist engine schema migrations. Current migrations preserve
// explicit percentages and legacy GB; only an unset size is Automatic.
normalized.migrateToCurrentSchema()
// Native MTP now requires opt-in. Never repair an explicit Off to Auto.
// Family defaults apply only to unset choices. Never replace a stored Off.
// Retire only defaults whose provenance was recorded by the old family
// selector, including API-only launches with no chat view present.
normalized.mtp = NativeMTPSelectionDefault.adaptiveSelection(
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -821,7 +821,7 @@ public enum SettingsSearchIndex {
section: "Speculative Decoding",
title: "Speculative Decoding",
keywords: [
"speculative", "mtp", "native mtp", "draft model", "speculative depth", "default off", "on adaptive", "off ar", "adaptive",
"speculative", "mtp", "native mtp", "draft model", "speculative depth", "bundle default", "default", "on adaptive", "off ar", "adaptive",
// The drafter picker lives in this card. Someone who has
// just downloaded a DFlash 2 checkpoint searches for its
// name, not for "speculative decoding".
Expand Down
4 changes: 4 additions & 0 deletions Packages/OsaurusCore/Networking/ServerController.swift
Original file line number Diff line number Diff line change
Expand Up @@ -630,6 +630,10 @@ final class ServerController: ObservableObject {
next: VMLXServerMTPSettings
) -> Bool {
(previous.mode == .off) != (next.mode == .off)
// The family default loads without the MTP head for every family except Qwen3.8 Flash-Next
// (`resolvedMTPLaunch`), so leaving or entering it can change what the load must contain. Without a
// reload, choosing On (Adaptive) for such a bundle kept a head-less resident model and never speculated.
|| (previous.mode == .familyDefault) != (next.mode == .familyDefault)
|| previous.dflash2DrafterPath != next.dflash2DrafterPath
|| previous.dflash2BlockSize != next.dflash2BlockSize
|| previous.keepDraftCacheSeparate != next.keepDraftCacheSeparate
Expand Down
2 changes: 1 addition & 1 deletion Packages/OsaurusCore/Package.resolved

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

2 changes: 1 addition & 1 deletion Packages/OsaurusCore/Package.swift
Original file line number Diff line number Diff line change
Expand Up @@ -327,7 +327,7 @@ let package = Package(
// text-only cache checkpoints use exact active-template prefix proofs.
.package(
url: "https://github.com/osaurus-ai/vmlx-swift",
revision: "648f328ca1c97afc3f51af5da91665fef632c6a5"
revision: "a468573fbe5b2c166d8d1dabea78c205c41e6716"
),
// FluidAudio 0.14.3 added a breaking `language:` parameter to TTS
// calls that osaurus's `TTSService` doesn't pass. Pinning to the
Expand Down
2 changes: 1 addition & 1 deletion Packages/OsaurusCore/Resources/Guide/guide-chat.md
Original file line number Diff line number Diff line change
Expand Up @@ -84,7 +84,7 @@ The Cloud shortlist contains favorites plus the current Cloud model. Use the sta

In the Cloud browser, search by model name or provider and filter by Category or Context. Category tags identify each model’s task; models with available minimum pricing show From credits beneath their name. Selecting a model closes the browser, while starring a model keeps it open. Manage Credits opens your Cloud account controls.

Selecting a model with options reveals an options column beside Provider and Model. When the model exposes a single option the column is named after it (for example Reasoning Effort, whether that's On/Off or Low through Extra High); otherwise it is titled Model options. Every option the model exposes lives there in the same row style: thinking (Default, On, Off), the reasoning level, speculative depth, and any toggles, each with the saved choice or model default checked and a Reset to default link when you have overridden it. Selections keep the card open; click outside or press Escape to close it. There is no separate Model options page.
Selecting a model with options reveals an options column beside Provider and Model. When the model exposes a single option the column is named after it (for example Reasoning Effort, whether that's On/Off or Low through Extra High); otherwise it is titled Model options. Every option the model exposes lives there in the same row style: thinking (Default, On, Off), the reasoning level, speculative decoding (Off or Adaptive), and any toggles, each with the saved choice or model default checked and a Reset to default link when you have overridden it. Selections keep the card open; click outside or press Escape to close it. There is no separate Model options page.

## Credits in chat

Expand Down
29 changes: 16 additions & 13 deletions Packages/OsaurusCore/Resources/Guide/guide-local-models.md
Original file line number Diff line number Diff line change
Expand Up @@ -41,19 +41,22 @@ remain errors. Metadata checks never download or replace model weights.

## Native MTP

Native MTP starts **Off (AR)**. **On (Adaptive)** requests the runtime's adaptive
speculative policy for an eligible local model, including Qwen 27B and Flash Next
in supported affine and JANGH bundles. The runtime chooses its depth; there are
no user depth buttons or native draft-token limits. Sampling remains bundle-driven
unless you explicitly override it.

On is a request, not proof of active speculation. Missing heads, blocked bundles
and missing verified tuning can keep ordinary decoding active. Server → Settings →
**Speculative Decoding** shows the loaded model's actual resolution and reason.
Remote models do not expose this local control. Explicit Off survives reload;
legacy manual depths migrate to Adaptive and are re-evaluated by runtime admission.
A selected **DFlash 2 Drafter** remains a separate external-drafter choice; remove
that selection to stop it. Native MTP Off does not disable an external drafter.
Speculative Decoding defaults to **Default**, resolved from the selected bundle.
Qwen Flash-Next with a usable native MTP head starts **On (Adaptive)**, including
supported affine and JANGH variants. Qwen 27B automatically uses a compatible
bundled DFlash2 drafter when present. A bundle without compatible draft support
runs ordinary autoregressive decoding.

The picker shows **Off (AR)** or **On (Adaptive)**. The runtime chooses depth;
there are no manual depth buttons. Explicit Off disables speculation, including
bundled and selected external drafters, and survives reload. **Reset to default**
restores bundle-aware policy, so a capable bundle can show Adaptive again.
Legacy manual depths migrate to Adaptive and are re-evaluated by runtime admission.

On is a request, not proof of active speculation. Unsupported media and
schema-constrained requests run AR. Server → Settings → **Speculative Decoding**
shows the loaded model's actual resolution and reason. Sampling stays bundle-driven
unless explicitly overridden. Remote models do not expose this local control.

## Apple Foundation Models

Expand Down
32 changes: 19 additions & 13 deletions Packages/OsaurusCore/Resources/Guide/guide-settings.md
Original file line number Diff line number Diff line change
Expand Up @@ -117,19 +117,25 @@ rewrite those server settings. Decisions report `ram_safety_enabled`; when false

## Native MTP

Native MTP starts **Off (AR)**. **On (Adaptive)** requests the runtime's adaptive
speculative policy for an eligible local model, including Qwen 27B and Flash Next
in supported affine and JANGH bundles. The runtime chooses its depth; there are
no user depth buttons or native draft-token limits. Sampling remains bundle-driven
unless you explicitly override it.

On is a request, not proof of active speculation. Missing heads, blocked bundles
and missing verified tuning can keep ordinary decoding active. Server → Settings →
**Speculative Decoding** shows the loaded model's actual resolution and reason.
Remote models do not expose this local control. Explicit Off survives reload;
legacy manual depths migrate to Adaptive and are re-evaluated by runtime admission.
A selected **DFlash 2 Drafter** remains a separate external-drafter choice; remove
that selection to stop it. Native MTP Off does not disable an external drafter.
Speculative Decoding defaults to **Default**, which resolves from the selected
bundle. Qwen Flash-Next bundles with a usable native MTP head start **On
(Adaptive)**, including supported affine and JANGH variants. Qwen 27B uses a
compatible bundled DFlash2 drafter automatically when one is present. A bundle
without compatible draft support runs ordinary autoregressive decoding.

The model picker shows **Off (AR)** or **On (Adaptive)** before weights load.
The runtime chooses depth; there are no manual D1/D2/D3 buttons. Explicit Off
turns speculation off, including external and bundled drafters. Saved explicit
choices survive reload. Legacy manual depths migrate to Adaptive and are
re-evaluated by runtime admission. **Reset to default** restores the selected
bundle's policy, so a capable bundle can show Adaptive again. Included evaluations
and benchmarks inherit that policy unless explicitly overridden.

On is a request, not proof that a particular request uses speculation. Unsupported
media and schema-constrained requests run AR. Server → Settings → **Speculative
Decoding** shows the loaded model's actual resolution and reason. Sampling stays
bundle-driven unless explicitly overridden. Remote models do not expose this
local control.

## Local model memory

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -139,6 +139,9 @@ enum AgentSubagentRunner {
/// a character-based estimate when the provider never reported one.
var completionTokens = 0
var tokensPerSecond: Double?
/// Tokens the runtime actually prefilled for this step (rendered template, tool schemas, history), from the
/// input-token hint or the stats sentinel. `nil` when the provider reported neither.
var promptTokens: Int?
}

/// Run a bounded subagent loop. The caller (kind) owns model resolution,
Expand Down Expand Up @@ -275,7 +278,10 @@ enum AgentSubagentRunner {
// Usage: prompt of the last step (largest composed prompt),
// completions summed. Estimator fallback for providers that
// never emit the stats sentinel.
usage.promptTokens = ContextBudgetManager.estimateTokens(for: effective)
// Prefer the runtime's own count; the message estimator omits the rendered template and the
// separately attached tool schemas (a 4-tool delegation reported 92 against 1,455 prefilled).
usage.promptTokens =
outcome.promptTokens ?? ContextBudgetManager.estimateTokens(for: effective)
var stepCompletion = outcome.completionTokens
if stepCompletion == 0 {
stepCompletion =
Expand Down Expand Up @@ -504,10 +510,15 @@ enum AgentSubagentRunner {
do {
for try await delta in stream {
if let cause = cancelCause() { throw RunCancelled(cause: cause) }
if let inputTokens = StreamingInputTokenHint.decode(delta) {
outcome.promptTokens = inputTokens
continue
}
if let stats = StreamingStatsHint.decode(delta) {
sawStats = true
outcome.completionTokens = stats.tokenCount
outcome.tokensPerSecond = stats.tokensPerSecond
if let inputTokens = stats.inputTokenCount { outcome.promptTokens = inputTokens }
onProgress?(stats.tokenCount, stats.tokensPerSecond)
continue
}
Expand Down
Loading
Loading