Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions Package.swift
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,9 @@ let package = Package(
.executableTarget(name: "KevCheck", dependencies: ["FluidUse"]),
.executableTarget(name: "ClefVisionCheck", dependencies: ["FluidUse"]),
.executableTarget(name: "InternDecisionCheck", dependencies: ["FluidUse"]),
.executableTarget(name: "Decision2Check", dependencies: ["FluidUse"]),
.executableTarget(
name: "IssueTriageDemo", dependencies: ["FluidUse"], exclude: ["README.md", "fetch-issues.sh"]),
.executableTarget(name: "ShortReplyCheck", dependencies: ["FluidUse"]),
.executableTarget(name: "ShortReplyDemo", dependencies: ["FluidUse"], exclude: ["README.md", "demo.sh", "mock-feed"]),
.executableTarget(name: "GLiClassServe", dependencies: ["FluidUse"]),
Expand Down
35 changes: 35 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -210,6 +210,41 @@ moves and switches as `choice` options it matches its 4B teacher (24-6 vs poke-e
heuristic player over 30 battles). Buckets are 512, 640 and 1,024 tokens with int8 weights: about 0.85 GB in memory
with one bucket in use, 1.9 GB to download, the same 90 ms per decision as fp16.

## Decision 2.0

`Decision2Manager` runs vLLM Semantic Router's [Decision 2.0](https://huggingface.co/collections/vllm-sr/decision-20-6ab7cf7bdfb506bf8269cb00)
models (Apache-2.0) on the GPU through Core ML: Kai 0.6B (Qwen3) and Eos 0.8B (Qwen3.5 hybrid). A request is a state
(text or JSON) and any number of named questions (choice, yes/no, score); every question is answered in one call. The
text the questions share is read once and each question continues from it exactly as in its own upstream row, so the
answers are upstream's (typed-decisions TEST, 2,000 decisions: 0 token mismatches, 5 / 4 near-tie flips vs the
upstream PyTorch runtime, all with a top-2 margin under 0.01). Long requests are split into chunks that each repeat the
shared text. The pinned snapshots download from
[FluidInference/decision-2.0-kai-coreml](https://huggingface.co/FluidInference/decision-2.0-kai-coreml) (1.1 GB) and
[FluidInference/decision-2.0-eos-coreml](https://huggingface.co/FluidInference/decision-2.0-eos-coreml) (1.4 GB) on
first use (macOS 15 / iOS 18).

```swift
let model = try await Decision2Manager.load(from: try await Decision2ModelStore.ensure(.kai))
try await model.warm()
let result = try await model.answer(
state: "The order arrived damaged yesterday. The customer has a receipt and asks for a replacement today.",
questions: [
("route", .choice("Which team should handle this request?", [
"returns": "Refunds, replacements and damaged deliveries",
"billing": "Payments, invoices and charges",
])),
("receipt", .yesNo("Does the customer have a receipt?")),
("urgency", .score("How urgent is this request?", ["Routine", "Soon", "Today"])),
])
print(result["route"]!.choice, result["receipt"]!.yes!, result["urgency"]!.score!)
```

On an M5 Pro that request takes 17 ms with Kai and 36 ms with Eos; a five-question typed-decisions request (~1,000
packed tokens) 63 / 113 ms. The upstream PyTorch runtime on the same Mac (MPS, fp32) takes 89 / 306 ms for the first and
418 / 1,595 ms for the second. `swift run -c release Decision2Check parity <model directory> <fixtures.json>` checks
the Swift host against the Python reference. [Sources/IssueTriageDemo](Sources/IssueTriageDemo/README.md) labels 1,000
GitHub issues live with Kai (type, owning team, priority, flags: five decisions per issue in one call).

## Short replies

`ShortReplyManager` drafts one short reply to a social post with a sub-1B model:
Expand Down
108 changes: 108 additions & 0 deletions Sources/Decision2Check/main.swift
Original file line number Diff line number Diff line change
@@ -0,0 +1,108 @@
import FluidUse
import Foundation

/// Decision 2.0 checks.
///
/// swift run -c release Decision2Check parity <model directory> <fixtures.json>
/// swift run -c release Decision2Check example kai|eos (downloads the pinned snapshot, answers the card example)
///
/// Fixtures: typed-decisions TEST requests with the expected token ids of every question row, the Python Core ML
/// runtime's probabilities and the upstream runtime's answers (decision2-kai/swift_fixtures.py).
@main
struct Decision2Check {
struct Fixture: Decodable {
let id: String
let state: String
let questions: String
let ids: [String: [Int]]
let python: [String: [String: Double]]
let upstream: [String: [String: Double]]
}

static func main() async throws {
let arguments = Array(CommandLine.arguments.dropFirst())
guard #available(macOS 15.0, *) else { fatalError("macOS 15 required") }
if arguments.first == "parity", arguments.count == 3 {
try await parity(directory: URL(fileURLWithPath: arguments[1]), fixtures: URL(fileURLWithPath: arguments[2]))
} else if arguments.first == "example", arguments.count == 2, let model = Decision2Model(rawValue: "decision-2.0-\(arguments[1])-coreml") {
try await example(model)
} else {
fputs("usage: Decision2Check parity <model dir> <fixtures.json> | example kai|eos\n", stderr)
exit(2)
}
}

@available(macOS 15.0, *)
static func example(_ model: Decision2Model) async throws {
let directory = try await Decision2ModelStore.ensure(model) { file, bytes in
if bytes > 0 { print("downloaded \(file) (\(bytes / 1_000_000) MB)") }
}
let manager = try await Decision2Manager.load(from: directory)
try await manager.warm()
var times: [Double] = []
var result: Decision2Result!
for _ in 0..<5 {
let start = Date()
result = try await manager.answer(
state: "The order arrived damaged yesterday. The customer has a receipt and asks for a replacement today.",
questions: [
("route", .choice("Which team should handle this request?", [
"returns": "Refunds, replacements and damaged deliveries",
"billing": "Payments, invoices and charges",
"technical": "Product setup and faults",
])),
("receipt", .yesNo("Does the customer have a receipt?")),
("urgency", .score("How urgent is this request?", ["Routine", "Soon", "Today"])),
])
times.append(Date().timeIntervalSince(start) * 1000)
}
for a in result.answers {
let detail = a.yes.map { String(format: "yes %.3f", $0) } ?? a.score.map { String(format: "score %.3f", $0) } ?? a.choice
print("\(a.id): \(detail) \(zip(a.keys, a.probabilities).map { "\($0)=\(String(format: "%.3f", $1))" }.joined(separator: " "))")
}
print("\(manager.modelName): \(result.answers.count) questions, \(result.calls) call, ms per request: "
+ times.map { String(format: "%.1f", $0) }.joined(separator: ", "))
}

@available(macOS 15.0, *)
static func parity(directory: URL, fixtures url: URL) async throws {
let fixtures = try JSONDecoder().decode([Fixture].self, from: Data(contentsOf: url))
let manager = try await Decision2Manager.load(from: directory)
try await manager.warm()
var tokenMismatch = 0, decisions = 0, flipsPython = 0, flipsUpstream = 0
var maxPython = 0.0, maxUpstream = 0.0
var times: [Double] = []
func label(_ probs: [String: Double], _ keys: [String]) -> String {
var best = keys[0]
for k in keys where probs[k]! > probs[best]! { best = k }
return best
}
for fixture in fixtures {
let state = try OrderedJSON.parse(fixture.state)
guard case .object(let members) = try OrderedJSON.parse(fixture.questions) else { fatalError("questions") }
let questions = try members.map { (id: $0.key, question: try Decision2Question(json: $0.value)) }
for (id, question) in questions where try manager.tokens(state: state, question: question) != fixture.ids[id]! {
tokenMismatch += 1
if tokenMismatch <= 3 { print("token mismatch \(fixture.id)/\(id)") }
}
let start = Date()
let result = try await manager.answer(state: state, questions: questions)
times.append(Date().timeIntervalSince(start) * 1000)
for a in result.answers {
decisions += 1
let mine = Dictionary(uniqueKeysWithValues: zip(a.keys, a.probabilities))
let py = fixture.python[a.id]!, up = fixture.upstream[a.id]!
maxPython = max(maxPython, a.keys.map { abs(mine[$0]! - py[$0]!) }.max()!)
maxUpstream = max(maxUpstream, a.keys.map { abs(mine[$0]! - up[$0]!) }.max()!)
flipsPython += label(mine, a.keys) != label(py, a.keys) ? 1 : 0
flipsUpstream += label(mine, a.keys) != label(up, a.keys) ? 1 : 0
}
}
times.sort()
print("\(manager.modelName): \(fixtures.count) requests, \(decisions) decisions")
print("token mismatches vs upstream encoder: \(tokenMismatch)")
print(String(format: "vs Python Core ML runtime: %d flips, max |dp| %.5f", flipsPython, maxPython))
print(String(format: "vs upstream PyTorch: %d flips, max |dp| %.4f", flipsUpstream, maxUpstream))
print(String(format: "request p50 %.1f ms, p95 %.1f ms", times[times.count / 2], times[times.count * 95 / 100]))
}
}
Loading
Loading