diff --git a/Package.swift b/Package.swift index d4e49dc..79387f8 100644 --- a/Package.swift +++ b/Package.swift @@ -56,6 +56,9 @@ let package = Package( .executableTarget(name: "KevCheck", dependencies: ["FluidUse"]), .executableTarget(name: "ClefVisionCheck", dependencies: ["FluidUse"]), .executableTarget(name: "InternDecisionCheck", dependencies: ["FluidUse"]), + .executableTarget(name: "Decision2Check", dependencies: ["FluidUse"]), + .executableTarget( + name: "IssueTriageDemo", dependencies: ["FluidUse"], exclude: ["README.md", "fetch-issues.sh"]), .executableTarget(name: "ShortReplyCheck", dependencies: ["FluidUse"]), .executableTarget(name: "ShortReplyDemo", dependencies: ["FluidUse"], exclude: ["README.md", "demo.sh", "mock-feed"]), .executableTarget(name: "GLiClassServe", dependencies: ["FluidUse"]), diff --git a/README.md b/README.md index eef3ea7..ef34352 100644 --- a/README.md +++ b/README.md @@ -210,6 +210,41 @@ moves and switches as `choice` options it matches its 4B teacher (24-6 vs poke-e heuristic player over 30 battles). Buckets are 512, 640 and 1,024 tokens with int8 weights: about 0.85 GB in memory with one bucket in use, 1.9 GB to download, the same 90 ms per decision as fp16. +## Decision 2.0 + +`Decision2Manager` runs vLLM Semantic Router's [Decision 2.0](https://huggingface.co/collections/vllm-sr/decision-20-6ab7cf7bdfb506bf8269cb00) +models (Apache-2.0) on the GPU through Core ML: Kai 0.6B (Qwen3) and Eos 0.8B (Qwen3.5 hybrid). A request is a state +(text or JSON) and any number of named questions (choice, yes/no, score); every question is answered in one call. The +text the questions share is read once and each question continues from it exactly as in its own upstream row, so the +answers are upstream's (typed-decisions TEST, 2,000 decisions: 0 token mismatches, 5 / 4 near-tie flips vs the +upstream PyTorch runtime, all with a top-2 margin under 0.01). Long requests are split into chunks that each repeat the +shared text. The pinned snapshots download from +[FluidInference/decision-2.0-kai-coreml](https://huggingface.co/FluidInference/decision-2.0-kai-coreml) (1.1 GB) and +[FluidInference/decision-2.0-eos-coreml](https://huggingface.co/FluidInference/decision-2.0-eos-coreml) (1.4 GB) on +first use (macOS 15 / iOS 18). + +```swift +let model = try await Decision2Manager.load(from: try await Decision2ModelStore.ensure(.kai)) +try await model.warm() +let result = try await model.answer( + state: "The order arrived damaged yesterday. The customer has a receipt and asks for a replacement today.", + questions: [ + ("route", .choice("Which team should handle this request?", [ + "returns": "Refunds, replacements and damaged deliveries", + "billing": "Payments, invoices and charges", + ])), + ("receipt", .yesNo("Does the customer have a receipt?")), + ("urgency", .score("How urgent is this request?", ["Routine", "Soon", "Today"])), + ]) +print(result["route"]!.choice, result["receipt"]!.yes!, result["urgency"]!.score!) +``` + +On an M5 Pro that request takes 17 ms with Kai and 36 ms with Eos; a five-question typed-decisions request (~1,000 +packed tokens) 63 / 113 ms. The upstream PyTorch runtime on the same Mac (MPS, fp32) takes 89 / 306 ms for the first and +418 / 1,595 ms for the second. `swift run -c release Decision2Check parity ` checks +the Swift host against the Python reference. [Sources/IssueTriageDemo](Sources/IssueTriageDemo/README.md) labels 1,000 +GitHub issues live with Kai (type, owning team, priority, flags: five decisions per issue in one call). + ## Short replies `ShortReplyManager` drafts one short reply to a social post with a sub-1B model: diff --git a/Sources/Decision2Check/main.swift b/Sources/Decision2Check/main.swift new file mode 100644 index 0000000..d39387a --- /dev/null +++ b/Sources/Decision2Check/main.swift @@ -0,0 +1,108 @@ +import FluidUse +import Foundation + +/// Decision 2.0 checks. +/// +/// swift run -c release Decision2Check parity +/// swift run -c release Decision2Check example kai|eos (downloads the pinned snapshot, answers the card example) +/// +/// Fixtures: typed-decisions TEST requests with the expected token ids of every question row, the Python Core ML +/// runtime's probabilities and the upstream runtime's answers (decision2-kai/swift_fixtures.py). +@main +struct Decision2Check { + struct Fixture: Decodable { + let id: String + let state: String + let questions: String + let ids: [String: [Int]] + let python: [String: [String: Double]] + let upstream: [String: [String: Double]] + } + + static func main() async throws { + let arguments = Array(CommandLine.arguments.dropFirst()) + guard #available(macOS 15.0, *) else { fatalError("macOS 15 required") } + if arguments.first == "parity", arguments.count == 3 { + try await parity(directory: URL(fileURLWithPath: arguments[1]), fixtures: URL(fileURLWithPath: arguments[2])) + } else if arguments.first == "example", arguments.count == 2, let model = Decision2Model(rawValue: "decision-2.0-\(arguments[1])-coreml") { + try await example(model) + } else { + fputs("usage: Decision2Check parity | example kai|eos\n", stderr) + exit(2) + } + } + + @available(macOS 15.0, *) + static func example(_ model: Decision2Model) async throws { + let directory = try await Decision2ModelStore.ensure(model) { file, bytes in + if bytes > 0 { print("downloaded \(file) (\(bytes / 1_000_000) MB)") } + } + let manager = try await Decision2Manager.load(from: directory) + try await manager.warm() + var times: [Double] = [] + var result: Decision2Result! + for _ in 0..<5 { + let start = Date() + result = try await manager.answer( + state: "The order arrived damaged yesterday. The customer has a receipt and asks for a replacement today.", + questions: [ + ("route", .choice("Which team should handle this request?", [ + "returns": "Refunds, replacements and damaged deliveries", + "billing": "Payments, invoices and charges", + "technical": "Product setup and faults", + ])), + ("receipt", .yesNo("Does the customer have a receipt?")), + ("urgency", .score("How urgent is this request?", ["Routine", "Soon", "Today"])), + ]) + times.append(Date().timeIntervalSince(start) * 1000) + } + for a in result.answers { + let detail = a.yes.map { String(format: "yes %.3f", $0) } ?? a.score.map { String(format: "score %.3f", $0) } ?? a.choice + print("\(a.id): \(detail) \(zip(a.keys, a.probabilities).map { "\($0)=\(String(format: "%.3f", $1))" }.joined(separator: " "))") + } + print("\(manager.modelName): \(result.answers.count) questions, \(result.calls) call, ms per request: " + + times.map { String(format: "%.1f", $0) }.joined(separator: ", ")) + } + + @available(macOS 15.0, *) + static func parity(directory: URL, fixtures url: URL) async throws { + let fixtures = try JSONDecoder().decode([Fixture].self, from: Data(contentsOf: url)) + let manager = try await Decision2Manager.load(from: directory) + try await manager.warm() + var tokenMismatch = 0, decisions = 0, flipsPython = 0, flipsUpstream = 0 + var maxPython = 0.0, maxUpstream = 0.0 + var times: [Double] = [] + func label(_ probs: [String: Double], _ keys: [String]) -> String { + var best = keys[0] + for k in keys where probs[k]! > probs[best]! { best = k } + return best + } + for fixture in fixtures { + let state = try OrderedJSON.parse(fixture.state) + guard case .object(let members) = try OrderedJSON.parse(fixture.questions) else { fatalError("questions") } + let questions = try members.map { (id: $0.key, question: try Decision2Question(json: $0.value)) } + for (id, question) in questions where try manager.tokens(state: state, question: question) != fixture.ids[id]! { + tokenMismatch += 1 + if tokenMismatch <= 3 { print("token mismatch \(fixture.id)/\(id)") } + } + let start = Date() + let result = try await manager.answer(state: state, questions: questions) + times.append(Date().timeIntervalSince(start) * 1000) + for a in result.answers { + decisions += 1 + let mine = Dictionary(uniqueKeysWithValues: zip(a.keys, a.probabilities)) + let py = fixture.python[a.id]!, up = fixture.upstream[a.id]! + maxPython = max(maxPython, a.keys.map { abs(mine[$0]! - py[$0]!) }.max()!) + maxUpstream = max(maxUpstream, a.keys.map { abs(mine[$0]! - up[$0]!) }.max()!) + flipsPython += label(mine, a.keys) != label(py, a.keys) ? 1 : 0 + flipsUpstream += label(mine, a.keys) != label(up, a.keys) ? 1 : 0 + } + } + times.sort() + print("\(manager.modelName): \(fixtures.count) requests, \(decisions) decisions") + print("token mismatches vs upstream encoder: \(tokenMismatch)") + print(String(format: "vs Python Core ML runtime: %d flips, max |dp| %.5f", flipsPython, maxPython)) + print(String(format: "vs upstream PyTorch: %d flips, max |dp| %.4f", flipsUpstream, maxUpstream)) + print(String(format: "request p50 %.1f ms, p95 %.1f ms", times[times.count / 2], times[times.count * 95 / 100])) + } +} diff --git a/Sources/FluidUse/Decision2/Decision2Manager.swift b/Sources/FluidUse/Decision2/Decision2Manager.swift new file mode 100644 index 0000000..4468666 --- /dev/null +++ b/Sources/FluidUse/Decision2/Decision2Manager.swift @@ -0,0 +1,439 @@ +@preconcurrency import CoreML +import Foundation + +/// Decision 2.0 (vLLM Semantic Router: Kai 0.6B on Qwen3, Eos 0.8B on Qwen3.5) on Core ML: every question of a +/// request in one call. The questions' rows share a token prefix (the context); it runs once, and each question's +/// suffix continues from it exactly as in its own upstream row. Long requests are split into chunks that each repeat +/// the prefix, on whichever function needs the fewest milliseconds. +/// +/// `directory` is a FluidInference/decision-2.0-*-coreml snapshot (`Decision2ModelStore.ensure`). +@available(macOS 15.0, iOS 18.0, *) +public final class Decision2Manager: Sendable { + static let maxOptions = 255 + static let maskValue: Float16 = -1e4 + static let chunk = 64 + static let lags = 3 + + enum Backbone: Sendable { case qwen3, qwen35 } + + struct Function: Sendable { + let name: String + /// Qwen3: L = packed tokens, S = 0. Qwen3.5: S = prefix tokens, L = packed question tokens. + let s: Int + let l: Int + let n: Int + let costMs: Double + } + + struct Row { + let ids: [Int] + let candidates: [Int] + let query: Int + } + + public let modelName: String + private let backbone: Backbone + private let compiled: URL + private let computeUnits: MLComputeUnits + private let tokenizer: QwenBPETokenizer + private let padID: Int + private let rotaryDim: Int + private let ropeTheta: Double + private let scoreBias: [Int: [Double]] + private let functions: [Function] + private let cache = FunctionCache() + + actor FunctionCache { + private var models: [String: MLModel] = [:] + + func model(_ name: String, at url: URL, units: MLComputeUnits) async throws -> MLModel { + if let model = models[name] { return model } + let configuration = MLModelConfiguration() + configuration.computeUnits = units + configuration.functionName = name + let model = try await MLModel.load(contentsOf: url, configuration: configuration) + models[name] = model + return model + } + } + + public static func load( + from directory: URL, computeUnits: MLComputeUnits = .cpuAndGPU + ) async throws -> Decision2Manager { + guard + let config = try JSONSerialization.jsonObject( + with: Data(contentsOf: directory.appendingPathComponent("coreml_config.json"))) as? [String: Any], + let modelName = config["model_name"] as? String, let package = config["package"] as? String, + let costs = config["functions"] as? [String: NSNumber], let pad = config["pad_token_id"] as? Int + else { throw Decision2Error.invalidAsset("coreml_config.json is missing fields") } + let backbone: Backbone = (config["backbone"] as? String) == "qwen3_5" ? .qwen35 : .qwen3 + var rotary = 0 + var theta = 0.0 + if backbone == .qwen35 { + guard let rope = config["rope"] as? [String: Any], let dim = rope["rotary_dim"] as? Int, + let base = (rope["rope_theta"] as? NSNumber)?.doubleValue + else { throw Decision2Error.invalidAsset("coreml_config.json has no rope") } + rotary = dim + theta = base + } + var functions: [Function] = [] + for (name, cost) in costs { + var dims: [Character: Int] = [:] + for part in name.split(separator: "_") { dims[part.first!] = Int(part.dropFirst()) } + if backbone == .qwen3, let l = dims["L"], let n = dims["N"] { + functions.append(Function(name: name, s: 0, l: l, n: n, costMs: cost.doubleValue)) + } else if backbone == .qwen35, let s = dims["S"], let c = dims["C"], let n = dims["N"] { + functions.append(Function(name: name, s: s, l: c, n: n, costMs: cost.doubleValue)) + } else { + throw Decision2Error.invalidAsset("unexpected function name \(name)") + } + } + var bias: [Int: [Double]] = [:] + let biasURL = directory.appendingPathComponent("score_bias.json") + if FileManager.default.fileExists(atPath: biasURL.path) { + guard let report = try JSONSerialization.jsonObject(with: Data(contentsOf: biasURL)) as? [String: Any], + let offsets = report["offsets"] as? [String: [NSNumber]] + else { throw Decision2Error.invalidAsset("score_bias.json") } + for (levels, values) in offsets { bias[Int(levels)!] = values.map(\.doubleValue) } + } + var compiled = directory.appendingPathComponent( + (package as NSString).deletingPathExtension + ".mlmodelc") + if !FileManager.default.fileExists(atPath: compiled.path) { + compiled = try await KevManager.compiled(directory.appendingPathComponent(package)) + } + return Decision2Manager( + modelName: modelName, backbone: backbone, compiled: compiled, computeUnits: computeUnits, + tokenizer: try QwenBPETokenizer(tokenizerJsonURL: directory.appendingPathComponent("tokenizer.json")), + padID: pad, rotaryDim: rotary, ropeTheta: theta, scoreBias: bias, + functions: functions.sorted { ($0.s, $0.l) < ($1.s, $1.l) }) + } + + init( + modelName: String, backbone: Backbone, compiled: URL, computeUnits: MLComputeUnits, + tokenizer: QwenBPETokenizer, padID: Int, rotaryDim: Int, ropeTheta: Double, scoreBias: [Int: [Double]], + functions: [Function] + ) { + self.modelName = modelName + self.backbone = backbone + self.compiled = compiled + self.computeUnits = computeUnits + self.tokenizer = tokenizer + self.padID = padID + self.rotaryDim = rotaryDim + self.ropeTheta = ropeTheta + self.scoreBias = scoreBias + self.functions = functions + } + + /// Loads every function and runs it once, so no request pays for loading or for the first GPU dispatch. Largest + /// first: switching functions costs a re-setup on the next call, so the small, common ones are left hot. + public func warm() async throws { + let row = Row(ids: [padID, padID, padID], candidates: [1], query: 2) + for function in functions.reversed() { _ = try await run([row], on: function) } + } + + public func answer(state: OrderedJSON, questions: [(id: String, question: Decision2Question)]) async throws + -> Decision2Result + { + try Self.validate(questions) + guard !questions.isEmpty else { return Decision2Result(answers: [], inputTokens: 0, calls: 0) } + let rows = try questions.map { try encode(state: state, question: $0.question) } + var logits = [[Double]](repeating: [], count: rows.count) + let calls = try plan(rows) + for (function, group) in calls { + let out = try await run(group.map { rows[$0] }, on: function) + for (slot, index) in group.enumerated() { logits[index] = out[slot] } + } + let answers = zip(questions, logits).map { item, values -> Decision2Answer in + let options = item.question.options + var z = values + if case .score = item.question, let offsets = scoreBias[z.count] { + z = zip(z, offsets).map { $0 + $1 } + } + let peak = z.max() ?? 0 + let e = z.map { exp($0 - peak) } + let sum = e.reduce(0, +) + return Decision2Answer( + id: item.id, type: item.question.taskType, keys: options.map(\.key), probabilities: e.map { $0 / sum }) + } + return Decision2Result(answers: answers, inputTokens: rows.reduce(0) { $0 + $1.ids.count }, calls: calls.count) + } + + /// Upstream's question rules: unique ids, 2…255 choice options, 2…10 score levels (yes/no always has 2). + static func validate(_ questions: [(id: String, question: Decision2Question)]) throws { + var seen = Set() + for (id, q) in questions { + guard seen.insert(id).inserted else { throw Decision2Error.invalidQuestion("duplicate question id \(id)") } + switch q { + case .choice(_, let options) where !(2...maxOptions).contains(options.count): + throw Decision2Error.invalidQuestion("choice \(id) needs 2 to \(maxOptions) options, got \(options.count)") + case .score(_, let levels) where !(2...10).contains(levels.count): + throw Decision2Error.invalidQuestion("score \(id) needs 2 to 10 levels, got \(levels.count)") + default: break + } + } + } + + /// The token ids of one question's upstream row (for parity checks). + public func tokens(state: OrderedJSON, question: Decision2Question) throws -> [Int] { + try Self.validate([("q", question)]) + return try encode(state: state, question: question).ids + } + + // MARK: Prompt (upstream decision_model.segments / encode) + + func encode(state: OrderedJSON, question: Decision2Question) throws -> Row { + let prefix = + "Context:\n\(state.decisionPayload)\n\nTask type: \(question.taskType)\nQuestion:\n" + + "\(question.instructions.decisionPayload)\nOptions:" + var ids = try tokenizer.encode(prefix) + var candidates: [Int] = [] + for option in question.options { + let body = OrderedJSON.object([("key", .string(option.key)), ("description", option.description)]) + let piece = try tokenizer.encode("\n") + guard !piece.isEmpty else { throw Decision2Error.invalidQuestion("empty option") } + ids += piece + candidates.append(ids.count - 1) + } + ids += try tokenizer.encode( + "\n\nSelect the single option best supported by the context and instructions.\nDecision:") + return Row(ids: ids, candidates: candidates, query: ids.count - 1) + } + + // MARK: Planning + + /// Common token prefix of the rows, cut before the first option endpoint and the last token of the shortest. + static func prefixLength(_ rows: [Row]) -> Int { + let limit = min(rows.map { $0.candidates[0] }.min()!, rows.map { $0.ids.count }.min()! - 1) + let first = rows[0].ids + for i in 0.. Bool { + let p = Self.prefixLength(rows) + let options = rows.reduce(0) { $0 + $1.candidates.count } + let suffix = rows.reduce(0) { $0 + $1.ids.count - p } + switch backbone { + case .qwen3: return p + suffix <= function.l && options <= function.n + case .qwen35: return p <= function.s && suffix <= function.l && options <= function.n + } + } + + /// One call in the smallest function that fits, or greedy in-order chunks on the cheapest function. + func plan(_ rows: [Row]) throws -> [(Function, [Int])] { + var best: (Double, [(Function, [Int])])? + candidates: for function in functions { + var groups: [[Int]] = [] + var group: [Int] = [] + for index in rows.indices { + if fits((group + [index]).map { rows[$0] }, function) { + group.append(index) + continue + } + guard !group.isEmpty, fits([rows[index]], function) else { continue candidates } + groups.append(group) + group = [index] + } + groups.append(group) + let cost = Double(groups.count) * function.costMs + if best == nil || cost < best!.0 { best = (cost, groups.map { (function, $0) }) } + } + guard let best else { + throw Decision2Error.tooLong("a question exceeds the largest function \(functions.last?.name ?? "")") + } + return best.1 + } + + // MARK: Inputs and prediction + + func run(_ rows: [Row], on function: Function) async throws -> [[Double]] { + let model: MLModel + do { + model = try await cache.model(function.name, at: compiled, units: computeUnits) + } catch { + throw Decision2Error.invalidAsset("function \(function.name): \(error.localizedDescription)") + } + let (features, owners) = try backbone == .qwen3 ? qwen3Inputs(rows, function) : qwen35Inputs(rows, function) + let output = try await model.prediction(from: MLDictionaryFeatureProvider(dictionary: features)) + guard let logits = output.featureValue(for: "logits")?.multiArrayValue else { + throw Decision2Error.invalidOutput("logits") + } + var per = [[Double]](repeating: [], count: rows.count) + for (slot, owner) in owners.enumerated() { per[owner].append(logits[slot].doubleValue) } + return per + } + + /// Kai: [1, L] ids and positions, additive [1, 1, L, L] mask (a suffix sees the prefix and itself, causally). + func qwen3Inputs(_ rows: [Row], _ function: Function) throws -> ([String: Any], [Int]) { + let L = function.l + let p = Self.prefixLength(rows) + var ids = Array(rows[0].ids[0.. ([String: Any], [Int]) { + let S = function.s + let C = function.l + let T = S + C + let lags = Self.lags + let p = Self.prefixLength(rows) + let inputIDs = try array([1, T], .int32) + let valid = try array([S], .float16) + let tail = try array([lags, S], .float16) + let segment = try array([C, C], .float16) + let keep = try array([lags, C], .float16) + let lagTail = try array([lags, C, lags], .float16) + let idp = inputIDs.dataPointer.assumingMemoryBound(to: Int32.self) + let vp = valid.dataPointer.assumingMemoryBound(to: Float16.self) + let tp = tail.dataPointer.assumingMemoryBound(to: Float16.self) + let sp = segment.dataPointer.assumingMemoryBound(to: Float16.self) + let kp = keep.dataPointer.assumingMemoryBound(to: Float16.self) + let lp = lagTail.dataPointer.assumingMemoryBound(to: Float16.self) + for i in 0..= 0 { tp[i * S + p - lags + i] = 1 } + for i in 0..= s { + kp[(s - 1) * C + start + k] = 1 + } else { + lp[((s - 1) * C + start + k) * lags + lags + k - s] = 1 + } + } + } + for c in row.candidates { + candidates.append(Int32(start + c - p)) + queries.append(Int32(start + row.query - p)) + owners.append(j) + } + start += n + } + // chunked delta-rule masks, from the segment matrix + let chunks = C / Self.chunk + let segChunks = try array([chunks, Self.chunk, Self.chunk], .float16) + let cont = try array([C], .float16) + let lastSeg = try array([chunks, Self.chunk], .float16) + let scp = segChunks.dataPointer.assumingMemoryBound(to: Float16.self) + let cp = cont.dataPointer.assumingMemoryBound(to: Float16.self) + let lsp = lastSeg.dataPointer.assumingMemoryBound(to: Float16.self) + for c in 0.. (MLMultiArray, MLMultiArray) { + let cos = try array([positions.count, rotaryDim], .float16) + let sin = try array([positions.count, rotaryDim], .float16) + let c = cos.dataPointer.assumingMemoryBound(to: Float16.self) + let s = sin.dataPointer.assumingMemoryBound(to: Float16.self) + let half = rotaryDim / 2 + for (row, position) in positions.enumerated() { + for i in 0.. MLMultiArray { + guard values.count <= n else { throw Decision2Error.tooLong("\(values.count) options > \(n) slots") } + let out = try array([n], .int32) + let pointer = out.dataPointer.assumingMemoryBound(to: Int32.self) + for (i, v) in values.enumerated() { pointer[i] = v } + return out + } + + private func array(_ dims: [Int], _ type: MLMultiArrayDataType) throws -> MLMultiArray { + let array = try MLMultiArray(shape: dims.map { NSNumber(value: $0) }, dataType: type) + switch type { + case .int32: array.dataPointer.initializeMemory(as: Int32.self, repeating: 0, count: array.count) + default: array.dataPointer.initializeMemory(as: Float16.self, repeating: 0, count: array.count) + } + return array + } +} diff --git a/Sources/FluidUse/Decision2/Decision2ModelStore.swift b/Sources/FluidUse/Decision2/Decision2ModelStore.swift new file mode 100644 index 0000000..4ddd29b --- /dev/null +++ b/Sources/FluidUse/Decision2/Decision2ModelStore.swift @@ -0,0 +1,127 @@ +import CryptoKit +import Foundation + +/// A published Decision 2.0 Core ML snapshot, laid out as `Decision2Manager.load(from:)` expects. +public enum Decision2Model: String, Sendable, CaseIterable { + /// Decision-2.0-Kai-0.6B (Qwen3): fastest, 1.1 GB fp16 (FluidInference/decision-2.0-kai-coreml). + case kai = "decision-2.0-kai-coreml" + /// Decision-2.0-Eos-0.8B (Qwen3.5 hybrid): 1.4 GB fp16 (FluidInference/decision-2.0-eos-coreml). + case eos = "decision-2.0-eos-coreml" + + var repository: String { "FluidInference/\(rawValue)" } + + var revision: String { + switch self { + case .kai: "73344136ce1d0b68be69ebc5ad2c014ff868ea66" + case .eos: "7bf832369ecdca988a2376e0212b65b0680163f0" + } + } + + var assets: [Decision2ModelStore.Asset] { + switch self { + case .kai: Decision2ModelStore.kaiAssets + case .eos: Decision2ModelStore.eosAssets + } + } +} + +/// Downloads a pinned, checksummed Decision 2.0 snapshot into the FluidUse cache. +public enum Decision2ModelStore { + public typealias Progress = @Sendable (_ file: String, _ bytes: Int64) -> Void + + struct Asset { + let path: String + let sha256: String + } + + static func package(_ name: String, model: String, weights: String, manifest: String) -> [Asset] { + [ + Asset(path: "\(name).mlpackage/Data/com.apple.CoreML/model.mlmodel", sha256: model), + Asset(path: "\(name).mlpackage/Data/com.apple.CoreML/weights/weight.bin", sha256: weights), + Asset(path: "\(name).mlpackage/Manifest.json", sha256: manifest), + ] + } + + static let kaiAssets: [Asset] = + [ + Asset(path: "coreml_config.json", sha256: "d24a234e8e471225c5ae7191527d1a2a675132e48ad1c14ca230c823ec90bfcb"), + Asset(path: "score_bias.json", sha256: "7d3a060fe1823aa7fe759776df3db872408de580f227120fbe08b9eef53dcf6e"), + Asset(path: "tokenizer.json", sha256: "be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506"), + ] + + package( + "Decision2KaiPacked", model: "7c4ee2802024c09d66b66b3d7a50f57e3195fd3c175facfd2902df54825abc77", + weights: "153390b2b6722768b8150b148f2aa6f5fcc960ede365754903a22ed60f853994", + manifest: "419d3a2521d3886ce608df0204761de533059261ede0276473804daf46216624") + + static let eosAssets: [Asset] = + [ + Asset(path: "coreml_config.json", sha256: "99be4d683b0c4163c00e103c0ff024e353ba247e61499ec2743126a181a75acf"), + Asset(path: "tokenizer.json", sha256: "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523"), + ] + + package( + "Decision2EosPacked", model: "e8a434580763e03408199536441dcfabbb3e58f479e2ef9e823e69d8b497e443", + weights: "680030e081b927dc5902d522e0df4f24bdf3a13910c228625d7fd98ac324e939", + manifest: "8933a45677029671d29f614359c9f9d8f2a4394ac30cfd38ee1b765d880c071b") + + /// Ensure the snapshot exists in the FluidUse cache and return its directory. Files are checksummed once per + /// pinned revision; later launches only check that they are present. + public static func ensure( + _ model: Decision2Model = .kai, cacheDirectory: URL? = nil, progress: Progress? = nil + ) async throws -> URL { + let root = cacheDirectory ?? LayaModelStore.defaultCacheDirectory() + let directory = root.appendingPathComponent(model.rawValue, isDirectory: true) + let manager = FileManager.default + let verified = directory.appendingPathComponent(".verified-\(model.revision)") + if manager.fileExists(atPath: verified.path), + model.assets.allSatisfy({ manager.fileExists(atPath: directory.appendingPathComponent($0.path).path) }) + { + return directory + } + try manager.createDirectory(at: directory, withIntermediateDirectories: true) + for asset in model.assets { + try Task.checkCancellation() + let destination = directory.appendingPathComponent(asset.path) + if manager.fileExists(atPath: destination.path), try checksum(of: destination) == asset.sha256 { + continue + } + try manager.createDirectory(at: destination.deletingLastPathComponent(), withIntermediateDirectories: true) + progress?(asset.path, 0) + let escaped = asset.path.addingPercentEncoding(withAllowedCharacters: .urlPathAllowed) ?? asset.path + guard + let url = URL(string: "https://huggingface.co/\(model.repository)/resolve/\(model.revision)/\(escaped)") + else { + throw Decision2Error.invalidAsset("Invalid Hugging Face asset URL for \(asset.path)") + } + let (temporary, response) = try await URLSession.shared.download(from: url) + defer { try? manager.removeItem(at: temporary) } + guard let http = response as? HTTPURLResponse, http.statusCode == 200 else { + throw Decision2Error.invalidAsset("Download failed for \(asset.path)") + } + let actual = try checksum(of: temporary) + guard actual == asset.sha256 else { + throw Decision2Error.invalidAsset( + "Checksum mismatch for \(asset.path): expected \(asset.sha256), got \(actual)") + } + let size = (try manager.attributesOfItem(atPath: temporary.path)[.size] as? NSNumber)?.int64Value ?? 0 + try LayaModelStore.installDownloadedFile(temporary, at: destination) + // A package file changed: the .mlmodelc compiled from the old package beside it must not outlive it. + if let range = asset.path.range(of: ".mlpackage/") { + let compiled = directory.appendingPathComponent(String(asset.path[.. String { + let handle = try FileHandle(forReadingFrom: file) + defer { try? handle.close() } + var digest = SHA256() + while let chunk = try handle.read(upToCount: 4_194_304), !chunk.isEmpty { + digest.update(data: chunk) + } + return digest.finalize().map { String(format: "%02x", $0) }.joined() + } +} diff --git a/Sources/FluidUse/Decision2/Decision2Types.swift b/Sources/FluidUse/Decision2/Decision2Types.swift new file mode 100644 index 0000000..36647d9 --- /dev/null +++ b/Sources/FluidUse/Decision2/Decision2Types.swift @@ -0,0 +1,155 @@ +import Foundation + +/// One typed question for a Decision 2.0 model (vLLM Semantic Router), as in its System One `questions` object. +/// Instructions and descriptions are text or structured JSON, rendered exactly as the upstream runtime renders them. +public enum Decision2Question: Sendable { + /// Pick one of `options` (key, description); keys are reported in this order. A `.null` description is allowed. + case choice(instructions: OrderedJSON, options: [(key: String, description: OrderedJSON)]) + /// Yes or no; reported under `false`, `true` in that order. Upstream's defaults are "No" / "Yes". + case noul(instructions: OrderedJSON, no: OrderedJSON = "No", yes: OrderedJSON = "Yes") + /// One of 2…10 ordered levels; keys are `"0"`, `"1"`, …. + case score(instructions: OrderedJSON, levels: [OrderedJSON]) + + public static func choice(_ instructions: String, _ options: KeyValuePairs) -> Decision2Question { + .choice(instructions: .string(instructions), options: options.map { ($0.key, .string($0.value)) }) + } + + public static func yesNo(_ instructions: String) -> Decision2Question { + .noul(instructions: .string(instructions)) + } + + public static func score(_ instructions: String, _ levels: [String]) -> Decision2Question { + .score(instructions: .string(instructions), levels: levels.map { .string($0) }) + } + + /// A question in the System One JSON form: `{"type": "choice" | "noul" | "score", "instructions": …, + /// "criteria": …}`. Choice criteria keep their key order. + public init(json: OrderedJSON) throws { + guard case .object(let members) = json else { throw Decision2Error.invalidQuestion("question is not an object") } + let field = { (name: String) in members.first(where: { $0.key == name })?.value } + guard case .string(let type)? = field("type") else { throw Decision2Error.invalidQuestion("missing type") } + guard let instructions = field("instructions"), instructions != .string(""), instructions != .null else { + throw Decision2Error.invalidQuestion("missing instructions") + } + switch (type, field("criteria")) { + case ("choice", .object(let options)?) where (2...255).contains(options.count): + self = .choice(instructions: instructions, options: options.map { (key: $0.key, description: $0.value) }) + case ("noul", nil), ("noul", .null?): + self = .noul(instructions: instructions) + case ("noul", .object(let sides)?) where Set(sides.map(\.key)).isSubset(of: ["false", "true"]): + let side = { (key: String) in sides.first(where: { $0.key == key })?.value } + self = .noul(instructions: instructions, no: side("false") ?? "No", yes: side("true") ?? "Yes") + case ("score", .array(let levels)?) where (2...10).contains(levels.count): + self = .score(instructions: instructions, levels: levels) + default: + throw Decision2Error.invalidQuestion("unsupported \(type) criteria") + } + } + + var taskType: String { + switch self { + case .choice: "choice" + case .noul: "noul" + case .score: "score" + } + } + + var instructions: OrderedJSON { + switch self { + case .choice(let instructions, _), .noul(let instructions, _, _), .score(let instructions, _): instructions + } + } + + /// `(key, description)` pairs in the order the model reads them. + var options: [(key: String, description: OrderedJSON)] { + switch self { + case .choice(_, let options): options + case .noul(_, let no, let yes): [("false", no), ("true", yes)] + case .score(_, let levels): levels.enumerated().map { (String($0.offset), $0.element) } + } + } +} + +/// One answer, with the same fields as upstream `system_one()`. +public struct Decision2Answer: Sendable { + public let id: String + /// `choice`, `noul` or `score`. + public let type: String + public let keys: [String] + /// Softmax over the options (after the package's Score offsets, when it has them). + public let probabilities: [Double] + + /// The most probable key; ties go to the earlier option, as upstream. + public var choice: String { + var best = 0 + for i in probabilities.indices where probabilities[i] > probabilities[best] { best = i } + return keys[best] + } + /// Probability of `true` for a yes/no question. + public var yes: Double? { type == "noul" ? probabilities[keys.firstIndex(of: "true")!] : nil } + /// Expected level for a Score question. + public var score: Double? { + type == "score" ? zip(keys, probabilities).reduce(0) { $0 + Double(Int($1.0)!) * $1.1 } : nil + } + /// 1 − normalized entropy (upstream's product confidence); nil for yes/no. + public var confidence: Double? { + guard type != "noul" else { return nil } + let entropy = -probabilities.filter { $0 > 0 }.reduce(0) { $0 + $1 * log($1) } + return max(0, min(1, 1 - entropy / log(Double(keys.count)))) + } + + public func probability(_ key: String) -> Double? { keys.firstIndex(of: key).map { probabilities[$0] } } +} + +public struct Decision2Result: Sendable { + /// In the order the questions were given. + public let answers: [Decision2Answer] + /// Upstream's `usage.input_tokens`: the sum of every question's own row. + public let inputTokens: Int + /// Core ML calls this request took (1 unless it had to be split). + public let calls: Int + public subscript(id: String) -> Decision2Answer? { answers.first(where: { $0.id == id }) } +} + +public enum Decision2Error: Error, LocalizedError, Sendable { + case invalidQuestion(String) + case invalidAsset(String) + case invalidOutput(String) + case tooLong(String) + + public var errorDescription: String? { + switch self { + case .invalidQuestion(let reason): "Invalid Decision 2.0 question: \(reason)" + case .invalidAsset(let reason): "Invalid Decision 2.0 asset: \(reason)" + case .invalidOutput(let reason): "Invalid Decision 2.0 output: \(reason)" + case .tooLong(let reason): "Decision 2.0 request too long: \(reason)" + } + } +} + +extension OrderedJSON { + /// Python `json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":"))`, the form Decision 2.0 + /// renders structured payloads in. Keys sort by code point, as Python's `str` ordering. + public var canonical: String { + switch self { + case .object(let members): + let sorted = members.sorted { a, b in + a.key.unicodeScalars.lexicographicallyPrecedes(b.key.unicodeScalars) { $0.value < $1.value } + } + return "{" + sorted.map { OrderedJSON.pythonString($0.key) + ":" + $0.value.canonical }.joined(separator: ",") + + "}" + case .array(let items): return "[" + items.map(\.canonical).joined(separator: ",") + "]" + case .string(let text): return OrderedJSON.pythonString(text) + case .integer(let value): return String(value) + case .number(let value): return OrderedJSON.pythonFloat(value) + case .bool(let value): return value ? "true" : "false" + case .null: return "null" + } + } + + /// Upstream `_payload`: text verbatim, anything else canonical JSON. + var decisionPayload: String { + if case .string(let text) = self { return text } + return canonical + } +} diff --git a/Sources/IssueTriageDemo/ContentView.swift b/Sources/IssueTriageDemo/ContentView.swift new file mode 100644 index 0000000..7f755bc --- /dev/null +++ b/Sources/IssueTriageDemo/ContentView.swift @@ -0,0 +1,512 @@ +import SwiftUI + +extension Color { + init(hex: UInt32, alpha: Double = 1) { + self.init( + .sRGB, red: Double((hex >> 16) & 0xff) / 255, green: Double((hex >> 8) & 0xff) / 255, + blue: Double(hex & 0xff) / 255, opacity: alpha) + } +} + +/// GitHub dark palette, as in the web demo. +enum Theme { + static let bg = Color(hex: 0x0d1117) + static let bg2 = Color(hex: 0x010409) + static let panel = Color(hex: 0x151b23) + static let line = Color(hex: 0x3d444d) + static let line2 = Color(hex: 0x2a313c) + static let ink = Color(hex: 0xf0f6fc) + static let dim = Color(hex: 0x9198a1) + static let link = Color(hex: 0x4493f8) + static let green = Color(hex: 0x3fb950) + static let purple = Color(hex: 0xab7df8) + static let btn = Color(hex: 0x212830) + static let live = Color(.sRGB, red: 137 / 255, green: 87 / 255, blue: 229 / 255, opacity: 0.16) + static let gradient = LinearGradient( + colors: [Color(hex: 0x8957e5), Color(hex: 0x4493f8)], startPoint: .leading, endPoint: .trailing) +} + +struct ContentView: View { + @EnvironmentObject var model: TriageModel + + var body: some View { + VStack(spacing: 0) { + Header(open: model.openCount) + VStack(spacing: 16) { + Toolbar() + HStack(alignment: .top, spacing: 16) { + Sidebar(stats: model.stats) + VStack(spacing: 16) { + if model.phase == .running || model.phase == .finished { + StatsBar(stats: model.stats, total: model.rows.count) + } + IssueBox(rows: model.rows, open: model.openCount, scroll: model.scroll) + } + } + .frame(maxHeight: .infinity, alignment: .top) + } + .padding(24) + .frame(maxWidth: 1560) + .frame(maxWidth: .infinity, maxHeight: .infinity) + } + .background(Theme.bg) + .foregroundStyle(Theme.ink) + .font(.system(size: 14)) + .preferredColorScheme(.dark) + .sheet(item: $model.selected) { row in DetailSheet(row: row) } + } +} + +// MARK: - Header and toolbar + +struct Header: View { + let open: Int + + var body: some View { + VStack(alignment: .leading, spacing: 10) { + HStack(spacing: 10) { + Image(systemName: "book.closed") + .font(.system(size: 13)) + .foregroundStyle(Theme.dim) + .frame(width: 32, height: 32) + .overlay(Circle().stroke(Theme.line)) + Text("routerlabs").font(.system(size: 16)) + Text("/").font(.system(size: 16)).foregroundStyle(Theme.dim) + Text("semantic-switch").font(.system(size: 16, weight: .semibold)) + Text("Public") + .font(.system(size: 12)).foregroundStyle(Theme.dim) + .padding(.horizontal, 7) + .overlay(Capsule().stroke(Theme.line)) + .padding(.leading, 6) + Spacer() + Text("Local demo · labels by an on-device model") + .font(.system(size: 11)).foregroundStyle(Theme.dim) + .padding(.horizontal, 8).padding(.vertical, 1) + .overlay(RoundedRectangle(cornerRadius: 6).stroke(Theme.line, style: StrokeStyle(dash: [3, 2]))) + } + .frame(height: 32) + HStack(spacing: 6) { + tab("Code") + HStack(spacing: 6) { + Image(systemName: "smallcircle.filled.circle") + Text("Issues").fontWeight(.semibold) + Text("\(open)") + .font(.system(size: 12, weight: .medium)) + .padding(.horizontal, 6) + .background(Capsule().fill(Color(hex: 0x2f3742))) + } + .padding(.horizontal, 10).padding(.vertical, 8) + .overlay(alignment: .bottom) { Rectangle().fill(Color(hex: 0xf78166)).frame(height: 2) } + ForEach(["Pull requests", "Discussions", "Actions", "Projects", "Insights"], id: \.self) { tab($0) } + } + } + .padding(.horizontal, 24).padding(.top, 12) + .frame(maxWidth: .infinity, alignment: .leading) + .background(Theme.bg2) + .overlay(alignment: .bottom) { Rectangle().fill(Theme.line2).frame(height: 1) } + } + + func tab(_ name: String) -> some View { + Text(name).padding(.horizontal, 10).padding(.vertical, 8) + } +} + +struct Toolbar: View { + @EnvironmentObject var model: TriageModel + + var body: some View { + HStack(spacing: 8) { + Text("is:issue sort:created-desc") + .foregroundStyle(Theme.dim) + .padding(.horizontal, 12).padding(.vertical, 5) + .frame(maxWidth: .infinity, alignment: .leading) + .background(RoundedRectangle(cornerRadius: 6).fill(Theme.bg)) + .overlay(RoundedRectangle(cornerRadius: 6).stroke(Theme.line)) + FakeButton(title: "Labels") + FakeButton(title: "Milestones") + Button { + Task { await model.triageAll() } + } label: { + Text(triageTitle) + .fontWeight(.semibold) + .padding(.horizontal, 16).padding(.vertical, 5) + .background(RoundedRectangle(cornerRadius: 6).fill(Theme.gradient)) + .opacity(model.phase == .ready ? 1 : 0.6) + } + .buttonStyle(.plain) + .disabled(model.phase != .ready) + FakeButton(title: "New issue", fill: Color(hex: 0x238636)) + } + } + + var triageTitle: String { + switch model.phase { + case .loadingIssues: "Loading issues…" + case .loadingModel: "Loading model…" + case .ready: "✦ Triage with Decision 2.0" + case .running: "✦ Triaging…" + case .finished: "✓ Triaged" + case .failed: "✦ Triage with Decision 2.0" + } + } +} + +struct FakeButton: View { + let title: String + var fill = Theme.btn + + var body: some View { + Text(title) + .fontWeight(.medium) + .padding(.horizontal, 16).padding(.vertical, 5) + .background(RoundedRectangle(cornerRadius: 6).fill(fill)) + .overlay(RoundedRectangle(cornerRadius: 6).stroke(Theme.line)) + } +} + +// MARK: - Sidebar and stats + +struct Sidebar: View { + @ObservedObject var stats: TriageStats + @EnvironmentObject var model: TriageModel + + var body: some View { + VStack(alignment: .leading, spacing: 0) { + Text("ROUTED TO WORKGROUPS") + .font(.system(size: 12, weight: .semibold)).kerning(0.5) + .foregroundStyle(Theme.dim) + .padding(.bottom, 12) + let maxCount = max(1, stats.workgroups.map(\.count).max() ?? 1) + ForEach(stats.workgroups) { wg in + VStack(spacing: 4) { + HStack { + Text(wg.name.replacingOccurrences(of: "wg/", with: "") + .replacingOccurrences(of: "owner/", with: "")) + Spacer() + Text("\(wg.count)").fontWeight(.bold).monospacedDigit() + } + .font(.system(size: 13)) + GeometryReader { geo in + ZStack(alignment: .leading) { + Capsule().fill(Theme.line2) + Capsule().fill(LabelChip.base(wg.name)) + .frame(width: geo.size.width * Double(wg.count) / Double(maxCount)) + } + } + .frame(height: 8) + } + .padding(.bottom, 10) + } + if let note { + Text(note).font(.system(size: 12)).foregroundStyle(Theme.dim).padding(.top, 4) + } + } + .padding(.horizontal, 16).padding(.vertical, 14) + .frame(width: 290, alignment: .leading) + .background(RoundedRectangle(cornerRadius: 6).fill(Theme.panel)) + .overlay(RoundedRectangle(cornerRadius: 6).stroke(Theme.line)) + } + + var note: String? { + switch model.phase { + case .loadingIssues: "Loading issues…" + case .loadingModel: "Loading Decision-2.0-Kai-0.6B…" + case .ready: "Press ✦ Triage to start" + case .running, .finished: nil + case .failed(let message): message + } + } +} + +struct StatsBar: View { + @ObservedObject var stats: TriageStats + let total: Int + + var body: some View { + VStack(spacing: 8) { + HStack(alignment: .firstTextBaseline, spacing: 18) { + stat("\(stats.done)", "/ \(total) issues triaged") + stat("\(stats.decisions)", "labels decided") + stat(stats.avgMs.map { String(format: "%.0f", $0) } ?? "–", "ms per issue") + stat(stats.rate.map { String(format: "%.1f", $0) } ?? "–", "issues / sec") + Spacer(minLength: 0) + (Text("5 decisions per issue in one call · ") + + Text("Decision-2.0-Kai-0.6B").foregroundColor(Theme.ink).bold() + + Text(" · Core ML on this Mac")) + .lineLimit(1) + .minimumScaleFactor(0.8) + .layoutPriority(1) + } + .font(.system(size: 13)) + .foregroundStyle(Theme.dim) + GeometryReader { geo in + ZStack(alignment: .leading) { + Capsule().fill(Theme.line2) + Capsule().fill(Theme.gradient) + .frame(width: geo.size.width * Double(stats.done) / Double(max(1, total))) + } + } + .frame(height: 4) + } + .padding(.horizontal, 14).padding(.vertical, 10) + .background(RoundedRectangle(cornerRadius: 6).fill(Theme.panel)) + .overlay(RoundedRectangle(cornerRadius: 6).stroke(Theme.line)) + } + + func stat(_ value: String, _ caption: String) -> some View { + HStack(alignment: .firstTextBaseline, spacing: 4) { + Text(value).font(.system(size: 18, weight: .bold)).monospacedDigit().foregroundStyle(Theme.ink) + Text(caption) + } + .fixedSize() + } +} + +// MARK: - Issue list + +struct IssueBox: View { + let rows: [RowState] + let open: Int + let scroll: ScrollState + + var body: some View { + VStack(spacing: 0) { + HStack(spacing: 16) { + HStack(spacing: 6) { + Image(systemName: "smallcircle.filled.circle") + Text("\(open) Open") + } + .foregroundStyle(Theme.ink).fontWeight(.semibold) + HStack(spacing: 6) { + Image(systemName: "checkmark") + Text("\(rows.count - open) Closed") + } + Spacer(minLength: 16) + HStack(spacing: 22) { + ForEach(["Author", "Labels", "Projects", "Milestones", "Assignees", "Sort"], id: \.self) { + Text("\($0) ▾") + } + } + .lineLimit(1) + } + .foregroundStyle(Theme.dim) + .padding(.horizontal, 16).padding(.vertical, 14) + .background(Theme.panel) + ScrollViewReader { proxy in + ScrollView { + LazyVStack(spacing: 0) { + ForEach(rows) { row in IssueRow(row: row) } + } + } + .background(ScrollDriver(scroll: scroll, proxy: proxy)) + } + } + .clipShape(RoundedRectangle(cornerRadius: 6)) + .overlay(RoundedRectangle(cornerRadius: 6).stroke(Theme.line)) + .frame(maxHeight: .infinity) + } +} + +/// Keeps the current row centered; the list itself observes nothing that changes during a run. +struct ScrollDriver: View { + @ObservedObject var scroll: ScrollState + let proxy: ScrollViewProxy + + var body: some View { + Color.clear.onChange(of: scroll.target) { _, target in + guard let target else { return } + let anchor = scroll.anchor + withAnimation(.easeInOut(duration: 0.25)) { proxy.scrollTo(target, anchor: anchor) } + guard anchor == .top else { return } + // lazy rows above are estimated until laid out; settle once the jump has landed + Task { @MainActor in + try? await Task.sleep(for: .milliseconds(400)) + proxy.scrollTo(target, anchor: anchor) + } + } + } +} + +struct IssueRow: View { + @ObservedObject var row: RowState + @EnvironmentObject var model: TriageModel + @State private var hover = false + + var body: some View { + let issue = row.issue + HStack(alignment: .top, spacing: 10) { + Image(systemName: issue.isOpen ? "smallcircle.filled.circle" : "checkmark.circle") + .foregroundStyle(issue.isOpen ? Theme.green : Theme.purple) + .padding(.top, 3) + VStack(alignment: .leading, spacing: 2) { + TitleFlow { + Text(issue.title).font(.system(size: 16, weight: .semibold)) + if let result = row.result { + let pop = row.labeledAt.map { Date().timeIntervalSince($0) < 0.5 } ?? false + ForEach(result.labels, id: \.self) { LabelChip(name: $0, pop: pop) } + } + } + Text("#\(issue.number) \(issue.isOpen ? "opened" : "was closed") \(issue.created) by \(issue.author)") + .font(.system(size: 12)).foregroundStyle(Theme.dim) + } + .frame(maxWidth: .infinity, alignment: .leading) + HStack(spacing: 3) { + if issue.comments > 0 { + Image(systemName: "bubble.left") + Text("\(issue.comments)") + } + } + .font(.system(size: 12)).foregroundStyle(Theme.dim) + .frame(width: 50, alignment: .trailing) + .padding(.top, 3) + } + .padding(.horizontal, 16).padding(.vertical, 8) + .background(row.live ? Theme.live : hover ? Theme.panel : Color.clear) + .overlay(alignment: .top) { Rectangle().fill(Theme.line2).frame(height: 1) } + .contentShape(Rectangle()) + .onHover { hover = $0 } + .onTapGesture { model.selected = row } + .id(row.id) + } +} + +/// A label chip: the label color at 18 % fill and 45 % border, text lightened 45 % toward white. +struct LabelChip: View { + let name: String + var pop = false + @State private var shown = false + + static func base(_ name: String) -> Color { + let (r, g, b) = LabelPalette.rgb(name) + return Color(.sRGB, red: r, green: g, blue: b) + } + + var body: some View { + let (r, g, b) = LabelPalette.rgb(name) + let light = { (v: Double) in v + (1 - v) * 0.45 } + Text(name) + .font(.system(size: 12, weight: .medium)) + .lineLimit(1) + .foregroundStyle(Color(.sRGB, red: light(r), green: light(g), blue: light(b))) + .padding(.horizontal, 7) + .frame(height: 20) + .background(Capsule().fill(Color(.sRGB, red: r, green: g, blue: b, opacity: 0.18))) + .overlay(Capsule().stroke(Color(.sRGB, red: r, green: g, blue: b, opacity: 0.45))) + .scaleEffect(!pop || shown ? 1 : 0.2) + .opacity(!pop || shown ? 1 : 0) + .onAppear { + guard pop else { return } + withAnimation(.spring(response: 0.35, dampingFraction: 0.75)) { shown = true } + } + } +} + +/// The title, then chips flowing after it on the same line when they fit, else wrapping below. +struct TitleFlow: Layout { + var spacing: CGFloat = 5 + var lineSpacing: CGFloat = 4 + + func sizeThatFits(proposal: ProposedViewSize, subviews: Subviews, cache: inout ()) -> CGSize { + let frames = arrange(width: proposal.width ?? 10_000, subviews) + let width = frames.map(\.maxX).max() ?? 0 + let height = frames.map(\.maxY).max() ?? 0 + return CGSize(width: proposal.width ?? width, height: height) + } + + func placeSubviews(in bounds: CGRect, proposal: ProposedViewSize, subviews: Subviews, cache: inout ()) { + for (frame, subview) in zip(arrange(width: bounds.width, subviews), subviews) { + subview.place( + at: CGPoint(x: bounds.minX + frame.minX, y: bounds.minY + frame.minY), + proposal: ProposedViewSize(frame.size)) + } + } + + private func arrange(width: CGFloat, _ subviews: Subviews) -> [CGRect] { + guard let first = subviews.first else { return [] } + let title = first.sizeThatFits(ProposedViewSize(width: width, height: nil)) + var frames = [CGRect(origin: .zero, size: title)] + // chips sit beside the title's last line; a wrapped title fills the width, so chips go below it + let firstLine: CGFloat = 21 + var x = title.width + spacing + 2 + var lineTop = max(0, title.height - firstLine) + var lineHeight = min(title.height, firstLine) + for subview in subviews.dropFirst() { + let size = subview.sizeThatFits(.unspecified) + if x + size.width > width, x > 0 { + x = 0 + lineTop += lineHeight + lineSpacing + lineHeight = size.height + } + frames.append(CGRect(x: x, y: lineTop + (lineHeight - size.height) / 2, width: size.width, height: size.height)) + x += size.width + spacing + } + return frames + } +} + +// MARK: - Detail sheet + +struct DetailSheet: View { + @ObservedObject var row: RowState + @Environment(\.dismiss) private var dismiss + + var body: some View { + VStack(alignment: .leading, spacing: 0) { + HStack(alignment: .firstTextBaseline) { + (Text(row.issue.title).font(.system(size: 18, weight: .semibold)) + + Text(" #\(row.issue.number)").font(.system(size: 18)).foregroundColor(Theme.dim)) + Spacer() + Button("Close") { dismiss() }.keyboardShortcut(.cancelAction) + } + if let r = row.result { + HStack(spacing: 4) { ForEach(r.labels, id: \.self) { LabelChip(name: $0) } } + .padding(.vertical, 6) + Text("5 decisions in one call · \(String(format: "%.0f", r.ms)) ms") + .font(.system(size: 12)).foregroundStyle(Theme.dim) + section("Type") + bars(r.type) + section("Owning workgroup") + bars(r.wg) + section("Priority · urgency score \(String(format: "%.2f", r.priorityScore)) / 2") + bars(r.priority) { ["0": "P2 · nice-to-have", "1": "P1 · important", "2": "P0 · critical"][$0] ?? $0 } + section("Flags") + bars([("needs-info", r.needsInfo), ("good first issue", r.goodFirst)]) + } else { + Text("Not triaged yet — press ✦ Triage with Decision 2.0.") + .foregroundStyle(Theme.dim).padding(.top, 12) + } + } + .padding(.horizontal, 20).padding(.vertical, 18) + .frame(width: 640, alignment: .leading) + .background(Theme.panel) + .foregroundStyle(Theme.ink) + .font(.system(size: 13)) + .preferredColorScheme(.dark) + } + + func section(_ title: String) -> some View { + Text(title.uppercased()) + .font(.system(size: 12)).kerning(0.5).foregroundStyle(Theme.dim) + .padding(.top, 12).padding(.bottom, 4) + } + + func bars(_ probs: [(String, Double)], _ format: @escaping (String) -> String = { $0 }) -> some View { + VStack(spacing: 4) { + ForEach(probs.sorted { $0.1 > $1.1 }, id: \.0) { key, p in + HStack(spacing: 8) { + Text(format(key)).frame(width: 230, alignment: .leading).lineLimit(1) + GeometryReader { geo in + ZStack(alignment: .leading) { + Capsule().fill(Theme.line2) + Capsule().fill(Theme.link).frame(width: geo.size.width * p) + } + } + .frame(height: 8) + Text("\(Int((p * 100).rounded()))%") + .monospacedDigit().foregroundStyle(Theme.dim) + .frame(width: 44, alignment: .trailing) + } + } + } + } +} diff --git a/Sources/IssueTriageDemo/Issue.swift b/Sources/IssueTriageDemo/Issue.swift new file mode 100644 index 0000000..a2d2286 --- /dev/null +++ b/Sources/IssueTriageDemo/Issue.swift @@ -0,0 +1,94 @@ +import CryptoKit +import FluidUse +import Foundation + +/// One issue as shown on the page. The real author is replaced by a stable made-up handle; the model state is built +/// once at load. +struct Issue: Identifiable, Sendable { + let number: Int + let title: String + let isOpen: Bool + let author: String + let created: String + let comments: Int + let state: OrderedJSON + + var id: Int { number } + + /// Loads `gh issue list … --json number,title,body,labels,state,author,createdAt,comments` output, newest first. + static func load(from url: URL) throws -> [Issue] { + guard let raw = try JSONSerialization.jsonObject(with: Data(contentsOf: url)) as? [[String: Any]] else { + throw CocoaError(.fileReadCorruptFile) + } + let iso = ISO8601DateFormatter() + let day = DateFormatter() + day.locale = Locale(identifier: "en_US") + day.dateFormat = "MMM d, yyyy" + return raw.compactMap { item -> Issue? in + guard let number = item["number"] as? Int, let title = item["title"] as? String else { return nil } + let login = (item["author"] as? [String: Any])?["login"] as? String ?? "ghost" + let createdAt = item["createdAt"] as? String ?? "" + return Issue( + number: number, title: title, isOpen: (item["state"] as? String) == "OPEN", + author: Issue.handle(login), + created: iso.date(from: createdAt).map { day.string(from: $0) } ?? createdAt, + comments: (item["comments"] as? [Any])?.count ?? 0, + state: .object([ + ("repository", .string("vllm-project/semantic-router")), + ("issue_title", .string(title)), + ("issue_body", .string(Issue.clean(item["body"] as? String ?? ""))), + ])) + } + .sorted { $0.number > $1.number } + } + + /// triage_core.clean: drop HTML comments, collapse 3+ newlines, strip, keep the first 700 code points. + static func clean(_ body: String, limit: Int = 700) -> String { + // `.` must cross newlines (Python re.S), set with the inline flag. + var text = body.replacingOccurrences(of: "(?s)", with: "", options: .regularExpression) + text = text.replacingOccurrences(of: "\n{3,}", with: "\n\n", options: .regularExpression) + text = text.trimmingCharacters(in: .whitespacesAndNewlines) + return String(String.UnicodeScalarView(text.unicodeScalars.prefix(limit))) + } + + static let adjectives = [ + "swift", "quiet", "brave", "lucky", "cosmic", "rusty", "sunny", "clever", "mellow", "pixel", "fuzzy", "neon", + ] + static let nouns = [ + "otter", "falcon", "maple", "comet", "badger", "lynx", "cactus", "harbor", "pebble", "willow", "koala", "ember", + ] + + /// server.handle: the SHA-256 of the login as a 256-bit integer h → "{ADJ[h % 12]}-{NOUN[(h // 12) % 12]}{h % 97}". + static func handle(_ login: String) -> String { + let digest = Array(SHA256.hash(data: Data(login.utf8))) + func mod(_ m: Int) -> Int { digest.reduce(0) { ($0 * 256 + Int($1)) % m } } + let h144 = mod(144) + return "\(adjectives[h144 % 12])-\(nouns[h144 / 12])\(mod(97))" + } +} + +/// The repository's label colors (labels.json), for the labels the model can give. +enum LabelPalette { + static let colors: [String: String] = [ + "bug": "d73a4a", "enhancement": "a2eeef", "documentation": "ededed", "question": "d876e3", + "proposal": "5319E7", + "wg/mom-routing": "7B3FE4", "wg/router-models-inference-runtime": "0E8A16", + "wg/data-plane-networking": "0366D6", "wg/enterprise-environment": "D93F0B", + "wg/evaluation-quality": "e247cd", "wg/developer-experience-ecosystem": "1D76DB", + "wg/agentic-context": "8250DF", "owner/maintainers": "5319E7", "wg/platform-operations": "D93F0B", + "priority/P0": "b60205", "priority/P1": "5f9df6", "priority/P2": "e4f815", + "needs-info": "D876E3", "good first issue": "7057ff", + ] + + /// Sidebar workgroups, in labels.json order (ties keep this order). + static let workgroups = [ + "wg/mom-routing", "wg/router-models-inference-runtime", "wg/data-plane-networking", + "wg/enterprise-environment", "wg/evaluation-quality", "wg/developer-experience-ecosystem", + "wg/agentic-context", "owner/maintainers", "wg/platform-operations", + ] + + static func rgb(_ name: String) -> (Double, Double, Double) { + let hex = UInt32(colors[name] ?? "8b949e", radix: 16) ?? 0x8b949e + return (Double((hex >> 16) & 0xff) / 255, Double((hex >> 8) & 0xff) / 255, Double(hex & 0xff) / 255) + } +} diff --git a/Sources/IssueTriageDemo/IssueTriageDemoApp.swift b/Sources/IssueTriageDemo/IssueTriageDemoApp.swift new file mode 100644 index 0000000..9ff2fdb --- /dev/null +++ b/Sources/IssueTriageDemo/IssueTriageDemoApp.swift @@ -0,0 +1,32 @@ +import AppKit +import SwiftUI + +@main +struct IssueTriageDemoApp: App { + @StateObject private var model = TriageModel() + + init() { + setvbuf(stdout, nil, _IOLBF, 0) + // Bare SwiftPM executables start as background processes; make this one a regular windowed app. + for key in UserDefaults.standard.dictionaryRepresentation().keys where key.hasPrefix("NSWindow Frame") { + UserDefaults.standard.removeObject(forKey: key) + } + // The issues path is an argument; AppKit would otherwise take it as a document to open and show no window. + var arguments = UserDefaults.standard.volatileDomain(forName: UserDefaults.argumentDomain) + arguments["NSTreatUnknownArgumentsAsOpen"] = "NO" + UserDefaults.standard.setVolatileDomain(arguments, forName: UserDefaults.argumentDomain) + NSApplication.shared.setActivationPolicy(.regular) + NSApplication.shared.activate(ignoringOtherApps: true) + } + + var body: some Scene { + WindowGroup("Issue Triage — Decision-2.0-Kai-0.6B on Core ML") { + ContentView() + .environmentObject(model) + .frame(minWidth: 1000, minHeight: 640) + .task { await model.start() } + } + .defaultSize(width: 1500, height: 950) + .windowResizability(.contentMinSize) + } +} diff --git a/Sources/IssueTriageDemo/README.md b/Sources/IssueTriageDemo/README.md new file mode 100644 index 0000000..9f12139 --- /dev/null +++ b/Sources/IssueTriageDemo/README.md @@ -0,0 +1,33 @@ +# IssueTriageDemo + +A GitHub-style dark issues page for a fictional repo, **routerlabs / semantic-switch**, listing 1,000 real open-source +issues (vllm-project/semantic-router) with their labels stripped and their authors replaced by made-up handles. Press +**✦ Triage with Decision 2.0** and Decision-2.0-Kai-0.6B on Core ML labels every issue live: type, owning workgroup, +priority, needs-info and good-first-issue, **5 decisions per issue in one model call**. + +- Label chips pop in as each issue is labeled; the list auto-scrolls to keep the current row centered. +- The sidebar routes issues to workgroups, re-sorted live by count; the stats bar shows issues triaged, labels decided, + ms per issue and issues / sec. +- Click an issue to see the model's probabilities per question (type, workgroup, priority with its urgency score, + flags). +- Priority cuts the Score answer's expected level (0-2): P0 at >= 1.452, P1 at >= 1.251, else P2 (this repo's p90 / + p50). needs-info and good first issue need yes >= 0.6. + +The model runs back to back in a detached task; the UI only consumes results. Each triaged issue prints one line to +stdout (`66.1 ms #4612 enhancement wg/router-models-inference-runtime priority/P2`), plus a summary at the end. + +## Run + +```sh +Sources/IssueTriageDemo/fetch-issues.sh # gh CLI → ~/Library/Caches/FluidUse/issue-triage/issues.json +swift run -c release IssueTriageDemo [issues.json] [--auto] +``` + +The issues file holds real author logins (shown only as hashed handles), so it stays out of the repository. `--auto` +starts triage 1.5 s after the model is ready, for hands-free recording. The model snapshot downloads on first run +(`Decision2ModelStore.ensure(.kai)`, 1.1 GB). + +## Measured + +M-series Mac, 1,000 issues: 66.8 ms per issue, 15.0 issues / sec, 5,000 decisions in 67 s, with the UI live. Labels +match the Python Core ML reference demo on every issue compared (626 of 626). diff --git a/Sources/IssueTriageDemo/TriageModel.swift b/Sources/IssueTriageDemo/TriageModel.swift new file mode 100644 index 0000000..c96e8da --- /dev/null +++ b/Sources/IssueTriageDemo/TriageModel.swift @@ -0,0 +1,271 @@ +import FluidUse +import Foundation +import SwiftUI + +/// The five questions (triage_core.questions), answered together in one model call. +enum TriageQuestions { + static let all: [(id: String, question: Decision2Question)] = [ + ( + "type", + .choice( + "What kind of issue is this?", + [ + "bug": "Something isn't working: errors, crashes, wrong behavior, failing tests", + "enhancement": "New feature or improvement request", + "documentation": "Docs, README, guides, examples or website content", + "question": "A usage question or request for help, not a defect", + "proposal": "Architecture, design, roadmap or research proposal", + ]) + ), + ( + "wg", + .choice( + "Which workgroup should own this issue?", + [ + "wg/mom-routing": + "Routing decisions, signals, model selection rules, Mixture-of-Models routing logic", + "wg/router-models-inference-runtime": + "Classifier/embedding models, model training, inference runtime (Candle, ONNX, Rust bindings)", + "wg/data-plane-networking": + "Envoy, ext_proc, gateway, proxy, request/response handling, networking", + "wg/enterprise-environment": + "Kubernetes, Helm, operators, auth, deployment environments, ROCm/GPU platforms", + "wg/evaluation-quality": + "Benchmarks, evaluations, accuracy, testing quality, CI performance checks", + "wg/developer-experience-ecosystem": + "Docs, CLI, dashboard UI, examples, install, local dev experience", + "wg/agentic-context": "Agents, tools, MCP, memory, context management, agent workflows", + "wg/platform-operations": + "Observability, metrics, logging, tracing, operations and reliability", + "owner/maintainers": "Repository governance, community, releases, process", + ]) + ), + ( + "priority", + .score( + "How urgent is this issue?", + [ + "P2: nice-to-have or exploratory", "P1: important, should be done", + "P0: critical, must be fixed now", + ]) + ), + ( + "needs_info", + .yesNo( + "Is important information missing (e.g. reproduction steps, versions, logs) so the maintainers must ask the author for more details?" + ) + ), + ("good_first", .yesNo("Is this a small, well-scoped task suitable for a first-time contributor?")), + ] +} + +/// One issue's labels and the probabilities behind them (server.triage). +struct TriageResult: Sendable { + // The priority score (expected level, 0-2) ranks issues well but its argmax is P0 for most issues; cut at this + // repo's p90 / p50 so P0 is the top ~10 % and P1 the next ~40 %. + static let p0At = 1.452 + static let p1At = 1.251 + + let ms: Double + let labels: [String] + let type: [(String, Double)] + let wg: [(String, Double)] + let priority: [(String, Double)] + let priorityScore: Double + let needsInfo: Double + let goodFirst: Double + + init(_ result: Decision2Result, ms: Double) { + func probs(_ id: String) -> [(String, Double)] { + guard let a = result[id] else { return [] } + return Array(zip(a.keys, a.probabilities)) + } + let level = result["priority"]?.score ?? 0 + self.ms = ms + type = probs("type") + wg = probs("wg") + priority = probs("priority") + priorityScore = level + needsInfo = result["needs_info"]?.yes ?? 0 + goodFirst = result["good_first"]?.yes ?? 0 + var labels = [ + result["type"]?.choice ?? "?", result["wg"]?.choice ?? "?", + level >= Self.p0At ? "priority/P0" : level >= Self.p1At ? "priority/P1" : "priority/P2", + ] + if needsInfo >= 0.6 { labels.append("needs-info") } + if goodFirst >= 0.6 { labels.append("good first issue") } + self.labels = labels + } +} + +/// Per-row state, so a result redraws only its own row. +@MainActor final class RowState: ObservableObject, Identifiable { + let issue: Issue + @Published var result: TriageResult? + @Published var live = false + var labeledAt: Date? + + nonisolated var id: Int { issue.number } + + init(issue: Issue) { self.issue = issue } +} + +struct WorkgroupCount: Identifiable, Equatable { + let name: String + var count: Int + var id: String { name } +} + +/// Counters for the stats bar and the sidebar. +@MainActor final class TriageStats: ObservableObject { + @Published var done = 0 + @Published var decisions = 0 + @Published var avgMs: Double? + @Published var rate: Double? + @Published var workgroups = LabelPalette.workgroups.map { WorkgroupCount(name: $0, count: 0) } +} + +/// The row the list should keep centered. +@MainActor final class ScrollState: ObservableObject { + @Published var target: Int? + var anchor = UnitPoint.center +} + +@MainActor final class TriageModel: ObservableObject { + enum Phase: Equatable { + case loadingIssues, loadingModel, ready, running, finished + case failed(String) + } + + typealias Answer = @Sendable (OrderedJSON) async throws -> Decision2Result + + @Published var phase = Phase.loadingIssues { + didSet { if case .failed(let message) = phase { print("error: \(message)") } } + } + @Published var rows: [RowState] = [] + @Published var selected: RowState? + let stats = TriageStats() + let scroll = ScrollState() + private var answer: Answer? + private var started = false + + var openCount: Int { rows.filter(\.issue.isOpen).count } + + static func issuesURL() -> URL { + if let path = CommandLine.arguments.dropFirst().first(where: { !$0.hasPrefix("-") }) { + return URL(fileURLWithPath: (path as NSString).expandingTildeInPath) + } + return FileManager.default.homeDirectoryForCurrentUser + .appendingPathComponent("Library/Caches/FluidUse/issue-triage/issues.json") + } + + func start() async { + guard !started else { return } + started = true + let url = Self.issuesURL() + guard FileManager.default.fileExists(atPath: url.path) else { + phase = .failed("No issues at \(url.path).\nRun Sources/IssueTriageDemo/fetch-issues.sh first.") + return + } + do { + let issues = try await Task.detached(priority: .userInitiated) { try Issue.load(from: url) }.value + rows = issues.map(RowState.init) + print("loaded \(issues.count) issues from \(url.path)") + phase = .loadingModel + guard #available(macOS 15.0, *) else { + phase = .failed("Decision 2.0 on Core ML needs macOS 15 or later.") + return + } + let loadStart = Date() + answer = try await Task.detached(priority: .userInitiated) { () -> Answer in + let directory = try await Decision2ModelStore.ensure(.kai) + let manager = try await Decision2Manager.load(from: directory) + try await manager.warm() + return { state in try await manager.answer(state: state, questions: TriageQuestions.all) } + }.value + print(String(format: "Decision-2.0-Kai-0.6B loaded and warmed in %.1f s", Date().timeIntervalSince(loadStart))) + phase = .ready + if CommandLine.arguments.contains("--auto") { + try await Task.sleep(for: .milliseconds(1500)) + await triageAll() + } + } catch { + phase = .failed(error.localizedDescription) + } + } + + func triageAll() async { + guard phase == .ready, let answer else { return } + phase = .running + let states = rows.map(\.issue.state) + let numbers = rows.map(\.issue.number) + // Inference runs back to back off the main actor; the UI only consumes results, so rendering never delays + // the next call. + let (results, continuation) = AsyncThrowingStream.makeStream(of: (Int, TriageResult).self) + let producer = Task.detached(priority: .userInitiated) { + do { + for (index, state) in states.enumerated() { + try Task.checkCancellation() + let start = DispatchTime.now().uptimeNanoseconds + let result = try await answer(state) + let ms = Double(DispatchTime.now().uptimeNanoseconds - start) / 1e6 + let triage = TriageResult(result, ms: ms) + print( + String(format: "%5.1f ms #%-5d ", ms, numbers[index]) + + triage.labels.joined(separator: " ")) + continuation.yield((index, triage)) + } + continuation.finish() + } catch { + continuation.finish(throwing: error) + } + } + defer { producer.cancel() } + + rows.first?.live = true + scroll.target = rows.first?.id + let t0 = Date() + var sumMs = 0.0 + var counts = Dictionary(uniqueKeysWithValues: LabelPalette.workgroups.map { ($0, 0) }) + do { + for try await (index, result) in results { + let row = rows[index] + row.labeledAt = Date() + row.result = result + withAnimation(.easeOut(duration: 0.6)) { row.live = false } + if index + 1 < rows.count { + rows[index + 1].live = true + if index % 2 == 1 { scroll.target = rows[index + 1].id } + } + counts[result.labels[1], default: 0] += 1 + sumMs += result.ms + let done = index + 1 + stats.done = done + stats.decisions = done * TriageQuestions.all.count + stats.avgMs = sumMs / Double(done) + stats.rate = Double(done) / Date().timeIntervalSince(t0) + if done % 3 == 0 || done == rows.count { updateWorkgroups(counts) } + } + } catch { + phase = .failed(error.localizedDescription) + return + } + let seconds = Date().timeIntervalSince(t0) + print( + String( + format: "triaged %d issues · %d decisions · %.1f ms per issue · %.1f issues/s · %.1f s total", + stats.done, stats.decisions, sumMs / Double(max(1, stats.done)), Double(stats.done) / seconds, seconds)) + phase = .finished + scroll.anchor = .top + scroll.target = rows.first?.id + } + + private func updateWorkgroups(_ counts: [String: Int]) { + let order = Dictionary(uniqueKeysWithValues: LabelPalette.workgroups.enumerated().map { ($1, $0) }) + let sorted = counts.map { WorkgroupCount(name: $0.key, count: $0.value) } + .sorted { ($0.count, -(order[$0.name] ?? 99)) > ($1.count, -(order[$1.name] ?? 99)) } + if sorted != stats.workgroups { + withAnimation(.easeInOut(duration: 0.25)) { stats.workgroups = sorted } + } + } +} diff --git a/Sources/IssueTriageDemo/fetch-issues.sh b/Sources/IssueTriageDemo/fetch-issues.sh new file mode 100755 index 0000000..fd4ef82 --- /dev/null +++ b/Sources/IssueTriageDemo/fetch-issues.sh @@ -0,0 +1,13 @@ +#!/bin/sh +# Fetch the issues the IssueTriageDemo labels (needs the GitHub CLI, `gh auth login`). +# +# Sources/IssueTriageDemo/fetch-issues.sh [output.json] +# +# Default output: ~/Library/Caches/FluidUse/issue-triage/issues.json (where the app looks when given no path). +# The file holds real issue authors; the app shows made-up handles in their place. Keep it out of the repository. +set -eu +out="${1:-$HOME/Library/Caches/FluidUse/issue-triage/issues.json}" +mkdir -p "$(dirname "$out")" +gh issue list -R vllm-project/semantic-router --state all --limit 1000 \ + --json number,title,body,labels,state,author,createdAt,comments > "$out" +echo "wrote $out"