diff --git a/Documentation/DecisionModelSupport.md b/Documentation/DecisionModelSupport.md index 889c47f..cc8473f 100644 --- a/Documentation/DecisionModelSupport.md +++ b/Documentation/DecisionModelSupport.md @@ -7,6 +7,7 @@ FluidUse serves the weighted sub-1B models on the [Jev Decision Index](https://h | Laya | `LayaManager` | [laya-coreml](https://huggingface.co/FluidInference/laya-coreml) | Native Swift | | GLiNER 2.5 small / base / multilingual | `GLiNER2Manager` (`.small`, `.base`, `.multilingual`) | [small](https://huggingface.co/FluidInference/gliner2-5-small-coreml), [base](https://huggingface.co/FluidInference/gliner2-5-base-coreml), [multi](https://huggingface.co/FluidInference/gliner2-5-multi-coreml) | Native Swift | | Verdict | `VerdictManager` | [verdict-coreml](https://huggingface.co/FluidInference/verdict-coreml) | Native Swift, calibrated, with trained abstention | +| Cua-S1-4B-0.2 (text / multimodal) | `CuaS1FourBManager` | [cua-s1-4b-coreml](https://huggingface.co/FluidInference/cua-s1-4b-coreml) | Native Swift, GPU, fp16 / w8 / gptq | | GLiClass Edge Apps v2 | `GLiClassManager` | [gliclass-edge-apps-coreml](https://huggingface.co/FluidInference/gliclass-edge-apps-coreml) | Native Swift | | Kev 0.5B / 0.6B | `PublishedCoreMLManager` + `evaluate(SystemOneRequest)` | [0.5B](https://huggingface.co/FluidInference/kev-0-5b-coreml), [0.6B](https://huggingface.co/FluidInference/kev-0.6b-coreml) | Bridge | | Decision 1.0 Kai / Lex | `PublishedCoreMLManager` + `evaluate(SystemOneRequest)` | [Kai](https://huggingface.co/FluidInference/decision-1.0-kai-coreml), [Lex](https://huggingface.co/FluidInference/decision-1.0-lex-coreml) | Bridge | diff --git a/Package.swift b/Package.swift index 7ab7069..01a08ac 100644 --- a/Package.swift +++ b/Package.swift @@ -30,6 +30,7 @@ let package = Package( dependencies: ["FluidUse", "Game2048", "LayaTetris", .product(name: "FluidAudio", package: "FluidAudio")] ), .executableTarget(name: "FluidUseOfficialBench", dependencies: ["FluidUse"]), + .executableTarget(name: "FluidUseCuaS1", dependencies: ["FluidUse"]), .executableTarget( name: "LayaTetrisDemo", dependencies: ["FluidUse", "LayaTetris"], diff --git a/README.md b/README.md index 2f6a9c6..f6d8fcf 100644 --- a/README.md +++ b/README.md @@ -107,6 +107,30 @@ The current ten-seed capped run averages 3,667 pieces for GLiClass LUT8 and 2,87 heuristic control. See [Benchmarks.md](Benchmarks.md) for the exact policy, per-seed results, and the limits of comparison with the older laya measurements. +## Cua-S1-4B GUI decisions + +`CuaS1FourBManager` runs [Cua-S1-4B-0.2](https://huggingface.co/cua-ai/cua-s1-4b-0.2) (Qwen3.5-4B + Cua's +LoRA adapters) on the GPU: one prefill pass scores a closed list of `(element, action)` options for an +accessibility tree (`.text`) or a screenshot (`.multimodal`). Models download pinned and SHA-256 checked from +[FluidInference/cua-s1-4b-coreml](https://huggingface.co/FluidInference/cua-s1-4b-coreml) on first use. + +```swift +let cua = try await CuaS1FourBManager.load(configuration: .init(modality: .text, variant: "gptq")) +let decision = try await cua.decide(CuaS1FourBState( + app: "portal", taskFamily: "login_auth", goal: "Log in", + accessibilityTree: "- [el_0] Button \"Log in\"", + options: [.init(elementId: "el_0", role: "Button", label: "Log in", action: "click"), + .init(elementId: "el_0", role: "Button", label: "Log in", action: "skip")])) +print(decision.bestPerElement()) +``` + +On a 613-task GUI-360 text split the fp16 and `gptq` (2.6 GB) builds both score 85.5% at about 1.1 s per +decision on an M5 Pro; the Swift runtime matches the Python Core ML path exactly (38/38 fixture parity for +both modalities). `swift run -c release FluidUseCuaS1 parity --models --fixtures ` +reruns the parity check against a local mobius build. Call `prewarm()` after `load` in an app: the first prediction of a fresh +4B Core ML graph spends about 100 s specializing GPU kernels (cached by the OS afterwards). The text decoder is 2.6-6.8 GB and the multimodal +one 4.0-7.4 GB, so this is a Mac-class model. + ## GLiNER 2.5 classification `GLiNER2Manager` runs the published base or multilingual classification head on device. Both diff --git a/Sources/FluidUse/CuaS1FourB/CuaS1FourBManager.swift b/Sources/FluidUse/CuaS1FourB/CuaS1FourBManager.swift new file mode 100644 index 0000000..2d476db --- /dev/null +++ b/Sources/FluidUse/CuaS1FourB/CuaS1FourBManager.swift @@ -0,0 +1,387 @@ +import Accelerate +@preconcurrency import CoreML +import Foundation + +/// On-device Cua-S1-4B-0.2 (Qwen3.5-4B + Cua LoRA) closed-option GUI decisions. +/// +/// One prefill pass per decision: the prompt is right-padded to a fixed bucket, run through four +/// Core ML decoder parts (8 Qwen3.5 layers each), and the answer-letter logits A..Z at the last +/// prompt position are read out. Nothing is generated, so there is no KV cache. The embedding +/// gather happens on the host (memory-mapped fp16 table) so screenshot features can be spliced in. +/// +/// A manager runs one modality: `text` (accessibility tree) or `multimodal` (screenshot); the two +/// adapters were trained separately and ship as separate decoders. Calls are serialized by the actor. +public actor CuaS1FourBManager { + public struct Configuration: Sendable { + public var modality: CuaS1FourBModality + /// Bucket lengths to load (each loads its own ~6.8 GB fp16 of decoder weights). + public var lengths: [Int] + /// Weight variant: `""` fp16 (6.8 GB), `"w8"` int8 linears (3.4 GB), `"gptq"` GPTQ MLP int4 + int8 + /// (2.6 GB, text only; same GUI-360 accuracy as fp16). + public var variant: String + /// The 4B prefill runs best on the GPU; the ANE path falls back to CPU for most of the graph. + public var computeUnits: MLComputeUnits + + public init( + modality: CuaS1FourBModality = .text, lengths: [Int]? = nil, variant: String = "", + computeUnits: MLComputeUnits = .cpuAndGPU + ) { + self.modality = modality + self.lengths = lengths ?? (modality == .text ? [1024] : [2048]) + self.variant = variant + self.computeUnits = computeUnits + } + } + + struct Bucket { + let length: Int + let parts: [MLModel] + let letterTokenIds: [Int] + } + + struct RopeParameters { + let rotaryDim: Int + let theta: Double + let mropeSection: [Int] + } + + nonisolated public let modality: CuaS1FourBModality + nonisolated public let tokenizer: QwenTokenizer + let buckets: [Bucket] + let embeddings: Data + let hiddenSize: Int + let rope: RopeParameters + let vision: CuaS1FourBVision? + + init( + modality: CuaS1FourBModality, tokenizer: QwenTokenizer, buckets: [Bucket], embeddings: Data, + hiddenSize: Int, rope: RopeParameters, vision: CuaS1FourBVision? + ) { + self.modality = modality + self.tokenizer = tokenizer + self.buckets = buckets.sorted { $0.length < $1.length } + self.embeddings = embeddings + self.hiddenSize = hiddenSize + self.rope = rope + self.vision = vision + } + + /// Loaded bucket lengths, ascending. + public var lengths: [Int] { buckets.map(\.length) } + + /// Load from a local directory laid out like the published repository: + /// `tokenizer.json`, `embeddings.f16`, `/L[-variant]/CuaS1Decoder_part{0..3}.mlmodelc|.mlpackage` + /// (+ `config.json`), and for multimodal `multimodal/vision/`. + public static func load( + from directory: URL, configuration: Configuration = Configuration() + ) async throws + -> CuaS1FourBManager + { + let manager = FileManager.default + let tokenizer = try QwenTokenizer(tokenizerJsonURL: directory.appendingPathComponent("tokenizer.json")) + let embeddingsURL = try firstExisting( + [ + directory.appendingPathComponent("embeddings.f16"), + directory.appendingPathComponent(configuration.modality.rawValue).appendingPathComponent( + "embeddings.f16"), + ], what: "embeddings.f16") + let embeddings = try Data(contentsOf: embeddingsURL, options: .alwaysMapped) + + let modelConfiguration = MLModelConfiguration() + modelConfiguration.computeUnits = configuration.computeUnits + var buckets: [Bucket] = [] + var hiddenSize = 0 + var rope: RopeParameters? + for length in configuration.lengths { + let name = "L\(length)" + (configuration.variant.isEmpty ? "" : "-\(configuration.variant)") + let bucketDir = directory.appendingPathComponent(configuration.modality.rawValue) + .appendingPathComponent(name) + let configData = try Data(contentsOf: bucketDir.appendingPathComponent("config.json")) + guard let config = try JSONSerialization.jsonObject(with: configData) as? [String: Any], + let seqLen = config["seq_len"] as? Int, seqLen == length, + let hidden = config["hidden_size"] as? Int, + let partsInfo = config["parts"] as? [Any], + let letterIds = config["letter_token_ids"] as? [Int], + let rotaryDim = config["rotary_dim"] as? Int, + let theta = config["rope_theta"] as? Double, + let section = config["mrope_section"] as? [Int] + else { + throw CuaS1FourBError.invalidAsset("bad config.json in \(bucketDir.path)") + } + guard letterIds.count == CuaS1FourBPrompt.letters.count else { + throw CuaS1FourBError.invalidAsset("config.json letter_token_ids must list A..Z") + } + for (letter, id) in zip(CuaS1FourBPrompt.letters, letterIds) where tokenizer.encode(letter) != [id] { + throw CuaS1FourBError.invalidAsset("letter \(letter) is not the single token \(id) in tokenizer.json") + } + hiddenSize = hidden + rope = RopeParameters(rotaryDim: rotaryDim, theta: theta, mropeSection: section) + var parts: [MLModel] = [] + for index in 0.. CuaS1FourBDecision { + try Task.checkCancellation() + let chat = try CuaS1FourBPrompt.chat(state: state, modality: modality) + var ids = tokenizer.encode(chat) + var image: CuaS1FourBVision.ImageFeatures? + if modality == .multimodal { + guard let vision, let screenshot = state.screenshot else { + throw CuaS1FourBError.invalidInput("multimodal modality requires a screenshot") + } + let features = try vision.features(for: screenshot) + ids = try vision.expandImagePads(ids, count: features.tokens) + image = features + } + let logits = try letterLogits(ids: ids, image: image) + let count = state.options.count + let used = Array(logits.prefix(count)) + let maxLogit = used.max() ?? 0 + let exps = used.map { expf($0 - maxLogit) } + let total = exps.reduce(0, +) + let scored = (0..assistant\n") + for bucket in buckets { + _ = try letterLogits(ids: ids, image: nil, bucket: bucket) + } + try vision?.prewarm() + } + + /// Letter logits A..Z for already tokenized ids (parity checks against the Python reference). + /// `image` carries spliced screenshot features and their M-RoPE grid for multimodal prompts. + func letterLogits(ids: [Int], image: CuaS1FourBVision.ImageFeatures?) throws -> [Float] { + try letterLogits(ids: ids, image: image, bucket: bucket(for: ids.count)) + } + + private func letterLogits( + ids: [Int], image: CuaS1FourBVision.ImageFeatures?, bucket: Bucket + ) throws + -> [Float] + { + let length = bucket.length + let hidden = try MLMultiArray( + shape: [1, NSNumber(value: length), NSNumber(value: hiddenSize)], dataType: .float16) + let rowBytes = hiddenSize * 2 + let vocabRows = embeddings.count / rowBytes + let hiddenPtr = hidden.dataPointer.bindMemory(to: UInt16.self, capacity: length * hiddenSize) + hiddenPtr.initialize(repeating: 0, count: length * hiddenSize) + var imageRow = 0 + try embeddings.withUnsafeBytes { raw in + for (t, id) in ids.enumerated() { + let dst = UnsafeMutableRawPointer(hiddenPtr + t * hiddenSize) + if let image, id == image.padTokenId { + image.copyRow(imageRow, to: dst) + imageRow += 1 + continue + } + guard id >= 0, id < vocabRows else { throw CuaS1FourBError.invalidInput("token id \(id) out of range") } + guard let base = raw.baseAddress else { throw CuaS1FourBError.invalidAsset("empty embeddings") } + dst.copyMemory(from: base + id * rowBytes, byteCount: rowBytes) + } + } + if let image, imageRow != image.tokens { + throw CuaS1FourBError.invalidInput("prompt has \(imageRow) image pads for \(image.tokens) features") + } + + let positions = + image.map { + Self.mropePositions( + ids: ids, padTokenId: $0.padTokenId, imageTokens: $0.tokens, gridRows: $0.gridRows, + gridCols: $0.gridCols) + } ?? (0.. Bucket { + guard let bucket = buckets.first(where: { tokens <= $0.length }) else { + throw CuaS1FourBError.promptTooLong(tokens: tokens, maximum: buckets.last?.length ?? 0) + } + return bucket + } + + /// Interleaved M-RoPE cos/sin [L, rotaryDim] (fp16) for (t, h, w) positions; padding rows use position 0. + func ropeTables(positions: [[Int]], length: Int) throws -> (MLMultiArray, MLMultiArray) { + let dim = rope.rotaryDim + let half = dim / 2 + var cosValues = [Float](repeating: 1, count: length * dim) + var sinValues = [Float](repeating: 0, count: length * dim) + let invFreq = (0.. [[Int]] + { + var positions: [[Int]] = [] + positions.reserveCapacity(ids.count) + var next = 0 + var t = 0 + while t < ids.count { + if ids[t] == padTokenId { + let start = next + for row in 0.. MLMultiArray { + let array = try MLMultiArray(shape: shape.map { NSNumber(value: $0) }, dataType: .float16) + var source = values + source.withUnsafeMutableBytes { src in + var input = vImage_Buffer( + data: src.baseAddress, height: 1, width: vImagePixelCount(values.count), rowBytes: values.count * 4) + var output = vImage_Buffer( + data: array.dataPointer, height: 1, width: vImagePixelCount(values.count), rowBytes: values.count * 2) + vImageConvert_PlanarFtoPlanar16F(&input, &output, 0) + } + return array + } + + /// Fresh densely packed fp16 copy. GPU outputs can carry padded strides and their backing buffers; + /// feeding them straight into the next model trips an MPSGraph shape/stride assertion. + static func contiguousCopy(_ array: MLMultiArray) throws -> MLMultiArray { + let shape = array.shape.map(\.intValue) + let copy = try MLMultiArray(shape: array.shape, dataType: .float16) + let rowLength = shape.last ?? 1 + let rows = array.count / rowLength + let strides = array.strides.map(\.intValue) + let dst = copy.dataPointer.bindMemory(to: UInt16.self, capacity: array.count) + let src = array.dataPointer.bindMemory(to: UInt16.self, capacity: array.count) + guard array.dataType == .float16, strides.last == 1 else { + throw CuaS1FourBError.invalidModel("unexpected decoder output layout \(array.dataType.rawValue) \(strides)") + } + var outer = [Int](repeating: 0, count: max(shape.count - 1, 0)) + for row in 0.. [Float] { + let count = array.count + if array.dataType == .float32 { + let ptr = array.dataPointer.bindMemory(to: Float.self, capacity: count) + return Array(UnsafeBufferPointer(start: ptr, count: count)) + } + var result = [Float](repeating: 0, count: count) + result.withUnsafeMutableBytes { dst in + var input = vImage_Buffer( + data: array.dataPointer, height: 1, width: vImagePixelCount(count), rowBytes: count * 2) + var output = vImage_Buffer( + data: dst.baseAddress, height: 1, width: vImagePixelCount(count), rowBytes: count * 4) + vImageConvert_Planar16FtoPlanarF(&input, &output, 0) + } + return result + } + + static func compiledModel(_ base: URL, manager: FileManager) async throws -> URL { + let compiled = base.appendingPathExtension("mlmodelc") + if manager.fileExists(atPath: compiled.path) { return compiled } + let package = base.appendingPathExtension("mlpackage") + guard manager.fileExists(atPath: package.path) else { + throw CuaS1FourBError.invalidAsset("missing \(base.lastPathComponent).mlmodelc or .mlpackage") + } + return try await MLModel.compileModel(at: package) + } + + static func firstExisting(_ urls: [URL], what: String) throws -> URL { + guard let url = urls.first(where: { FileManager.default.fileExists(atPath: $0.path) }) else { + throw CuaS1FourBError.invalidAsset("missing \(what) (looked in \(urls.map(\.path)))") + } + return url + } +} diff --git a/Sources/FluidUse/CuaS1FourB/CuaS1FourBModelStore.swift b/Sources/FluidUse/CuaS1FourB/CuaS1FourBModelStore.swift new file mode 100644 index 0000000..c0c1842 --- /dev/null +++ b/Sources/FluidUse/CuaS1FourB/CuaS1FourBModelStore.swift @@ -0,0 +1,92 @@ +import Foundation + +/// Pinned, checksum-verified download of `FluidInference/cua-s1-4b-coreml`. +/// +/// `Resources/cua-s1-4b-manifest.json` records every file's size and SHA-256 at one Hub revision +/// (regenerate with `Tools/pin_cua_s1_4b.py`). Only the files one configuration needs are fetched: +/// the shared tokenizer and embedding table, the requested decoder bucket(s), and for multimodal the +/// vision tower. Files land in `~/Library/Application Support/FluidUse/Models/cua-s1-4b-coreml`. +public enum CuaS1FourBModelStore { + public static let repository = "FluidInference/cua-s1-4b-coreml" + public typealias Progress = @Sendable (_ file: String, _ bytes: Int64) -> Void + + struct Manifest: Decodable { + let repository: String + let revision: String + let files: [PublishedCoreMLModelStore.Manifest.File] + } + + static func manifest() throws -> Manifest { + guard + let url = Bundle.module.url( + forResource: "cua-s1-4b-manifest", withExtension: "json", subdirectory: "Resources") + else { throw CuaS1FourBError.invalidAsset("cua-s1-4b-manifest.json is not bundled") } + let manifest = try JSONDecoder().decode(Manifest.self, from: Data(contentsOf: url)) + guard manifest.repository == repository else { + throw CuaS1FourBError.invalidAsset("manifest pins \(manifest.repository), expected \(repository)") + } + return manifest + } + + /// Path prefixes one configuration needs. + static func requiredPrefixes(_ configuration: CuaS1FourBManager.Configuration) -> [String] { + var prefixes = ["tokenizer.json", "embeddings.f16", "LICENSE", "NOTICE"] + let suffix = configuration.variant.isEmpty ? "" : "-\(configuration.variant)" + for length in configuration.lengths { + prefixes.append("\(configuration.modality.rawValue)/L\(length)\(suffix)/") + } + if configuration.modality == .multimodal { prefixes.append("multimodal/vision/") } + return prefixes + } + + /// Ensure the files for `configuration` exist under `cacheDirectory/cua-s1-4b-coreml`, downloading + /// missing or mismatched ones. Returns the repository directory. + public static func ensure( + configuration: CuaS1FourBManager.Configuration, cacheDirectory: URL? = nil, progress: Progress? = nil + ) async throws -> URL { + let manifest = try manifest() + let prefixes = requiredPrefixes(configuration) + let files = manifest.files.filter { file in prefixes.contains { file.path.hasPrefix($0) } } + for prefix in prefixes where prefix.hasSuffix("/") && !files.contains(where: { $0.path.hasPrefix(prefix) }) { + throw CuaS1FourBError.invalidAsset("\(prefix) is not in the pinned \(repository) revision") + } + let root = cacheDirectory ?? LayaModelStore.defaultCacheDirectory() + let directory = root.appendingPathComponent("cua-s1-4b-coreml", isDirectory: true) + let manager = FileManager.default + for file in files { + try Task.checkCancellation() + let destination = directory.appendingPathComponent(file.path) + if try PublishedCoreMLModelStore.matches(destination, file) { continue } + try manager.createDirectory(at: destination.deletingLastPathComponent(), withIntermediateDirectories: true) + let escaped = file.path.addingPercentEncoding(withAllowedCharacters: .urlPathAllowed) ?? file.path + guard + let url = URL(string: "https://huggingface.co/\(repository)/resolve/\(manifest.revision)/\(escaped)") + else { throw CuaS1FourBError.invalidAsset("invalid download URL for \(file.path)") } + progress?(file.path, 0) + let (temporary, response) = try await URLSession.shared.download(from: url) + defer { try? manager.removeItem(at: temporary) } + guard let http = response as? HTTPURLResponse, http.statusCode == 200 else { + throw CuaS1FourBError.invalidAsset( + "download of \(file.path) failed (\((response as? HTTPURLResponse)?.statusCode ?? -1))") + } + guard try PublishedCoreMLModelStore.matches(temporary, file) else { + throw CuaS1FourBError.invalidAsset("size or checksum mismatch for \(file.path)") + } + try LayaModelStore.installDownloadedFile(temporary, at: destination) + progress?(file.path, file.size) + } + return directory + } +} + +extension CuaS1FourBManager { + /// Download (pinned, verified) and load one configuration from the FluidUse model cache. + public static func load( + configuration: Configuration = Configuration(), cacheDirectory: URL? = nil, + progress: CuaS1FourBModelStore.Progress? = nil + ) async throws -> CuaS1FourBManager { + let directory = try await CuaS1FourBModelStore.ensure( + configuration: configuration, cacheDirectory: cacheDirectory, progress: progress) + return try await load(from: directory, configuration: configuration) + } +} diff --git a/Sources/FluidUse/CuaS1FourB/CuaS1FourBPrompt.swift b/Sources/FluidUse/CuaS1FourB/CuaS1FourBPrompt.swift new file mode 100644 index 0000000..34e13a2 --- /dev/null +++ b/Sources/FluidUse/CuaS1FourB/CuaS1FourBPrompt.swift @@ -0,0 +1,56 @@ +import Foundation + +/// Prompt contract of `cua_s1.four_b` (letters, `build_prompt`) rendered with the Qwen3.5 chat template. +/// +/// The template's generation prompt opens a thinking block (`\n`); Cua trains and evaluates the +/// adapters with exactly that suffix, so the letter logits are read after it. +public enum CuaS1FourBPrompt { + public static let letters = (UInt8(ascii: "A")...UInt8(ascii: "Z")).map { String(UnicodeScalar($0)) } + + public static let systemPrompt = + "You are a one-pass computer-use decision model. You are shown the current state of a screen and a " + + "fixed, closed list of candidate (element, action) options, each given a single letter. Choose exactly " + + "one option: the single best next action to take. Answer with ONLY that option's letter -- no words, " + + "no punctuation, no explanation." + + /// Placeholder the host expands to one `<|image_pad|>` per merged image token. + public static let imagePlaceholder = "<|vision_start|><|image_pad|><|vision_end|>" + + public static func optionLine(letter: String, option: CuaS1FourBOption) -> String { + var action = option.action + if option.action == "fill", let entity = option.entityId, !entity.isEmpty { + action += " (with entity '\(entity)')" + } + return "\(letter). \(option.role) \"\(option.label)\" -> \(action)" + } + + /// `build_prompt`'s user text. + public static func userText(state: CuaS1FourBState, modality: CuaS1FourBModality) throws -> String { + guard !state.options.isEmpty else { throw CuaS1FourBError.invalidInput("no options") } + guard state.options.count <= letters.count else { + throw CuaS1FourBError.invalidInput("\(state.options.count) options exceeds the 26-letter budget") + } + let lines = zip(letters, state.options).map { optionLine(letter: $0, option: $1) }.joined(separator: "\n") + var text = "" + if let goal = state.goal, !goal.isEmpty { text += "Goal: \(goal)\n\n" } + text += "App: \(state.app)\nTask family: \(state.taskFamily)\n\n" + switch modality { + case .text: + guard let tree = state.accessibilityTree, !tree.isEmpty else { + throw CuaS1FourBError.invalidInput("text modality requires an accessibility tree") + } + text += "Accessibility tree:\n\(tree)\n\n" + case .multimodal: + text += "The current screenshot is attached.\n\n" + } + return text + "Options:\n\(lines)\n\nAnswer with a single letter." + } + + /// The full chat string passed to the tokenizer (`apply_chat_template(..., add_generation_prompt=True)`). + public static func chat(state: CuaS1FourBState, modality: CuaS1FourBModality) throws -> String { + let user = try userText(state: state, modality: modality) + let content = modality == .multimodal ? imagePlaceholder + user : user + return "<|im_start|>system\n\(systemPrompt)<|im_end|>\n<|im_start|>user\n\(content)<|im_end|>\n" + + "<|im_start|>assistant\n\n" + } +} diff --git a/Sources/FluidUse/CuaS1FourB/CuaS1FourBTypes.swift b/Sources/FluidUse/CuaS1FourB/CuaS1FourBTypes.swift new file mode 100644 index 0000000..1d780fb --- /dev/null +++ b/Sources/FluidUse/CuaS1FourB/CuaS1FourBTypes.swift @@ -0,0 +1,102 @@ +import CoreGraphics +import Foundation + +/// Errors from the Cua-S1-4B runtime. +public enum CuaS1FourBError: Error, LocalizedError, Sendable { + case invalidAsset(String) + case invalidModel(String) + case invalidInput(String) + case promptTooLong(tokens: Int, maximum: Int) + + public var errorDescription: String? { + switch self { + case .invalidAsset(let detail): return "Cua-S1-4B asset: \(detail)" + case .invalidModel(let detail): return "Cua-S1-4B model: \(detail)" + case .invalidInput(let detail): return "Cua-S1-4B input: \(detail)" + case .promptTooLong(let tokens, let maximum): + return "Cua-S1-4B prompt is \(tokens) tokens; the largest loaded bucket holds \(maximum)" + } + } +} + +/// Which LoRA adapter (and so which converted decoder) a manager runs. +public enum CuaS1FourBModality: String, Sendable, CaseIterable { + /// Accessibility-tree text state. + case text + /// Screenshot state (vision tower + decoder trained with the multimodal adapter). + case multimodal +} + +/// One candidate `(element, action)` decision for a screen state, as in `cua_s1.four_b.Option`. +public struct CuaS1FourBOption: Sendable, Hashable { + public var elementId: String + public var role: String + public var label: String + public var action: String + /// Only meaningful for `fill`: which extracted value would be entered. + public var entityId: String? + + public init(elementId: String, role: String, label: String, action: String, entityId: String? = nil) { + self.elementId = elementId + self.role = role + self.label = label + self.action = action + self.entityId = entityId + } +} + +/// The screen state and closed option list for one decision. +public struct CuaS1FourBState: Sendable { + public var app: String + public var taskFamily: String + /// The episode goal when the state itself does not show it. + public var goal: String? + /// Accessibility tree text (text modality). + public var accessibilityTree: String? + /// Screenshot (multimodal modality). + public var screenshot: CGImage? + public var options: [CuaS1FourBOption] + + public init( + app: String, taskFamily: String, goal: String? = nil, accessibilityTree: String? = nil, + screenshot: CGImage? = nil, options: [CuaS1FourBOption] + ) { + self.app = app + self.taskFamily = taskFamily + self.goal = goal + self.accessibilityTree = accessibilityTree + self.screenshot = screenshot + self.options = options + } +} + +/// Scored options for one state, in the caller's option order. +public struct CuaS1FourBDecision: Sendable { + public struct Scored: Sendable { + public let option: CuaS1FourBOption + public let letter: String + /// Raw answer-letter logit at the last prompt position. + public let logit: Float + /// Softmax over all option letters (the `FourBModel.forward` readout). + public let probability: Float + } + + public let options: [Scored] + /// Prompt length in tokens and the bucket it ran in. + public let tokens: Int + public let bucketLength: Int + + /// The single best option overall (nil only for an empty decision, which `decide` never returns). + public var best: Scored? { options.max { $0.logit < $1.logit } } + + /// Per element, the best of that element's own options (Cua's benchmark readout). + public func bestPerElement() -> [String: Scored] { + var result: [String: Scored] = [:] + for scored in options { + let id = scored.option.elementId + if let current = result[id], current.logit >= scored.logit { continue } + result[id] = scored + } + return result + } +} diff --git a/Sources/FluidUse/CuaS1FourB/CuaS1FourBVision.swift b/Sources/FluidUse/CuaS1FourB/CuaS1FourBVision.swift new file mode 100644 index 0000000..1fe13c3 --- /dev/null +++ b/Sources/FluidUse/CuaS1FourB/CuaS1FourBVision.swift @@ -0,0 +1,369 @@ +import Accelerate +@preconcurrency import CoreML +import CoreGraphics +import Foundation + +/// Host side of the Qwen3.5 vision tower: the `Qwen2VLImageProcessor` preprocessing and the +/// grid-dependent inputs the static Core ML graph takes (see mobius `qwen35_vision.py`). +/// +/// 1. `smart_resize` to multiples of 32 px (pixel budget capped at the model's patch budget); +/// 2. bicubic resampling with antialiasing on 8-bit RGB (PIL/torchvision uint8 semantics); +/// 3. patches in spatial-merge-window order, normalized to [-1, 1]; +/// 4. learned position table resampled bilinearly (align_corners) to the patch grid; +/// 5. axial 2D rotary tables; padded patches are masked out of attention. +final class CuaS1FourBVision { + struct ImageFeatures { + let array: MLMultiArray + let tokens: Int + let gridRows: Int + let gridCols: Int + let padTokenId: Int + let hiddenSize: Int + + func copyRow(_ row: Int, to destination: UnsafeMutableRawPointer) { + let bytes = hiddenSize * 2 + destination.copyMemory(from: array.dataPointer + row * bytes, byteCount: bytes) + } + } + + let model: MLModel + let maxPatches: Int + let positionTable: [Float] // [side * side, dim] + let side: Int + let dim: Int + let heads: Int + let patchSize: Int + let merge: Int + let ropeTheta: Double + let outHidden: Int + let padTokenId: Int + let minPixels = 65_536 + let maxPixels = 16_777_216 + + private init( + model: MLModel, maxPatches: Int, positionTable: [Float], side: Int, dim: Int, heads: Int, patchSize: Int, + merge: Int, ropeTheta: Double, outHidden: Int, padTokenId: Int + ) { + self.model = model + self.maxPatches = maxPatches + self.positionTable = positionTable + self.side = side + self.dim = dim + self.heads = heads + self.patchSize = patchSize + self.merge = merge + self.ropeTheta = ropeTheta + self.outHidden = outHidden + self.padTokenId = padTokenId + } + + static func load( + from directory: URL, tokenizer: QwenTokenizer, computeUnits: MLComputeUnits + ) async throws + -> CuaS1FourBVision + { + let manager = FileManager.default + let names = (try? manager.contentsOfDirectory(atPath: directory.path)) ?? [] + guard let bundle = names.filter({ $0.hasPrefix("CuaS1Vision_P") }).sorted().first else { + throw CuaS1FourBError.invalidAsset("no CuaS1Vision_P*.mlmodelc in \(directory.path)") + } + let base = directory.appendingPathComponent((bundle as NSString).deletingPathExtension) + let url = try await CuaS1FourBManager.compiledModel(base, manager: manager) + let configuration = MLModelConfiguration() + configuration.computeUnits = computeUnits + let model = try await MLModel.load(contentsOf: url, configuration: configuration) + guard let shape = model.modelDescription.inputDescriptionsByName["patches"]?.multiArrayConstraint?.shape, + let maxPatches = shape.first?.intValue + else { + throw CuaS1FourBError.invalidModel("vision model has no patches input") + } + + let configData = try Data(contentsOf: directory.appendingPathComponent("vision_config.json")) + guard let config = try JSONSerialization.jsonObject(with: configData) as? [String: Any], + let dim = config["hidden_size"] as? Int, let heads = config["num_heads"] as? Int, + let patch = config["patch_size"] as? Int, let merge = config["spatial_merge_size"] as? Int, + let positions = config["num_position_embeddings"] as? Int, let out = config["out_hidden_size"] as? Int + else { + throw CuaS1FourBError.invalidAsset("bad vision_config.json") + } + let theta = ((config["rope_parameters"] as? [String: Any])?["rope_theta"] as? Double) ?? 10_000 + let side = Int(Double(positions).squareRoot()) + let tableData = try Data(contentsOf: directory.appendingPathComponent("pos_embed_table.f16")) + guard tableData.count == positions * dim * 2 else { + throw CuaS1FourBError.invalidAsset("pos_embed_table.f16 is \(tableData.count) bytes") + } + var table = [Float](repeating: 0, count: positions * dim) + tableData.withUnsafeBytes { src in + table.withUnsafeMutableBytes { dst in + var input = vImage_Buffer( + data: UnsafeMutableRawPointer(mutating: src.baseAddress), height: 1, + width: vImagePixelCount(positions * dim), rowBytes: positions * dim * 2) + var output = vImage_Buffer( + data: dst.baseAddress, height: 1, width: vImagePixelCount(positions * dim), + rowBytes: positions * dim * 4) + vImageConvert_Planar16FtoPlanarF(&input, &output, 0) + } + } + guard let pad = tokenizer.tokenId("<|image_pad|>") else { + throw CuaS1FourBError.invalidAsset("tokenizer has no <|image_pad|>") + } + return CuaS1FourBVision( + model: model, maxPatches: maxPatches, positionTable: table, side: side, dim: dim, heads: heads, + patchSize: patch, merge: merge, ropeTheta: theta, outHidden: out, padTokenId: pad) + } + + /// One prediction on an all-padding input to trigger GPU specialization. + func prewarm() throws { + let patchDim = 3 * patchSize * patchSize + let headDim = dim / heads + let zeros = { (count: Int) in [Float](repeating: 0, count: count) } + let inputs: [String: Any] = [ + "patches": try CuaS1FourBManager.half(zeros(maxPatches * patchDim), shape: [maxPatches, patchDim]), + "pos_embed": try CuaS1FourBManager.half(zeros(maxPatches * dim), shape: [maxPatches, dim]), + "cos": try CuaS1FourBManager.half(zeros(maxPatches * headDim), shape: [maxPatches, headDim]), + "sin": try CuaS1FourBManager.half(zeros(maxPatches * headDim), shape: [maxPatches, headDim]), + "key_mask": try CuaS1FourBManager.half(zeros(maxPatches), shape: [1, maxPatches]), + ] + _ = try autoreleasepool { try model.prediction(from: MLDictionaryFeatureProvider(dictionary: inputs)) } + } + + func expandImagePads(_ ids: [Int], count: Int) throws -> [Int] { + guard let index = ids.firstIndex(of: padTokenId), ids.filter({ $0 == padTokenId }).count == 1 else { + throw CuaS1FourBError.invalidInput("prompt must contain exactly one <|image_pad|>") + } + return Array(ids[.. (height: Int, width: Int) { + Self.smartResize( + height: height, width: width, factor: patchSize * merge, minPixels: minPixels, + maxPixels: min(maxPixels, maxPatches * patchSize * patchSize)) + } + + /// `transformers.models.qwen2_vl.image_processing_qwen2_vl.smart_resize`. + static func smartResize( + height: Int, width: Int, factor: Int, minPixels: Int, maxPixels: Int + ) + -> (height: Int, width: Int) + { + let f = Double(factor) + let h = Double(height) + let w = Double(width) + var hBar = (h / f).rounded(.toNearestOrEven) * f + var wBar = (w / f).rounded(.toNearestOrEven) * f + if hBar * wBar > Double(maxPixels) { + let beta = (h * w / Double(maxPixels)).squareRoot() + hBar = max(f, (h / beta / f).rounded(.down) * f) + wBar = max(f, (w / beta / f).rounded(.down) * f) + } else if hBar * wBar < Double(minPixels) { + let beta = (Double(minPixels) / (h * w)).squareRoot() + hBar = (h * beta / f).rounded(.up) * f + wBar = (w * beta / f).rounded(.up) * f + } + return (Int(hBar), Int(wBar)) + } + + func features(for image: CGImage) throws -> ImageFeatures { + let rgb = try Self.rgbBytes(image) + let (height, width) = targetSize(height: image.height, width: image.width) + let resized = Self.resizeBicubicAA( + rgb, width: image.width, height: image.height, toWidth: width, toHeight: height) + let gh = height / patchSize + let gw = width / patchSize + let count = gh * gw + guard count <= maxPatches else { + throw CuaS1FourBError.invalidInput("\(count) patches exceed the vision budget \(maxPatches)") + } + + let patchDim = 3 * patchSize * patchSize + var patches = [Float](repeating: 0, count: maxPatches * patchDim) + var rows = [Int](repeating: 0, count: count) + var cols = [Int](repeating: 0, count: count) + var n = 0 + for bh in 0..<(gh / merge) { + for bw in 0..<(gw / merge) { + for mh in 0.. (Int, Int, Float) { + let src = Float(index) * Float(side - 1) / Float(max(size - 1, 1)) + let lower = Int(src.rounded(.down)) + return (min(lower, side - 1), min(lower + 1, side - 1), src - Float(lower)) + } + + /// 8-bit RGB, row-major, exactly the stored pixel values (no color management). + static func rgbBytes(_ image: CGImage) throws -> [UInt8] { + let width = image.width + let height = image.height + var rgba = [UInt8](repeating: 0, count: width * height * 4) + let space = + image.colorSpace.flatMap { $0.model == .rgb ? $0 : nil } ?? CGColorSpaceCreateDeviceRGB() + let drawn = rgba.withUnsafeMutableBytes { buffer -> Bool in + guard + let context = CGContext( + data: buffer.baseAddress, width: width, height: height, bitsPerComponent: 8, bytesPerRow: width * 4, + space: space, bitmapInfo: CGImageAlphaInfo.noneSkipLast.rawValue) + else { return false } + context.interpolationQuality = .none + context.draw(image, in: CGRect(x: 0, y: 0, width: width, height: height)) + return true + } + guard drawn else { throw CuaS1FourBError.invalidInput("could not read screenshot pixels") } + var rgb = [UInt8](repeating: 0, count: width * height * 3) + for i in 0..<(width * height) { + rgb[i * 3] = rgba[i * 4] + rgb[i * 3 + 1] = rgba[i * 4 + 1] + rgb[i * 3 + 2] = rgba[i * 4 + 2] + } + return rgb + } + + /// PIL-style separable bicubic (a = -0.5) with antialiasing on downscale, horizontal pass then + /// vertical, rounding to 8 bits between passes -- torchvision's uint8 `resize(antialias=True)`. + static func resizeBicubicAA(_ src: [UInt8], width: Int, height: Int, toWidth: Int, toHeight: Int) -> [UInt8] { + var current = src + var w = width + if toWidth != width { + let coeffs = coefficients(inSize: width, outSize: toWidth) + var out = [UInt8](repeating: 0, count: toWidth * height * 3) + for y in 0.. UInt8 { + UInt8(max(0, min(255, value.rounded(.toNearestOrAwayFromZero)))) + } + + private static func coefficients(inSize: Int, outSize: Int) -> [(Int, [Double])] { + let scale = Double(inSize) / Double(outSize) + let filterScale = max(scale, 1) + let support = 2 * filterScale + return (0.. Double { + let a = -0.5 + let x = abs(x) + if x < 1 { return ((a + 2) * x - (a + 3)) * x * x + 1 } + if x < 2 { return (((x - 5) * x + 8) * x - 4) * a } + return 0 + } +} diff --git a/Sources/FluidUse/CuaS1FourB/QwenTokenizer.swift b/Sources/FluidUse/CuaS1FourB/QwenTokenizer.swift new file mode 100644 index 0000000..2416a47 --- /dev/null +++ b/Sources/FluidUse/CuaS1FourB/QwenTokenizer.swift @@ -0,0 +1,163 @@ +import Foundation + +/// Byte-level BPE encoder for the Qwen3.5 `tokenizer.json` (encode only). +/// +/// Mirrors the HuggingFace `tokenizers` pipeline the reference uses: split out added (special) +/// tokens verbatim, NFC-normalize the remaining text, pre-tokenize with the file's split regex, +/// map UTF-8 bytes through the GPT-2 byte alphabet, then apply BPE merges by rank. +public final class QwenTokenizer: Sendable { + private let vocab: [String: Int] + private let mergeRank: [String: Int] + private let splitRegex: NSRegularExpression + private let byteChars: [String] + /// Added tokens, longest first so overlapping prefixes resolve like the reference trie. + private let addedTokens: [(text: String, id: Int)] + + public init(tokenizerJsonURL: URL) throws { + let data = try Data(contentsOf: tokenizerJsonURL) + guard let root = try JSONSerialization.jsonObject(with: data) as? [String: Any], + let model = root["model"] as? [String: Any], + let vocabAny = model["vocab"] as? [String: Any], + let mergesAny = model["merges"] as? [Any] + else { + throw CuaS1FourBError.invalidAsset("tokenizer.json is missing model.vocab / model.merges") + } + var vocab = [String: Int](minimumCapacity: vocabAny.count) + for (token, id) in vocabAny { + guard let id = id as? Int else { continue } + vocab[token] = id + } + guard vocab.count == vocabAny.count else { + throw CuaS1FourBError.invalidAsset( + "tokenizer.json vocab keys collided (\(vocabAny.count) -> \(vocab.count))") + } + self.vocab = vocab + + var ranks = [String: Int](minimumCapacity: mergesAny.count) + for (rank, merge) in mergesAny.enumerated() { + if let pair = merge as? [String], pair.count == 2 { + ranks["\(pair[0]) \(pair[1])"] = rank + } else if let text = merge as? String { + ranks[text] = rank + } + } + self.mergeRank = ranks + + var added: [(String, Int)] = [] + for entry in root["added_tokens"] as? [[String: Any]] ?? [] { + if let content = entry["content"] as? String, let id = entry["id"] as? Int { + added.append((content, id)) + } + } + self.addedTokens = added.sorted { $0.0.count > $1.0.count }.map { (text: $0.0, id: $0.1) } + + self.splitRegex = try NSRegularExpression(pattern: Self.splitPattern(root)) + self.byteChars = Self.bytesToUnicode() + } + + /// Token ids for `text`, exactly as `tokenizer(text)["input_ids"]` (no BOS/EOS is added by Qwen). + public func encode(_ text: String) -> [Int] { + var ids: [Int] = [] + var rest = Substring(text) + while !rest.isEmpty { + if let (range, id) = firstAddedToken(in: rest) { + encodeOrdinary(String(rest[rest.startIndex.. Int? { + addedTokens.first { $0.text == token }?.id ?? vocab[token] + } + + private func firstAddedToken(in text: Substring) -> (Range, Int)? { + var best: (Range, Int)? + for token in addedTokens { + guard let range = text.range(of: token.text, options: .literal) else { continue } + if let current = best, current.0.lowerBound <= range.lowerBound { continue } + best = (range, token.id) + } + return best + } + + private func encodeOrdinary(_ text: String, into ids: inout [Int]) { + guard !text.isEmpty else { return } + let normalized = text.precomposedStringWithCanonicalMapping as NSString + for match in splitRegex.matches( + in: normalized as String, range: NSRange(location: 0, length: normalized.length)) + { + let piece = normalized.substring(with: match.range) + let symbols = piece.utf8.map { byteChars[Int($0)] } + for token in bpe(symbols) { + if let id = vocab[token] { ids.append(id) } + } + } + } + + private func bpe(_ initial: [String]) -> [String] { + var symbols = initial + while symbols.count > 1 { + var bestRank = Int.max + var bestIndex = -1 + for i in 0..<(symbols.count - 1) { + if let rank = mergeRank["\(symbols[i]) \(symbols[i + 1])"], rank < bestRank { + bestRank = rank + bestIndex = i + } + } + guard bestIndex >= 0 else { break } + let left = symbols[bestIndex] + let right = symbols[bestIndex + 1] + var merged: [String] = [] + merged.reserveCapacity(symbols.count - 1) + var i = 0 + while i < symbols.count { + if i < symbols.count - 1, symbols[i] == left, symbols[i + 1] == right { + merged.append(left + right) + i += 2 + } else { + merged.append(symbols[i]) + i += 1 + } + } + symbols = merged + } + return symbols + } + + private static func splitPattern(_ root: [String: Any]) throws -> String { + let pre = root["pre_tokenizer"] as? [String: Any] + let steps = (pre?["pretokenizers"] as? [[String: Any]]) ?? (pre.map { [$0] } ?? []) + for step in steps where step["type"] as? String == "Split" { + if let pattern = step["pattern"] as? [String: Any], let regex = pattern["Regex"] as? String { + return regex + } + } + throw CuaS1FourBError.invalidAsset("tokenizer.json has no Split pre-tokenizer regex") + } + + /// GPT-2 `bytes_to_unicode`: printable bytes map to themselves, the rest to U+0100 onwards. + private static func bytesToUnicode() -> [String] { + var printable = Array(33...126) + Array(161...172) + Array(174...255) + var codepoints = printable + var next = 0 + for byte in 0..<256 where !printable.contains(byte) { + printable.append(byte) + codepoints.append(256 + next) + next += 1 + } + var table = [String](repeating: "", count: 256) + for (byte, codepoint) in zip(printable, codepoints) { + // all code points are below U+0144, so the scalar always exists + if let scalar = Unicode.Scalar(UInt32(codepoint)) { table[byte] = String(scalar) } + } + return table + } +} diff --git a/Sources/FluidUse/Resources/cua-s1-4b-manifest.json b/Sources/FluidUse/Resources/cua-s1-4b-manifest.json new file mode 100644 index 0000000..3408ca2 --- /dev/null +++ b/Sources/FluidUse/Resources/cua-s1-4b-manifest.json @@ -0,0 +1,591 @@ +{ + "files": [ + { + "path": ".gitattributes", + "sha256": "58a5f4c3d744ff888064aaf7518ca488782d1d6afda47f44fe4a0a5b8b7ac611", + "size": 1695 + }, + { + "path": "LICENSE", + "sha256": "cfc7749b96f63bd31c3c42b5c471bf756814053e847c10f3eb003417bc523d30", + "size": 11358 + }, + { + "path": "NOTICE", + "sha256": "9f73f8a6523e9aea3af0d427f8647b92044b2b7cd94b14d8b569850c9c981cbc", + "size": 541 + }, + { + "path": "README.md", + "sha256": "cd58d8b80235a5fd143586465689a02f8536edf01eab8135bc0cd15d77c96afc", + "size": 4719 + }, + { + "path": "embeddings.f16", + "sha256": "c78988d979f52a340f39cd69edd8e4b0a80218a5fa8d95b6158bfce4a3de2a34", + "size": 1271398400 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part0.mlmodelc/analytics/coremldata.bin", + "sha256": "ad1a7513cfe12f660466f8caf5a0330304dd39f2f8dc1d494029ee12386f5d48", + "size": 243 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part0.mlmodelc/coremldata.bin", + "sha256": "5af0f95d7b234833e08a067d77b70591172d9e67be14691ee05b2de2b78c0d12", + "size": 490 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part0.mlmodelc/model.mil", + "sha256": "5c299efe129e57a9c1f3e761e0b105dc14dcc559beb9aba8b781f1579c792198", + "size": 1915172 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part0.mlmodelc/weights/weight.bin", + "sha256": "bd8571457c3fec6e7a06f6799ec9ecb0d3eceed807051aafdfb20f3e0de97186", + "size": 908130560 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part1.mlmodelc/analytics/coremldata.bin", + "sha256": "2ab9c401fa03cea94220109b5a99d8d3e126bc4dfe9a8b8795c4af6f43631c37", + "size": 243 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part1.mlmodelc/coremldata.bin", + "sha256": "5a28fa2aa5602f54bc58465b89e687b3df281bc41ecdab6ef9f967b9942ff7e9", + "size": 491 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part1.mlmodelc/model.mil", + "sha256": "5c299efe129e57a9c1f3e761e0b105dc14dcc559beb9aba8b781f1579c792198", + "size": 1915172 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part1.mlmodelc/weights/weight.bin", + "sha256": "8c40820ba83ca3431486dcc7c248567bda8998989161da1a369a18064d5558bd", + "size": 908130560 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part2.mlmodelc/analytics/coremldata.bin", + "sha256": "337ad2064346632c34caad4880a565b33de715f08a06704ca44689aaaa6cc763", + "size": 243 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part2.mlmodelc/coremldata.bin", + "sha256": "7aa5553841ec5740e71790666238b058e9e1f5f3ce00f9454c6bb4e566216534", + "size": 492 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part2.mlmodelc/model.mil", + "sha256": "5c299efe129e57a9c1f3e761e0b105dc14dcc559beb9aba8b781f1579c792198", + "size": 1915172 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part2.mlmodelc/weights/weight.bin", + "sha256": "41fdc02d62ed44356ec8b7f686f444367196c24906a19375655e71f944a258b8", + "size": 908130560 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part3.mlmodelc/analytics/coremldata.bin", + "sha256": "6b02c5c85c40afb288cf7f3d32528ef4a278fa50adf1599ea878547b481d6033", + "size": 243 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part3.mlmodelc/coremldata.bin", + "sha256": "6514c969c42b8c968495ec9693ea3ccfe094addf8f2201dc22b9ebad76d8561e", + "size": 520 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part3.mlmodelc/model.mil", + "sha256": "a85777af6772cf8a135d7ef7ac040693ee59823eb0ef7e8ab382442dd70e5cb6", + "size": 1918965 + }, + { + "path": "multimodal/L2048-w8/CuaS1Decoder_part3.mlmodelc/weights/weight.bin", + "sha256": "8ab3bdef52431f6c73c6d726aebb41d218c1acfcad2bcc2dced710727da00e75", + "size": 908202612 + }, + { + "path": "multimodal/L2048-w8/config.json", + "sha256": "b2a052ddecae07bb7d57310d52a1222bf0dceacd68b8d0c94eb72535e1b216a8", + "size": 758 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part0.mlmodelc/analytics/coremldata.bin", + "sha256": "89399356131eef0793daeb7423d8c17c0bbcf4a103e473f7bdc8ec2067fcbd21", + "size": 243 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part0.mlmodelc/coremldata.bin", + "sha256": "33db45daf20f1a7dff87105e8cf3ce49a0ae4d5372afd75296739dd3ede8403a", + "size": 490 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part0.mlmodelc/model.mil", + "sha256": "70cf728c8bbffe502235d0536614808df4196ce9266ce6d63e09fae1675dfd67", + "size": 1905561 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part0.mlmodelc/weights/weight.bin", + "sha256": "f7d6f7b1a6917c354ec6649f764276e80f0ff5960e41467e01d43367c4044a32", + "size": 1799342464 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part1.mlmodelc/analytics/coremldata.bin", + "sha256": "e320683c4e4682da883193445674ed89a8995299d2c4e75640b9b5e922167fb4", + "size": 243 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part1.mlmodelc/coremldata.bin", + "sha256": "f6121e09cf9001fd1ae5a77e095a26b9b2116b714d424b6037ddb9ea99001982", + "size": 491 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part1.mlmodelc/model.mil", + "sha256": "70cf728c8bbffe502235d0536614808df4196ce9266ce6d63e09fae1675dfd67", + "size": 1905561 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part1.mlmodelc/weights/weight.bin", + "sha256": "bbc62aeaf7fdb8d865b0b50bd9752b9d4fa910477b9b3ef84e94329c89c02641", + "size": 1799342464 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part2.mlmodelc/analytics/coremldata.bin", + "sha256": "076a0a0e3aa4e011d7f18fcfda34f8658605c14600e95dbf10fa028a703b0c36", + "size": 243 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part2.mlmodelc/coremldata.bin", + "sha256": "461d48cd16badf5b3eab914e404806631f2ba8b3f03b8286c05b9d7a8781311a", + "size": 492 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part2.mlmodelc/model.mil", + "sha256": "70cf728c8bbffe502235d0536614808df4196ce9266ce6d63e09fae1675dfd67", + "size": 1905561 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part2.mlmodelc/weights/weight.bin", + "sha256": "ef70b8d0bb8834b80c4b4f801f77c3528df279781ed6b5dd9f589c6b54ed897f", + "size": 1799342464 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part3.mlmodelc/analytics/coremldata.bin", + "sha256": "0c9fa6ca037d4f97ede2e2b1d0b9f83bc08f197b570893218e94c63df12e7e6d", + "size": 243 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part3.mlmodelc/coremldata.bin", + "sha256": "28ef16f96434c95a4194a2ca33c04a44c3a1467f6b27fb9a072a2c52d145a7f3", + "size": 520 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part3.mlmodelc/model.mil", + "sha256": "2452cf544b60eb5be9112d7ba6d36e71474ea520fa2ccc51528782f0dab324a4", + "size": 1909184 + }, + { + "path": "multimodal/L2048/CuaS1Decoder_part3.mlmodelc/weights/weight.bin", + "sha256": "3cc518c6008c82f92ed14ade0178698558804ba60ce7474701ef924f23ed6d09", + "size": 1799480948 + }, + { + "path": "multimodal/L2048/config.json", + "sha256": "b2a052ddecae07bb7d57310d52a1222bf0dceacd68b8d0c94eb72535e1b216a8", + "size": 758 + }, + { + "path": "multimodal/vision/CuaS1Vision_P4096.mlmodelc/analytics/coremldata.bin", + "sha256": "499972aa6a54f539ad6ef9b298c3721ab46c3078946baca1866ea2b09f8f6271", + "size": 243 + }, + { + "path": "multimodal/vision/CuaS1Vision_P4096.mlmodelc/coremldata.bin", + "sha256": "b485dd2b246bd19113405091337b6202929cb166e43ae431caea044d199d9596", + "size": 568 + }, + { + "path": "multimodal/vision/CuaS1Vision_P4096.mlmodelc/model.mil", + "sha256": "f56fdf8fca461ed9cfa0e79000f217fb7146d9832cfa51e43727b0b6a3e91461", + "size": 358523 + }, + { + "path": "multimodal/vision/CuaS1Vision_P4096.mlmodelc/weights/weight.bin", + "sha256": "c71bbb8ab70bc861dc038ee57145ee1f5c2d3c8b97f2248c25a150cbfd4ea243", + "size": 660756032 + }, + { + "path": "multimodal/vision/pos_embed_table.f16", + "sha256": "fce3f0a15737ac29e31921f887e28235bda9cafa921230952cb8f20db6f95d84", + "size": 4718592 + }, + { + "path": "multimodal/vision/vision_config.json", + "sha256": "bd0cd72ccf05ce4dd7aac1eae658524717cde3c4a744573eacd740e6af439b48", + "size": 372 + }, + { + "path": "reports/gui360-text-coreml-fp16.json", + "sha256": "f266db33a2647ddfc68cfad4866f392c3eb017936455a9f2777f0f6b3956d2f9", + "size": 199509 + }, + { + "path": "reports/gui360-text-coreml-gptq.json", + "sha256": "85c0518bc9428db4ffd2de4670a02098fb945742c4c37ae8937579a747e79aa5", + "size": 199513 + }, + { + "path": "reports/gui360-text-torch-bf16-first100.json", + "sha256": "d3797749ed358dbff608a4335d8f1ae7d0a9bfb04477c8b3ccfa20be895b4aae", + "size": 28162 + }, + { + "path": "reports/parity-coreml-ane-L1024.json", + "sha256": "0393c9681b79c2f3bbdabc2343d474e790ef8cf955ae7fb4411ee255e5b64cd8", + "size": 5916 + }, + { + "path": "reports/parity-coreml-gpu-L1024-gptq.json", + "sha256": "0662dc7f017d49d1c412677ac96d0832d2044bda96f01ef17708330be2b805f6", + "size": 5949 + }, + { + "path": "reports/parity-coreml-gpu-L1024-m4.json", + "sha256": "3329a9a89dce53510b7b175bbf2462cdc66c374e75af808105a3b77828ec76da", + "size": 5944 + }, + { + "path": "reports/parity-coreml-gpu-L1024-m4b16.json", + "sha256": "451958d6dc06fb8ec38989c5b3d7cfe725607ef2cc880d745d1b117ae6190056", + "size": 5944 + }, + { + "path": "reports/parity-coreml-gpu-L1024-m4gu.json", + "sha256": "c90d3171b71d2df902af501879b7eeea659f3253c111dfaf6d8576b491c171fd", + "size": 5949 + }, + { + "path": "reports/parity-coreml-gpu-L1024-p2.json", + "sha256": "45fcd9a8f5ce0da344b48f68080a54d5193b84b01b9eb7067ee67d68166cc78a", + "size": 5921 + }, + { + "path": "reports/parity-coreml-gpu-L1024-p3.json", + "sha256": "f0d894db364ea9b678f4bdf77c377e89b8299cb14b7eedefccb85baa55bbcd67", + "size": 5919 + }, + { + "path": "reports/parity-coreml-gpu-L1024-p4.json", + "sha256": "6df65625979b42d3611059a2313150a2baeadd382a7203387bd852f5737cb296", + "size": 5928 + }, + { + "path": "reports/parity-coreml-gpu-L1024-w4.json", + "sha256": "d96df9fd463185205a2b08d134d3234646c989a7dc5b77fa60df215d33d3384e", + "size": 5930 + }, + { + "path": "reports/parity-coreml-gpu-L1024-w8.json", + "sha256": "adc8addf44ef5aa7a5076c0e09f44e044da0bcb69eccf958e1d4ed6e62c3aeaf", + "size": 5967 + }, + { + "path": "reports/parity-coreml-gpu-L1024.json", + "sha256": "6ed181a23d5b040182dc771c864458ed37ed658213d03a4251b40498acdfe201", + "size": 5996 + }, + { + "path": "reports/parity-coreml-gpu-multimodal-L2048-torchvision.json", + "sha256": "290b4c35924b6a5c671a83a6414bc7105020583d8d17aaaa09123bcf7049f351", + "size": 5999 + }, + { + "path": "reports/parity-coreml-gpu-multimodal-L2048-visionfp32.json", + "sha256": "b69cb0a7aed22446a4cee21533a39ab633855a8f560f82c03d0538b3e395b353", + "size": 5995 + }, + { + "path": "reports/parity-coreml-gpu-multimodal-L2048-visionmixed.json", + "sha256": "f21faa1992c3d50660410996e0e8308cd8b8f5376856afbe9d31fd3d940b9c11", + "size": 5985 + }, + { + "path": "reports/parity-coreml-gpu-multimodal-L2048-w8.json", + "sha256": "368811969d60efd96e79902b4c19a1a79edf487b561ef8a07951f2d4cfc51c25", + "size": 5961 + }, + { + "path": "reports/parity-coreml-gpu-multimodal-L2048.json", + "sha256": "15554e4204f0c97d13178c545c468e0e4df9cba4b2bfeaa0d143f8c19a766dff", + "size": 5985 + }, + { + "path": "reports/parity-torch-L1024.json", + "sha256": "10e90b31a77e2c0769fe43067e0b6901dfffb2f9b8e6a0558be64cf7ac686caa", + "size": 5969 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part0.mlmodelc/analytics/coremldata.bin", + "sha256": "5f057041d9b75c7e8a89a228c9080bc9430d2d574f1b1f688bdc0cbcbf3bf77a", + "size": 243 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part0.mlmodelc/coremldata.bin", + "sha256": "d676b12319624f44d416216471ab327e6ba540208b48fc16c44ecef4bb3f9aaa", + "size": 485 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part0.mlmodelc/model.mil", + "sha256": "8aba3fe21506c208470c8c89ab3650f49ef259f8a19790a3d8a7c7ec773daa5d", + "size": 1211713 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part0.mlmodelc/weights/weight.bin", + "sha256": "9cff1a783ab78c5ff888b1117db2ed328ba107235391b6b93fac152e2d8c1556", + "size": 688039040 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part1.mlmodelc/analytics/coremldata.bin", + "sha256": "28690bec1b738b61ec907704b581c95f1a0e1d594f88f6d59508c65353f482ce", + "size": 243 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part1.mlmodelc/coremldata.bin", + "sha256": "ef3181c83b193ed98c0990953c6a004358562586b8a89d1e63f4ca51ca3fe0a0", + "size": 486 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part1.mlmodelc/model.mil", + "sha256": "8aba3fe21506c208470c8c89ab3650f49ef259f8a19790a3d8a7c7ec773daa5d", + "size": 1211713 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part1.mlmodelc/weights/weight.bin", + "sha256": "f0d914d381790821013cf28811587fccb3a3609e9b404dd4231c50be578f2083", + "size": 688039040 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part2.mlmodelc/analytics/coremldata.bin", + "sha256": "168454b9b46085d76872e3c324f7ad38906390882ddb1f67469e8c508275b460", + "size": 243 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part2.mlmodelc/coremldata.bin", + "sha256": "6d3a0f20b971a5e22af03a89066839fad53de4d036ada8c1d10615e25db77bf4", + "size": 487 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part2.mlmodelc/model.mil", + "sha256": "8aba3fe21506c208470c8c89ab3650f49ef259f8a19790a3d8a7c7ec773daa5d", + "size": 1211713 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part2.mlmodelc/weights/weight.bin", + "sha256": "f955ff6c9c7ae75464606859a5792bdd05efc175f6fa59fbedd5cd041189b329", + "size": 688039040 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part3.mlmodelc/analytics/coremldata.bin", + "sha256": "33ca5c90baf1b5abdd0d7b5b4588a7753b350fbf74a02c0b52dfa77ed2e763d4", + "size": 243 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part3.mlmodelc/coremldata.bin", + "sha256": "6170608b66b11d5459f2ef08abb8962aebc0d439850beb841624892920faae4d", + "size": 515 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part3.mlmodelc/model.mil", + "sha256": "9fb20147c4640df58502746095dfa64abb24fe6a54a6c02917b8ff02b2c14707", + "size": 1215330 + }, + { + "path": "text/L1024-gptq/CuaS1Decoder_part3.mlmodelc/weights/weight.bin", + "sha256": "1b61d05e6a036595aeb5ce35fe45aea24420b0cc2adaf49eb1cf3ff643186f00", + "size": 688177524 + }, + { + "path": "text/L1024-gptq/config.json", + "sha256": "bcd3a46886bead9855137c225ebe03c0491da8fafdd5c67e57d20201fb76bcaf", + "size": 752 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part0.mlmodelc/analytics/coremldata.bin", + "sha256": "aff8740950c72e389255f63405bf15dad6305ccec0a782d0e71343f02bd5432c", + "size": 243 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part0.mlmodelc/coremldata.bin", + "sha256": "441761f964d66792cdfd581c1cd51b1091dc935eb16020277944b23864cc6d29", + "size": 484 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part0.mlmodelc/model.mil", + "sha256": "ebaf277aa760e7865482d4111a4dc6a396df1ad1226f0995a07877868cded14e", + "size": 1201453 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part0.mlmodelc/weights/weight.bin", + "sha256": "c32195cf98200cbde881592af0477bf5169dd23f9c6507d0045a72bbca3e4035", + "size": 899709184 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part1.mlmodelc/analytics/coremldata.bin", + "sha256": "2c4267a4dcdba14ddc6376c96547a2f573c2d089c71ee0f2b876da755f28fdb7", + "size": 243 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part1.mlmodelc/coremldata.bin", + "sha256": "ab31fb70dd34037e747b9cee43144b408654ec02e45c1930f6b26a9bf4c0cfc2", + "size": 485 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part1.mlmodelc/model.mil", + "sha256": "ebaf277aa760e7865482d4111a4dc6a396df1ad1226f0995a07877868cded14e", + "size": 1201453 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part1.mlmodelc/weights/weight.bin", + "sha256": "1163b1608b4738dffe6f9f29304f99f6cae755ada55cddb7dd1a348f36759a6e", + "size": 899709184 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part2.mlmodelc/analytics/coremldata.bin", + "sha256": "a298cf5ef478418a9b9ec770e3b5c04d412662f9141641d0e858b25a69554814", + "size": 243 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part2.mlmodelc/coremldata.bin", + "sha256": "f5a9a1c7af82eeced756ee6bd886c08261804c04883b14f5bf211f97b9097488", + "size": 486 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part2.mlmodelc/model.mil", + "sha256": "ebaf277aa760e7865482d4111a4dc6a396df1ad1226f0995a07877868cded14e", + "size": 1201453 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part2.mlmodelc/weights/weight.bin", + "sha256": "06934d68d52baab18b867675ec6f05e3dd4a1b7b2a6f15be02a9308536cd5d08", + "size": 899709184 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part3.mlmodelc/analytics/coremldata.bin", + "sha256": "30c676d4eb5d5053a053cf2b845c92f1382b6d8ba0685ef88f71720accd4d158", + "size": 243 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part3.mlmodelc/coremldata.bin", + "sha256": "7682b7ae5fd182ede378482d47c6f6c7b8b1328442e41a1b01f3bac22a3b3622", + "size": 514 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part3.mlmodelc/model.mil", + "sha256": "8cfa50f984b5ef9f4f33f2fb279adba68abd1a14a98418f41e57617d9e2a3e73", + "size": 1205070 + }, + { + "path": "text/L1024-w8/CuaS1Decoder_part3.mlmodelc/weights/weight.bin", + "sha256": "efdb32974f09ad7e6ead6587045a3216cf968c67149df66793cadb3e35ec6eb9", + "size": 899847668 + }, + { + "path": "text/L1024-w8/config.json", + "sha256": "bcd3a46886bead9855137c225ebe03c0491da8fafdd5c67e57d20201fb76bcaf", + "size": 752 + }, + { + "path": "text/L1024/CuaS1Decoder_part0.mlmodelc/analytics/coremldata.bin", + "sha256": "e1d93a8b00f5dd24374504efb7f4b726abc1a34e6466d001003b488eba84709e", + "size": 243 + }, + { + "path": "text/L1024/CuaS1Decoder_part0.mlmodelc/coremldata.bin", + "sha256": "bf62fee7317b824b73eef15e81f6e601736f893b6042135a112b936481b35763", + "size": 484 + }, + { + "path": "text/L1024/CuaS1Decoder_part0.mlmodelc/model.mil", + "sha256": "7fcafcb8b2119f817325c1ada6b3152fd87c78175cee8bee1f023fecc3ff2bc0", + "size": 1191842 + }, + { + "path": "text/L1024/CuaS1Decoder_part0.mlmodelc/weights/weight.bin", + "sha256": "e4147295a5eb9e8f7ffcd6c37a36dffb7e1b9303660f56f7ae5370f0e11835f8", + "size": 1790921088 + }, + { + "path": "text/L1024/CuaS1Decoder_part1.mlmodelc/analytics/coremldata.bin", + "sha256": "ce721e03b16433e53b1feb3669ce4d336c0901c4436ef08dfc9aac1860f2ad4c", + "size": 243 + }, + { + "path": "text/L1024/CuaS1Decoder_part1.mlmodelc/coremldata.bin", + "sha256": "ab31fb70dd34037e747b9cee43144b408654ec02e45c1930f6b26a9bf4c0cfc2", + "size": 485 + }, + { + "path": "text/L1024/CuaS1Decoder_part1.mlmodelc/model.mil", + "sha256": "7fcafcb8b2119f817325c1ada6b3152fd87c78175cee8bee1f023fecc3ff2bc0", + "size": 1191842 + }, + { + "path": "text/L1024/CuaS1Decoder_part1.mlmodelc/weights/weight.bin", + "sha256": "7a2231738313d9acf3aba58252a3e42489f6db1fc80100354317c24e61046749", + "size": 1790921088 + }, + { + "path": "text/L1024/CuaS1Decoder_part2.mlmodelc/analytics/coremldata.bin", + "sha256": "2d331b7b8db44b30e76406f85a62d9f1e952e97bcc23ccffe3897d7ef3b9972b", + "size": 243 + }, + { + "path": "text/L1024/CuaS1Decoder_part2.mlmodelc/coremldata.bin", + "sha256": "93221912892daa93baafa4fbc2af130752b5698d1c937e4592c36a91ea2d3d3a", + "size": 486 + }, + { + "path": "text/L1024/CuaS1Decoder_part2.mlmodelc/model.mil", + "sha256": "7fcafcb8b2119f817325c1ada6b3152fd87c78175cee8bee1f023fecc3ff2bc0", + "size": 1191842 + }, + { + "path": "text/L1024/CuaS1Decoder_part2.mlmodelc/weights/weight.bin", + "sha256": "d497c68fbd744fcc5b396db3c3f3ebb3877f127f833701d421f35ff9d8d44d12", + "size": 1790921088 + }, + { + "path": "text/L1024/CuaS1Decoder_part3.mlmodelc/analytics/coremldata.bin", + "sha256": "9ad7443eb081e642196344a11b649acb6087c3846a8d3b9590514fa5ca85a263", + "size": 243 + }, + { + "path": "text/L1024/CuaS1Decoder_part3.mlmodelc/coremldata.bin", + "sha256": "b1760b68a72888d8be00227d028d6b45fe8fdc4ef59c9e12e5401b454b3f3169", + "size": 514 + }, + { + "path": "text/L1024/CuaS1Decoder_part3.mlmodelc/model.mil", + "sha256": "1a65128cca492bf69bda7c9cd9766b6567d797e287320fafe792644e19338d46", + "size": 1195462 + }, + { + "path": "text/L1024/CuaS1Decoder_part3.mlmodelc/weights/weight.bin", + "sha256": "7cce28a842f2894fbc541cf0c8967041e3b55ad304f46a7655243ac5108338a5", + "size": 1791059572 + }, + { + "path": "text/L1024/config.json", + "sha256": "bcd3a46886bead9855137c225ebe03c0491da8fafdd5c67e57d20201fb76bcaf", + "size": 752 + }, + { + "path": "tokenizer.json", + "sha256": "5f9e4d4901a92b997e463c1f46055088b6cca5ca61a6522d1b9f64c4bb81cb42", + "size": 12807982 + } + ], + "repository": "FluidInference/cua-s1-4b-coreml", + "revision": "735de16e16534b9983af559860f443d7135e4ca0" +} diff --git a/Sources/FluidUseCuaS1/main.swift b/Sources/FluidUseCuaS1/main.swift new file mode 100644 index 0000000..b0301dc --- /dev/null +++ b/Sources/FluidUseCuaS1/main.swift @@ -0,0 +1,153 @@ +import CoreML +import Foundation +import FluidUse +import ImageIO + +/// `fluiduse-cua-s1` -- Cua-S1-4B-0.2 Core ML runtime checks. +/// +/// parity --models |hub [--cache ] --fixtures +/// [--screens ] [--variant w8|gptq] +/// +/// Rebuilds each fixture prompt with `CuaS1FourBPrompt`, checks the chat string and token ids against the +/// Python reference, runs the Core ML model and compares the letter softmax with the fp32 reference. + +struct FixtureOption: Decodable { + let elementId: String + let role: String + let label: String + let action: String + let entityId: String? +} + +struct FixtureTask: Decodable { + let id: String + let app: String + let taskFamily: String + let goal: String? + let axTree: String? + let screenshot: String? + let options: [FixtureOption] + let expected: [String: String] + let chat: String + let inputIds: [Int] + let letterLogits: [Float] +} + +struct FixtureFile: Decodable { + let modality: String + let tasks: [FixtureTask] +} + +func snakeCaseDecoder() -> JSONDecoder { + let decoder = JSONDecoder() + decoder.keyDecodingStrategy = .convertFromSnakeCase + return decoder +} + +func value(_ flag: String, in args: [String]) -> String? { + guard let i = args.firstIndex(of: flag), i + 1 < args.count else { return nil } + return args[i + 1] +} + +func softmax(_ x: [Float]) -> [Float] { + let m = x.max() ?? 0 + let e = x.map { expf($0 - m) } + let s = e.reduce(0, +) + return e.map { $0 / s } +} + +func loadImage(_ url: URL) throws -> CGImage { + guard let source = CGImageSourceCreateWithURL(url as CFURL, nil), + let image = CGImageSourceCreateImageAtIndex(source, 0, nil) + else { throw CuaS1FourBError.invalidInput("cannot read \(url.path)") } + return image +} + +func parity(_ args: [String]) async throws { + guard let models = value("--models", in: args), let fixturesPath = value("--fixtures", in: args) else { + print( + "usage: parity --models |hub [--cache ] --fixtures [--screens ] [--variant w8|gptq]" + ) + exit(2) + } + let fixtures = try snakeCaseDecoder().decode( + FixtureFile.self, from: Data(contentsOf: URL(fileURLWithPath: fixturesPath))) + guard let modality = CuaS1FourBModality(rawValue: fixtures.modality) else { exit(2) } + let screens = URL(fileURLWithPath: value("--screens", in: args) ?? "fixtures/screens") + var configuration = CuaS1FourBManager.Configuration(modality: modality) + configuration.variant = value("--variant", in: args) ?? "" + if let lengths = value("--lengths", in: args) { + configuration.lengths = lengths.split(separator: ",").compactMap { Int($0) } + } + let loadStart = Date() + let manager: CuaS1FourBManager + if models == "hub" { + // pinned, SHA-256 checked download into --cache (default: the FluidUse model cache) + let cache = value("--cache", in: args).map { URL(fileURLWithPath: $0) } + manager = try await CuaS1FourBManager.load(configuration: configuration, cacheDirectory: cache) { file, bytes in + if bytes > 0 { print(" downloaded \(file) (\(bytes / 1_048_576) MB)") } + } + } else { + manager = try await CuaS1FourBManager.load(from: URL(fileURLWithPath: models), configuration: configuration) + } + print(String(format: "loaded %@ in %.1f s", modality.rawValue, Date().timeIntervalSince(loadStart))) + let warmStart = Date() + try await manager.prewarm() + print(String(format: "prewarmed in %.1f s", Date().timeIntervalSince(warmStart))) + + var chatOK = 0 + var idsOK = 0 + var argmaxOK = 0 + var maxDp: Float = 0 + var times: [Double] = [] + for task in fixtures.tasks { + let state = CuaS1FourBState( + app: task.app, taskFamily: task.taskFamily, goal: task.goal, accessibilityTree: task.axTree, + screenshot: try task.screenshot.map { try loadImage(screens.appendingPathComponent($0)) }, + options: task.options.map { + CuaS1FourBOption( + elementId: $0.elementId, role: $0.role, label: $0.label, action: $0.action, entityId: $0.entityId) + }) + let chat = try CuaS1FourBPrompt.chat(state: state, modality: modality) + if chat == task.chat { chatOK += 1 } else { print(" chat mismatch: \(task.id)") } + if modality == .text { + let ids = manager.tokenizer.encode(task.chat) + if ids == task.inputIds { + idsOK += 1 + } else { + print(" token mismatch: \(task.id) \(ids.count) vs \(task.inputIds.count)") + } + } + let start = Date() + let decision = try await manager.decide(state) + times.append(Date().timeIntervalSince(start) * 1000) + if modality == .multimodal { + if decision.tokens == task.inputIds.count { + idsOK += 1 + } else { + print(" length mismatch: \(task.id) \(decision.tokens) vs \(task.inputIds.count)") + } + } + let got = softmax(decision.options.map(\.logit)) + let want = softmax(task.letterLogits) + let dp = zip(got, want).map { abs($0 - $1) }.max() ?? 0 + maxDp = max(maxDp, dp) + let gotArg = got.indices.max { got[$0] < got[$1] }! + let wantArg = want.indices.max { want[$0] < want[$1] }! + if gotArg == wantArg { argmaxOK += 1 } else { print(" argmax mismatch: \(task.id) dp=\(dp)") } + } + let sorted = times.dropFirst().sorted() + let median = sorted.isEmpty ? times[0] : sorted[sorted.count / 2] + let n = fixtures.tasks.count + print("chat \(chatOK)/\(n) tokens \(idsOK)/\(n) argmax \(argmaxOK)/\(n) max|dp| \(maxDp)") + print(String(format: "median decision %.0f ms (first after prewarm %.0f ms)", median, times[0])) +} + +let args = Array(CommandLine.arguments.dropFirst()) +switch args.first { +case "parity": + try await parity(Array(args.dropFirst())) +default: + print("usage: fluiduse-cua-s1 parity ...") + exit(2) +} diff --git a/Tests/FluidUseTests/CuaS1FourBTests.swift b/Tests/FluidUseTests/CuaS1FourBTests.swift new file mode 100644 index 0000000..1061766 --- /dev/null +++ b/Tests/FluidUseTests/CuaS1FourBTests.swift @@ -0,0 +1,93 @@ +import XCTest + +@testable import FluidUse + +final class CuaS1FourBTests: XCTestCase { + private func snakeCaseDecoder() -> JSONDecoder { + let decoder = JSONDecoder() + decoder.keyDecodingStrategy = .convertFromSnakeCase + return decoder + } + + private struct PromptFixture: Decodable { + struct Option: Decodable { + let elementId: String + let role: String + let label: String + let action: String + let entityId: String? + } + let app: String + let taskFamily: String + let goal: String? + let axTree: String? + let options: [Option] + let chat: String + } + + /// Chat string rendered by `cua_s1.four_b.build_prompt` + the Qwen3.5 chat template. + func testChatMatchesReference() throws { + let url = try XCTUnwrap( + Bundle.module.url(forResource: "cua-s1-4b-prompt", withExtension: "json", subdirectory: "Fixtures")) + let fixture = try snakeCaseDecoder().decode(PromptFixture.self, from: Data(contentsOf: url)) + let state = CuaS1FourBState( + app: fixture.app, taskFamily: fixture.taskFamily, goal: fixture.goal, + accessibilityTree: fixture.axTree, + options: fixture.options.map { + CuaS1FourBOption( + elementId: $0.elementId, role: $0.role, label: $0.label, action: $0.action, entityId: $0.entityId) + }) + XCTAssertEqual(try CuaS1FourBPrompt.chat(state: state, modality: .text), fixture.chat) + } + + func testFillOptionNamesEntityAndMultimodalPlaceholder() throws { + let state = CuaS1FourBState( + app: "portal", taskFamily: "form_filling", goal: "Sign up", + options: [ + CuaS1FourBOption(elementId: "el_0", role: "Edit", label: "Email", action: "fill", entityId: "ent_1"), + CuaS1FourBOption(elementId: "el_0", role: "Edit", label: "Email", action: "skip"), + ]) + let chat = try CuaS1FourBPrompt.chat(state: state, modality: .multimodal) + XCTAssertTrue(chat.contains("<|im_start|>user\n<|vision_start|><|image_pad|><|vision_end|>Goal: Sign up\n\n")) + XCTAssertTrue(chat.contains("A. Edit \"Email\" -> fill (with entity 'ent_1')\nB. Edit \"Email\" -> skip\n")) + XCTAssertTrue(chat.contains("The current screenshot is attached.\n\n")) + XCTAssertTrue(chat.hasSuffix("<|im_start|>assistant\n\n")) + XCTAssertThrowsError(try CuaS1FourBPrompt.chat(state: state, modality: .text)) + } + + func testRejectsMoreThanTwentySixOptions() { + let options = (0..<27).map { + CuaS1FourBOption(elementId: "el_\($0)", role: "Button", label: "B", action: "skip") + } + let state = CuaS1FourBState(app: "a", taskFamily: "f", accessibilityTree: "-", options: options) + XCTAssertThrowsError(try CuaS1FourBPrompt.chat(state: state, modality: .text)) + } + + /// `Qwen3_5Model.get_rope_index` on 3 text tokens, a 4x6-patch image (2x3 merged) and 2 text tokens. + func testMRopePositionsMatchReference() { + let pad = 248_056 + let ids = [1, 2, 3] + Array(repeating: pad, count: 6) + [4, 5] + let positions = CuaS1FourBManager.mropePositions( + ids: ids, padTokenId: pad, imageTokens: 6, gridRows: 2, gridCols: 3) + XCTAssertEqual( + positions, + [ + [0, 0, 0], [1, 1, 1], [2, 2, 2], [3, 3, 3], [3, 3, 4], [3, 3, 5], [3, 4, 3], [3, 4, 4], [3, 4, 5], + [6, 6, 6], [7, 7, 7], + ]) + } + + /// `smart_resize(h, w, factor=32, min_pixels=65536, max_pixels=...)` from transformers. + func testSmartResizeMatchesReference() { + let cases: [(Int, Int, Int, Int, Int)] = [ + (580, 760, 16_777_216, 576, 768), (1056, 760, 16_777_216, 1056, 768), (316, 760, 16_777_216, 320, 768), + (1080, 1920, 16_777_216, 1088, 1920), (1080, 1920, 1_048_576, 768, 1344), (100, 100, 16_777_216, 256, 256), + ] + for (h, w, maxPixels, wantH, wantW) in cases { + let got = CuaS1FourBVision.smartResize( + height: h, width: w, factor: 32, minPixels: 65_536, maxPixels: maxPixels) + XCTAssertEqual(got.height, wantH, "\(h)x\(w)") + XCTAssertEqual(got.width, wantW, "\(h)x\(w)") + } + } +} diff --git a/Tests/FluidUseTests/Fixtures/cua-s1-4b-prompt.json b/Tests/FluidUseTests/Fixtures/cua-s1-4b-prompt.json new file mode 100644 index 0000000..c1b8178 --- /dev/null +++ b/Tests/FluidUseTests/Fixtures/cua-s1-4b-prompt.json @@ -0,0 +1 @@ +{"id": "directory_pager-20783164-d32388b2", "app": "directory_pager", "task_family": "pagination", "goal": null, "ax_tree": "# Employee Directory - Page 1 of 4\n\nThis is ONE turn. Judge every element against the screen's CURRENT state as shown below, NOT against the state it would be in after your other choices this turn. So: do not submit or advance while ANY field on this screen is still empty and has a value available in the source record, or a required box is still unticked -- even if you are also choosing to fill or tick it in this same turn; advancing comes on a later turn. An empty field the record has no value for is not fillable and never blocks advancing. Only fill a field when the record holds a value that genuinely belongs in THAT field: never repurpose a value that belongs to a different field, a different person, or a different point in time.\nGoal: Page forward through the employee directory one page at a time. Stop as soon as the page counter shows you are already on the last page -- do not page past the end.\n\n- [el_0] Button \"Next page\"", "screenshot": null, "options": [{"element_id": "el_0", "role": "Button", "label": "Next page", "action": "skip", "entity_id": null}, {"element_id": "el_0", "role": "Button", "label": "Next page", "action": "click", "entity_id": null}], "expected": {"el_0": "click"}, "chat": "<|im_start|>system\nYou are a one-pass computer-use decision model. You are shown the current state of a screen and a fixed, closed list of candidate (element, action) options, each given a single letter. Choose exactly one option: the single best next action to take. Answer with ONLY that option's letter -- no words, no punctuation, no explanation.<|im_end|>\n<|im_start|>user\nApp: directory_pager\nTask family: pagination\n\nAccessibility tree:\n# Employee Directory - Page 1 of 4\n\nThis is ONE turn. Judge every element against the screen's CURRENT state as shown below, NOT against the state it would be in after your other choices this turn. So: do not submit or advance while ANY field on this screen is still empty and has a value available in the source record, or a required box is still unticked -- even if you are also choosing to fill or tick it in this same turn; advancing comes on a later turn. An empty field the record has no value for is not fillable and never blocks advancing. Only fill a field when the record holds a value that genuinely belongs in THAT field: never repurpose a value that belongs to a different field, a different person, or a different point in time.\nGoal: Page forward through the employee directory one page at a time. Stop as soon as the page counter shows you are already on the last page -- do not page past the end.\n\n- [el_0] Button \"Next page\"\n\nOptions:\nA. Button \"Next page\" -> skip\nB. Button \"Next page\" -> click\n\nAnswer with a single letter.<|im_end|>\n<|im_start|>assistant\n\n", "input_ids": [248045, 8678, 198, 2523, 513, 264, 799, 45398, 6165, 23895, 5307, 1558, 13, 1394, 513, 6625, 279, 1428, 1528, 314, 264, 4034, 321, 264, 8097, 11, 7629, 1103, 314, 8871, 318, 5911, 11, 1852, 8, 2519, 11, 1754, 2574, 264, 3074, 6321, 13, 21513, 6681, 799, 2904, 25, 279, 3074, 1786, 1727, 1852, 310, 1831, 13, 21134, 440, 25835, 421, 2904, 579, 6321, 1137, 874, 4105, 11, 874, 59429, 11, 874, 15673, 13, 248046, 198, 248045, 846, 198, 2095, 25, 6025, 605, 1361, 198, 6065, 2902, 25, 27565, 271, 82418, 4757, 25, 198, 2, 16358, 17494, 471, 5577, 220, 16, 314, 220, 19, 271, 1919, 369, 23287, 2404, 13, 19594, 1396, 2315, 2272, 279, 4034, 579, 41685, 1528, 430, 6625, 3559, 11, 4045, 2272, 279, 1528, 424, 1000, 381, 303, 1238, 678, 975, 11125, 411, 2404, 13, 1987, 25, 635, 524, 9042, 466, 11573, 1345, 4001, 2002, 383, 411, 4034, 369, 1990, 4147, 321, 682, 264, 869, 2420, 303, 279, 2450, 3150, 11, 466, 264, 2483, 3618, 369, 1990, 12689, 17952, 1137, 1442, 413, 488, 513, 1048, 18207, 310, 4990, 466, 9063, 424, 303, 411, 1788, 2404, 26, 41647, 3905, 383, 264, 2843, 2404, 13, 1473, 4147, 2002, 279, 3150, 682, 874, 869, 364, 369, 524, 4990, 470, 321, 2496, 9714, 41647, 13, 8020, 4990, 264, 2002, 948, 279, 3150, 9687, 264, 869, 421, 34032, 16673, 303, 24467, 2002, 25, 2496, 1996, 28292, 264, 869, 421, 16673, 310, 264, 2086, 2002, 11, 264, 2086, 1637, 11, 466, 264, 2086, 1406, 303, 854, 13, 198, 38663, 25, 5577, 4487, 1472, 279, 9086, 6025, 799, 2081, 506, 264, 854, 13, 13809, 430, 4970, 430, 279, 2081, 5373, 4774, 488, 513, 2582, 383, 279, 1483, 2081, 1137, 635, 524, 2081, 3162, 279, 809, 13, 271, 12, 498, 300, 62, 15, 60, 6393, 328, 5666, 2081, 1, 271, 3670, 25, 198, 32, 13, 6393, 328, 5666, 2081, 1, 1411, 10390, 198, 33, 13, 6393, 328, 5666, 2081, 1, 1411, 4066, 271, 15666, 440, 264, 3074, 6321, 13, 248046, 198, 248045, 74455, 198, 248068, 198], "image_grid_thw": null, "letter_logits": [16.526920318603516, 24.02028465270996]} \ No newline at end of file diff --git a/Tools/pin_cua_s1_4b.py b/Tools/pin_cua_s1_4b.py new file mode 100644 index 0000000..6b3fcef --- /dev/null +++ b/Tools/pin_cua_s1_4b.py @@ -0,0 +1,27 @@ +#!/usr/bin/env python3 +"""Regenerate the pinned file manifest for FluidInference/cua-s1-4b-coreml. + + python3 Tools/pin_cua_s1_4b.py > Sources/FluidUse/Resources/cua-s1-4b-manifest.json +""" + +from __future__ import annotations + +import json +import sys + +from pin_published_coreml import files + +REPOSITORY = "FluidInference/cua-s1-4b-coreml" + + +def main() -> None: + if len(sys.argv) != 2: + raise SystemExit(__doc__) + revision = sys.argv[1] + manifest = {"repository": REPOSITORY, "revision": revision, "files": files(REPOSITORY, revision)} + json.dump(manifest, sys.stdout, indent=1, sort_keys=True) + sys.stdout.write("\n") + + +if __name__ == "__main__": + main()