Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@ DerivedData/
*.xcodeproj
.DS_Store
.venv/
.mobius/
Tools/doom/sauerkraut/models/
__pycache__/
_vizdoom.ini
Expand Down
2 changes: 2 additions & 0 deletions Package.swift
Original file line number Diff line number Diff line change
Expand Up @@ -55,6 +55,8 @@ let package = Package(
.executableTarget(name: "ImageSortDemo", dependencies: ["ImageSort"], exclude: ["README.md"]),
.executableTarget(name: "KevCheck", dependencies: ["FluidUse"]),
.executableTarget(name: "InternDecisionCheck", dependencies: ["FluidUse"]),
.executableTarget(name: "ShortReplyCheck", dependencies: ["FluidUse"]),
.executableTarget(name: "ShortReplyDemo", dependencies: ["FluidUse"], exclude: ["README.md", "demo.sh", "mock-feed"]),
.executableTarget(name: "GLiClassServe", dependencies: ["FluidUse"]),
.executableTarget(
name: "KevGuessWhoDemo", dependencies: ["FluidUse", "SortAnything"], exclude: ["README.md", "demo.sh"]),
Expand Down
18 changes: 18 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -187,6 +187,24 @@ moves and switches as `choice` options it matches its 4B teacher (24-6 vs poke-e
heuristic player over 30 battles). Buckets are 512, 640 and 1,024 tokens with int8 weights: about 0.85 GB in memory
with one bucket in use, 1.9 GB to download, the same 90 ms per decision as fp16.

## Short replies

`ShortReplyManager` drafts one short reply to a social post with a sub-1B model:
[short-reply-0.6b-coreml](https://huggingface.co/FluidInference/short-reply-0.6b-coreml), a Qwen3-0.6B fine-tune
(Apache-2.0) in one 724 MB package whose prefill runs on the Neural Engine and whose decode runs on the GPU. About
190 ms per reply on an M5 Pro; drafts are for a person to review, nothing is posted automatically.

```swift
let replies = try await ShortReplyManager.load(from: modelDirectory) // config.json, tokenizer.json, .mlpackage
let draft = try await replies.draft(for: "Finally passed my driving test on the third try.")
print(draft.reply, draft.timing.totalSeconds) // "Congrats on passing!" 0.19
```

`ShortReplyDemo` is a menu-bar app: open a post's reply box in any app (X in Chrome or Safari, Slack, Mail), press
**9**, and the draft is pasted into the box; **0** regenerates it.
`Sources/ShortReplyDemo/demo.sh --x` launches it with a macmon + log terminal
([Sources/ShortReplyDemo](Sources/ShortReplyDemo/README.md)).

## Demo

```bash
Expand Down
27 changes: 25 additions & 2 deletions Sources/FluidUse/Qwen/QwenBPETokenizer.swift
Original file line number Diff line number Diff line change
Expand Up @@ -2,14 +2,18 @@ import Foundation
import os

/// Byte-level BPE tokenizer for Qwen `tokenizer.json` files (Qwen2 through Qwen3.5): NFC normalization, added tokens
/// matched literally, Qwen's split regex, GPT-2 byte-to-unicode mapping, then merges by rank. Encoding only; no special
/// tokens are added.
/// matched literally, Qwen's split regex, GPT-2 byte-to-unicode mapping, then merges by rank. `encode` adds no
/// special tokens; `decode` drops them.
public final class QwenBPETokenizer: Sendable {
private let vocabulary: [String: Int]
private let mergeRanks: [String: Int]
/// Added tokens, longest first, so a longer token wins over one it contains.
private let addedTokens: [(content: String, id: Int)]
private let byteToCharacter: [Character]
/// Reverse tables for `decode`.
private let tokenForID: [Int: String]
private let byteForCharacter: [Character: UInt8]
private let specialIDs: Set<Int>
/// Encoded pieces; ordinary text repeats the same words, so this saves most of the merge loops.
private let cache = OSAllocatedUnfairLock<[String: [Int]]>(initialState: [:])

Expand Down Expand Up @@ -44,6 +48,13 @@ public final class QwenBPETokenizer: Sendable {
return (content, id)
}.sorted { $0.content.count > $1.content.count }
byteToCharacter = Self.bytesToUnicode()
var tokenForID = [Int: String](minimumCapacity: vocab.count)
for (token, id) in vocab { tokenForID[id] = token }
self.tokenForID = tokenForID
var byteForCharacter = [Character: UInt8](minimumCapacity: 256)
for (byte, character) in byteToCharacter.enumerated() { byteForCharacter[character] = UInt8(byte) }
self.byteForCharacter = byteForCharacter
specialIDs = Set(addedTokens.map(\.id))
_ = try Self.splitter()
}

Expand Down Expand Up @@ -72,6 +83,18 @@ public final class QwenBPETokenizer: Sendable {
return ids
}

/// Text for `ids`, dropping added (special) tokens; invalid byte sequences decode lossily.
public func decode(_ ids: [Int]) -> String {
var bytes: [UInt8] = []
for id in ids where !specialIDs.contains(id) {
guard let token = tokenForID[id] else { continue }
for character in token {
if let byte = byteForCharacter[character] { bytes.append(byte) }
}
}
return String(decoding: bytes, as: UTF8.self)
}

public func id(for token: String) -> Int? {
addedTokens.first { $0.content == token }?.id ?? vocabulary[token]
}
Expand Down
Loading
Loading