CTRunGetGlyphs返回的CGGlyph不存在时,如何手动合并字符?
问题:CTRun连字与Unicode映射处理
背景
我正在使用CTRunGetGlyphs处理NSAttributedString,目标是获取字符串"ffifion 🇬🇧"的字形并逐个传入其他函数。注意:ff应作为连字,字符串末尾包含Unicode旗帜表情,但当前使用的字体中没有ff连字。
返回的字形如下:
1st CTRun - 0 : 398 (ff) - 1 : 257 - 2 : 399 (fi) - 3 : 295 - 4 : 287 - 5 : 1 2nd CTRun - 0 : 428
当前遍历字形的代码
extension NSMutableAttributedString { func readGlyphs(size: CGSize) { var output: [Glyph] = [] var characterCounter: Int = 0 let attributedString = self let string = attributedString.string let framesetter = CTFramesetterCreateWithAttributedString(attributedString) let bounds = CGRect(x: 0, y: 0, width: size.width, height: size.height) let frame = CTFramesetterCreateFrame(framesetter, CFRange(), CGPath(rect: bounds, transform: nil), nil) let lines = CTFrameGetLines(frame) as? [CTLine] ?? [] let linesCount = lines.count let lineOrigins: [CGPoint] = [CGPoint](unsafeUninitializedCapacity: linesCount) { (bufferPointer, count) in if let baseAddress = bufferPointer.baseAddress { CTFrameGetLineOrigins(frame, CFRange(), baseAddress) count = linesCount } } for i in 0..<lineOrigins.count { let line = lines[i] guard let runs = CTLineGetGlyphRuns(line) as? [CTRun] else { continue } for run in runs { let runGlyphsCount = CTRunGetGlyphCount(run) let glyphPositions = [CGPoint](unsafeUninitializedCapacity: runGlyphsCount) { (bufferPointer, count) in if let baseAddress = bufferPointer.baseAddress { CTRunGetPositions(run, CFRange(), baseAddress) count = runGlyphsCount } } let glyphs = [CGGlyph](unsafeUninitializedCapacity: runGlyphsCount) { (bufferPointer, count) in if let baseAddress = bufferPointer.baseAddress { CTRunGetGlyphs(run, CFRange(), baseAddress) count = runGlyphsCount } } guard var attributes: [String: Any] = (CTRunGetAttributes(run) as NSDictionary as? [String: Any]) else { return } attributes = attributes .reduce([:]) { (partialResult: [String: Any], tuple: (key: String, value: Any)) in var result = partialResult result[tuple.key] = tuple.value return result } // swiftlint:disable force_cast let font = attributes["NSFont"] as! CTFont // swiftlint:enable force_cast let map = createUnicodeFontMap(ctFont: font) var indices = Array(repeating: CFIndex(), count: runGlyphsCount) CTRunGetStringIndices(run, CFRange(), &indices) for k in 0..<glyphs.count { let char: Character = Array(string)[k] let scalar: UnicodeScalar? = map[glyphs[k]] let indicie = indices[k] let charValue: String if let scalar { charValue = String(scalar) } else { charValue = String(char) } //processCharacter(char: ?????) characterCounter += 1 } } } } }
调用方式:
let _ = NSAttributedString("ffifion 🇬🇧").readGlyphs(size: CGSize(width: 380, height: 10_000))
核心问题
第一个CGGlyph无法找到对应的Unicode标量,如何将第一个字符与下一个合并为"ff",同时确保表情符号和其他有效Glyph正常工作?尝试过循环合并字符直到获取有效标量,但方法不够通用,需要能适配任意字符串的解决方案。
用于建立CGGlyph到UnicodeScalar映射的函数
func createUnicodeFontMap(ctFont: CTFont) -> [CGGlyph: UnicodeScalar] { let charset = CTFontCopyCharacterSet(ctFont) as CharacterSet var glyphToUnicode = [CGGlyph: UnicodeScalar]() // Start with empty map. // Enumerate all Unicode scalar values from the character set: for plane: UInt8 in 0...16 where charset.hasMember(inPlane: plane) { for unicode in UTF32Char(plane) << 16 ..< UTF32Char(plane + 1) << 16 { if let uniChar = UnicodeScalar(unicode), charset.contains(uniChar) { // Get glyph for this `uniChar` ... let utf16 = Array(uniChar.utf16) var glyphs = [CGGlyph](repeating: 0, count: utf16.count) if CTFontGetGlyphsForCharacters(ctFont, utf16, &glyphs, utf16.count) { // ... and add it to the map. glyphToUnicode[glyphs[0]] = uniChar } } } } return glyphToUnicode }
可行的解决方案(更新)
以下代码可满足需求,通过CTRunGetStringIndices返回的索引范围来获取每个Glyph对应的原字符串片段:
var char: String = "" let firstIndex = indices[k] if indices.indices.contains(k+1) { let slice = firstIndex..<indices[k+1] char = String(Array(string)[slice]) } else { if Array(string).indices.contains(firstIndex) { char = String(Array(string)[firstIndex]) } }
字体与Unicode的对应关系确实是一个复杂的领域,需要结合CoreText提供的索引映射来准确关联Glyph和原字符串内容。
内容的提问来源于stack exchange,提问作者Chris
相关产品推荐
相关产品推荐

