- 给全部 78 个 Swift 源文件添加详细的中文注释(文件级、类级、方法级) - 删除 LegacyRDReaderController/ 死代码目录(16 文件 4592 行) - 根目录翻页容器文件移入 ReaderView/ 目录 - Resources/ 移入 EPUBCore/Resources/(与使用者归属一致) - RDEPUBTextIndexTable.swift 移入 EPUBTextRendering/(消除反向依赖) - RDURLReaderController.swift 移入 EPUBUI/(入口控制器归入 UI 层) - 更新 podspec 资源路径
132 lines
5.4 KiB
Swift
132 lines
5.4 KiB
Swift
// RDEPUBSearchEngine.swift
|
|
// EPUB 全文搜索引擎
|
|
// 遍历 spine 中的线性 HTML/XHTML 资源,提取纯文本后执行大小写不敏感的关键词搜索,
|
|
// 返回匹配项列表(含 href、progression、previewText 等)。
|
|
|
|
import Foundation
|
|
import UIKit
|
|
|
|
/// 搜索引擎协议,定义全文搜索接口
|
|
protocol RDEPUBSearchEngine {
|
|
func search(keyword: String) -> [RDEPUBSearchMatch]
|
|
}
|
|
|
|
/// HTML 全文搜索引擎实现
|
|
/// 通过解析 HTML 为纯文本,逐章执行关键词匹配
|
|
final class RDEPUBHTMLSearchEngine: RDEPUBSearchEngine {
|
|
/// EPUB 解析器(用于读取 HTML 内容)
|
|
private let parser: RDEPUBParser
|
|
/// EPUB 出版物(用于遍历 spine 和规范化 href)
|
|
private let publication: RDEPUBPublication
|
|
|
|
init(parser: RDEPUBParser, publication: RDEPUBPublication) {
|
|
self.parser = parser
|
|
self.publication = publication
|
|
}
|
|
|
|
/// 执行全文搜索:遍历所有线性 spine 项,提取纯文本后匹配关键词
|
|
func search(keyword: String) -> [RDEPUBSearchMatch] {
|
|
let normalizedKeyword = keyword.trimmingCharacters(in: .whitespacesAndNewlines)
|
|
guard !normalizedKeyword.isEmpty else {
|
|
return []
|
|
}
|
|
|
|
var searchMatches: [RDEPUBSearchMatch] = []
|
|
for item in publication.spine where item.linear {
|
|
guard item.mediaType.contains("html") || item.mediaType.contains("xhtml"),
|
|
let html = parser.htmlString(forRelativePath: item.href) else {
|
|
continue
|
|
}
|
|
|
|
let plainText = plainText(fromHTML: html, baseURL: parser.fileURL(forRelativePath: item.href)?.deletingLastPathComponent())
|
|
let normalizedHref = publication.resourceResolver.normalizedHref(item.href) ?? item.href
|
|
searchMatches.append(contentsOf: matches(in: plainText, href: normalizedHref, keyword: normalizedKeyword))
|
|
}
|
|
|
|
return searchMatches
|
|
}
|
|
|
|
/// 将 HTML 转换为纯文本(优先使用 NSAttributedString,失败时用正则兜底)
|
|
private func plainText(fromHTML html: String, baseURL: URL?) -> String {
|
|
guard let data = html.data(using: .utf8) else {
|
|
return fallbackPlainText(fromHTML: html)
|
|
}
|
|
|
|
var options: [NSAttributedString.DocumentReadingOptionKey: Any] = [
|
|
.documentType: NSAttributedString.DocumentType.html,
|
|
.characterEncoding: String.Encoding.utf8.rawValue
|
|
]
|
|
if let baseURL {
|
|
options[NSAttributedString.DocumentReadingOptionKey(rawValue: "NSBaseURLDocumentOption")] = baseURL
|
|
}
|
|
|
|
if let attributed = try? NSAttributedString(data: data, options: options, documentAttributes: nil) {
|
|
return attributed.string
|
|
}
|
|
return fallbackPlainText(fromHTML: html)
|
|
}
|
|
|
|
/// 兜底方案:用正则去除 HTML 标签并解码 HTML 实体
|
|
private func fallbackPlainText(fromHTML html: String) -> String {
|
|
let stripped = html.replacingOccurrences(of: "<[^>]+>", with: " ", options: .regularExpression)
|
|
return stripped
|
|
.replacingOccurrences(of: " ", with: " ")
|
|
.replacingOccurrences(of: "&", with: "&")
|
|
.replacingOccurrences(of: "<", with: "<")
|
|
.replacingOccurrences(of: ">", with: ">")
|
|
.replacingOccurrences(of: "'", with: "'")
|
|
.replacingOccurrences(of: """, with: "\"")
|
|
}
|
|
|
|
/// 在纯文本中查找所有关键词匹配项(大小写不敏感)
|
|
private func matches(in text: String, href: String, keyword: String) -> [RDEPUBSearchMatch] {
|
|
let source = text as NSString
|
|
let fullLength = source.length
|
|
guard fullLength > 0 else {
|
|
return []
|
|
}
|
|
|
|
var results: [RDEPUBSearchMatch] = []
|
|
var localMatchIndex = 0
|
|
var searchRange = NSRange(location: 0, length: fullLength)
|
|
|
|
while searchRange.length > 0 {
|
|
let foundRange = source.range(of: keyword, options: [.caseInsensitive], range: searchRange)
|
|
guard foundRange.location != NSNotFound else {
|
|
break
|
|
}
|
|
|
|
let progressionDenominator = max(fullLength - 1, 1)
|
|
let progression = Double(foundRange.location) / Double(progressionDenominator)
|
|
results.append(
|
|
RDEPUBSearchMatch(
|
|
href: href,
|
|
progression: progression,
|
|
previewText: previewText(in: source, matchRange: foundRange),
|
|
localMatchIndex: localMatchIndex,
|
|
rangeLocation: foundRange.location,
|
|
rangeLength: foundRange.length
|
|
)
|
|
)
|
|
|
|
localMatchIndex += 1
|
|
let nextLocation = foundRange.location + max(foundRange.length, 1)
|
|
if nextLocation >= fullLength {
|
|
break
|
|
}
|
|
searchRange = NSRange(location: nextLocation, length: fullLength - nextLocation)
|
|
}
|
|
|
|
return results
|
|
}
|
|
|
|
/// 截取匹配位置前后的文本作为预览(前后各 12 个字符)
|
|
private func previewText(in text: NSString, matchRange: NSRange) -> String {
|
|
let previewRadius = 12
|
|
let start = max(matchRange.location - previewRadius, 0)
|
|
let end = min(matchRange.location + matchRange.length + previewRadius, text.length)
|
|
let range = NSRange(location: start, length: max(end - start, 0))
|
|
return text.substring(with: range).trimmingCharacters(in: .whitespacesAndNewlines)
|
|
}
|
|
}
|