mirror of
https://github.com/siyuan-note/siyuan.git
synced 2026-07-03 14:09:06 +02:00
36f4e3720e
* ✨ 搜索支持区分繁简选项 Add Simplified/Traditional Chinese sensitivity option for search 新增搜索配置 hanSensitive(区分繁简,默认开启 = 现状行为,命名与语义 向 caseSensitive 对齐)。关闭后简体与繁体可互相检索。 - FTS 层:关闭区分繁简时,blocks_fts / blocks_fts_case_insensitive 建表 追加 han_insensitive 分词器参数(依赖 88250/go-sqlite3 配套补丁), 繁体在分词时按单字折叠为简体,token 偏移仍指向原文; - 切换选项复用区分大小写现有的 FullReindex 流程; - 高亮:EncloseHighlighting 在关闭区分繁简时将关键字逐字符展开为 繁简等价字符类(如 诗经 -> [诗詩][经經]),与分词器使用同一份 OpenCC TSCharacters 映射数据; - 兼容:旧版本 conf.json 缺 hanSensitive 字段时按既往行为置为开启; setSearch 请求未携带该字段(现行前端/第三方调用)时保持当前值, 避免被零值意外翻转并触发重建索引; - 遵照配置界面重写计划(#17601),本次不改动设置界面与 i18n,仅添加 配置数据结构字段与前端类型定义(config.d.ts)。 范围说明:本次仅覆盖块全文搜索(关键字/查询语法)的命中与高亮; 历史/资源文件搜索、数据库属性过滤、虚拟引用、查找替换的替换动作 不在本次范围内(替换保持字节精确以避免误改原文)。 Fixes #13304 * 🐛 add missing util.SearchHanSensitive referenced by highlighting and sql.SetHanSensitive Copilot 评审指出 util.SearchHanSensitive 未定义导致无法编译:补上该全局 变量(默认 true,与既往行为一致,由 sql.SetHanSensitive 维护),并修正 conf/search.go 的 gofmt 对齐。已在本地以 dev 当前 go.mod(已合并的 88250/go-sqlite3 分词器)完成 go build -tags fts5 与 kernel/search 单测验证。
67 lines
2.1 KiB
Go
67 lines
2.1 KiB
Go
// SiYuan - Refactor your thinking
|
|
// Copyright (c) 2020-present, b3log.org
|
|
//
|
|
// This program is free software: you can redistribute it and/or modify
|
|
// it under the terms of the GNU Affero General Public License as published by
|
|
// the Free Software Foundation, either version 3 of the License, or
|
|
// (at your option) any later version.
|
|
//
|
|
// This program is distributed in the hope that it will be useful,
|
|
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
// GNU Affero General Public License for more details.
|
|
//
|
|
// You should have received a copy of the GNU Affero General Public License
|
|
// along with this program. If not, see <https://www.gnu.org/licenses/>.
|
|
|
|
package search
|
|
|
|
import (
|
|
"regexp"
|
|
"sort"
|
|
"strings"
|
|
)
|
|
|
|
// hanSimpToTrads 是 hanTradToSimp 的反向索引:简体字 -> 折叠到该简体的全部繁体字。
|
|
var hanSimpToTrads = map[rune][]rune{}
|
|
|
|
func init() {
|
|
for t, s := range hanTradToSimp {
|
|
hanSimpToTrads[s] = append(hanSimpToTrads[s], t)
|
|
}
|
|
for _, ts := range hanSimpToTrads {
|
|
sort.Slice(ts, func(i, j int) bool { return ts[i] < ts[j] })
|
|
}
|
|
}
|
|
|
|
// hanCharClass 返回与 r 繁简等价的所有字符(含 r 自身对应的简体),用于构造高亮正则。
|
|
func hanCharClass(r rune) (ret []rune) {
|
|
canon := r
|
|
if s, ok := hanTradToSimp[r]; ok {
|
|
canon = s
|
|
}
|
|
ret = append(ret, canon)
|
|
ret = append(ret, hanSimpToTrads[canon]...)
|
|
return
|
|
}
|
|
|
|
// hanInsensitiveRegexp 将关键字逐字符展开为繁简等价字符类,例如 "诗经" -> "[诗詩][经經]"。
|
|
// 仅用于搜索结果高亮;等价关系与 go-sqlite3 中 siyuan 分词器 han_insensitive 的映射表
|
|
// 来自同一份 OpenCC TSCharacters 数据,必须保持一致。
|
|
func hanInsensitiveRegexp(k string) string {
|
|
var b strings.Builder
|
|
for _, r := range k {
|
|
class := hanCharClass(r)
|
|
if 1 == len(class) {
|
|
b.WriteString(regexp.QuoteMeta(string(r)))
|
|
continue
|
|
}
|
|
b.WriteString("[")
|
|
for _, c := range class {
|
|
b.WriteString(regexp.QuoteMeta(string(c)))
|
|
}
|
|
b.WriteString("]")
|
|
}
|
|
return b.String()
|
|
}
|