From cb72759b568baf34281bd15f821ac9834c17bbb9 Mon Sep 17 00:00:00 2001
From: ai_xiaopei <xiaopei@aisim.cn>
Date: Thu, 03 Sep 2026 14:24:29 +0800
Subject: [PATCH] merge: 合并 origin/master schema v2 重构(RWR/增量对账/explore),保留 tag/entity 实验于 experiment 分支 + 本地 expandPath 修复与 draft list 过滤

---
 internal/search/explore.go |  170 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++
 1 files changed, 170 insertions(+), 0 deletions(-)

diff --git a/internal/search/explore.go b/internal/search/explore.go
new file mode 100644
index 0000000..c0cb05d
--- /dev/null
+++ b/internal/search/explore.go
@@ -0,0 +1,170 @@
+package search
+
+import (
+	"regexp"
+	"strings"
+
+	"github.com/aisim/kb-cli/internal/index"
+)
+
+// headingRe 1-4 级 Markdown 标题
+var headingRe = regexp.MustCompile(`^#{1,4} .+$`)
+
+// splitParagraphs 按 1-4 级标题切段。无标题的文档整体为一段。
+func splitParagraphs(content string) []string {
+	lines := strings.Split(content, "\n")
+	var paras []string
+	var cur []string
+	flush := func() {
+		if len(cur) > 0 {
+			paras = append(paras, strings.TrimRight(strings.Join(cur, "\n"), "\n")+"\n")
+			cur = nil
+		}
+	}
+	for _, l := range lines {
+		if headingRe.MatchString(l) {
+			flush()
+		}
+		cur = append(cur, l)
+	}
+	flush()
+	return paras
+}
+
+// extractRelevantParagraphs 返回命中关键词的段落(整段不截半句)。
+// 文档总长 <= budget 时整篇输出;无命中段落时输出空串;
+// 预算约束:累计超预算的段落丢弃(不截半段)。
+func extractRelevantParagraphs(content string, keywords []string, budget int) string {
+	if budget <= 0 || len(content) <= budget {
+		return content
+	}
+	var out []string
+	for _, p := range splitParagraphs(content) {
+		for _, kw := range keywords {
+			if strings.Contains(p, kw) {
+				out = append(out, p)
+				break
+			}
+		}
+	}
+	// 预算约束:累计超预算的段落丢弃(不截半段)
+	var total int
+	kept := []string{}
+	for _, p := range out {
+		if total+len(p) > budget {
+			break
+		}
+		total += len(p)
+		kept = append(kept, p)
+	}
+	return strings.Join(kept, "")
+}
+
+// extractWithBudget extractRelevantParagraphs 的预算版(budget<=0 视为无预算)
+func extractWithBudget(content string, keywords []string, budget int) string {
+	if budget <= 0 {
+		budget = 100000
+	}
+	return extractRelevantParagraphs(content, keywords, budget)
+}
+
+// ExploreOptions explore 参数
+type ExploreOptions struct {
+	Budget     int // 字节预算(0 = 用配置默认)
+	TopN       int // 0 = 用配置默认
+	HardBudget int // 字节硬上限(0 = 用代码缺省 32000)
+}
+
+// ExploredDoc 入选文档及其输出正文
+type ExploredDoc struct {
+	Path    string `json:"path"`
+	Title   string `json:"title"`
+	Section string `json:"section"`
+	Score   int    `json:"score"`
+	Body    string `json:"body"`
+}
+
+// ExploreResult explore 结果
+type ExploreResult struct {
+	Docs            []ExploredDoc       `json:"docs"`
+	Related         map[string][]string `json:"related"`
+	UnresolvedLinks []string            `json:"unresolved_links"`
+}
+
+// Explore 一次调用返回相关文档原文 + 关联清单 + 悬空链接
+func Explore(store *index.Store, keywords []string, cfg ExploreOptions) (*ExploreResult, error) {
+	if len(keywords) == 0 {
+		return nil, nil
+	}
+	// 段落命中用原词(精确),FTS/LIKE 检索用展开后的 bigram
+	searchKeywords := ExpandCJKKeywords(keywords)
+	budget := cfg.Budget
+	if budget <= 0 {
+		budget = 16000 // 代码缺省(config 读取在 cmd 层完成)
+	}
+	if budget > 32000 {
+		budget = 32000 // 硬上限缺省
+	}
+	// 配置显式给出的硬上限优先(cmd 层读 config.yaml 的 explore.hard_budget)
+	if cfg.HardBudget > 0 && budget > cfg.HardBudget {
+		budget = cfg.HardBudget
+	}
+	topN := cfg.TopN
+	if topN <= 0 {
+		topN = 5
+	}
+
+	opts := SearchOptions{TopN: topN}
+	results, err := Search(store, searchKeywords, opts)
+	if err != nil {
+		return nil, err
+	}
+	if len(results) == 0 {
+		return &ExploreResult{Related: map[string][]string{}}, nil
+	}
+
+	// 按分数降序分配预算:每文档至少 800 字节
+	res := &ExploreResult{Related: map[string][]string{}}
+	perDoc := budget / len(results)
+	if perDoc < 800 {
+		perDoc = 800
+	}
+	for _, r := range results {
+		content, _, _, err := store.GetNodeContent(r.ID)
+		if err != nil || content == "" {
+			continue
+		}
+		// 两级回退:先用原词提取(精确);原词在任何段落都不连续出现时
+		//(复合 CJK 词常见:bigram 召回了文档,但原词不连续)回退到展开词提取,
+		// 避免大文档被 continue 静默丢弃、最终输出「未找到相关文档」。
+		body := extractRelevantParagraphs(content, keywords, perDoc)
+		if body == "" {
+			body = extractRelevantParagraphs(content, searchKeywords, perDoc)
+		}
+		if body == "" {
+			// 无命中段落但文档入选 → 整篇(若放得下),否则跳过
+			if len(content) <= perDoc {
+				body = content
+			} else {
+				continue
+			}
+		}
+		res.Docs = append(res.Docs, ExploredDoc{
+			Path: r.Path, Title: r.Title, Section: r.Section,
+			Score: r.Score, Body: body,
+		})
+		// 关联清单
+		if links, err := store.GetNodeLinks(r.ID); err == nil {
+			res.Related[r.Path] = links
+		}
+	}
+
+	// 悬空链接提示:入选文档的 wikilinks 中未解析的
+	for _, d := range res.Docs {
+		links, err := store.GetUnresolvedLinks(d.Path)
+		if err == nil {
+			res.UnresolvedLinks = append(res.UnresolvedLinks, links...)
+		}
+	}
+	return res, nil
+}

--
Gitblit v1.10.0