Initial commit: knox knowledge index

This commit is contained in:
2026-08-29 02:52:32 -07:00
commit 8eecde18b4
31 changed files with 6855 additions and 0 deletions
+258
View File
@@ -0,0 +1,258 @@
package index
import (
"math"
"regexp"
"sort"
"strings"
"github.com/david/knox/internal/db"
)
type TopicCluster struct {
Name string
Keywords []string
Score float64
Entries []db.Entry
BySource map[string]int
}
var tokenRE = regexp.MustCompile(`[a-z]{4,}`)
var hexTokenRE = regexp.MustCompile(`^[a-f0-9]{8,}$`)
func Tokenize(text string) []string { return tokenize(text) }
func tokenize(text string) []string {
text = strings.ToLower(text)
text = strings.ReplaceAll(text, "-", " ")
text = strings.ReplaceAll(text, "_", " ")
return tokenRE.FindAllString(text, -1)
}
type document struct {
id int
title string
text string
source string
entry db.Entry
}
// TF-IDF index built from all observations
type TFIDFIndex struct {
docCount int
docFreq map[string]int // term → how many docs contain it
termDocs []map[string]float64 // per-doc: term → TF score
documents []document
}
func BuildTFIDF(entries []db.Entry) *TFIDFIndex {
idx := &TFIDFIndex{
docFreq: make(map[string]int),
termDocs: make([]map[string]float64, 0, len(entries)),
}
for _, e := range entries {
text := e.Title + " " + e.Summary + " " + e.SourceID + " " + e.Project
tokens := tokenize(text)
if len(tokens) == 0 {
continue
}
doc := document{
id: idx.docCount,
title: e.Title,
text: text,
source: e.SourceID,
entry: e,
}
// Term frequency in this doc
tf := make(map[string]float64)
seen := make(map[string]bool)
for _, t := range tokens {
if hexTokenRE.MatchString(t) || len(t) > 25 {
continue
}
tf[t]++
if !seen[t] {
idx.docFreq[t]++
seen[t] = true
}
}
// Normalize TF by doc length
docLen := float64(len(tokens))
for t, c := range tf {
tf[t] = c / docLen
}
idx.termDocs = append(idx.termDocs, tf)
idx.documents = append(idx.documents, doc)
idx.docCount++
}
return idx
}
func (idx *TFIDFIndex) TFIDF(term string, docIdx int) float64 {
tf, ok := idx.termDocs[docIdx][term]
if !ok {
return 0
}
df := idx.docFreq[term]
if df == 0 {
return 0
}
idf := math.Log(float64(idx.docCount) / float64(df))
return tf * idf
}
// TopTerms returns the N highest TF-IDF scoring terms for a document
func (idx *TFIDFIndex) TopTerms(docIdx int, n int) []struct {
Term string
Score float64
} {
var scored []struct {
Term string
Score float64
}
for t := range idx.termDocs[docIdx] {
s := idx.TFIDF(t, docIdx)
if s > 0 {
scored = append(scored, struct {
Term string
Score float64
}{t, s})
}
}
sort.Slice(scored, func(i, j int) bool { return scored[i].Score > scored[j].Score })
if len(scored) > n {
scored = scored[:n]
}
return scored
}
// Cluster groups observations into topics by shared top TF-IDF terms
func (idx *TFIDFIndex) Cluster(minSharedTerms int, maxTopics int) []TopicCluster {
if maxTopics <= 0 {
maxTopics = 20
}
if minSharedTerms <= 0 {
minSharedTerms = 2
}
// Get top 5 terms per doc
type docTerms struct {
docIdx int
terms []string
}
var docTermList []docTerms
for i := 0; i < idx.docCount; i++ {
tt := idx.TopTerms(i, 5)
if len(tt) >= minSharedTerms {
terms := make([]string, len(tt))
for j, t := range tt {
terms[j] = t.Term
}
docTermList = append(docTermList, docTerms{i, terms})
}
}
// Greedy clustering: docs sharing >= minSharedTerms terms become a topic
var clusters []TopicCluster
assigned := make(map[int]bool)
for _, dt := range docTermList {
if assigned[dt.docIdx] {
continue
}
cluster := TopicCluster{
Keywords: dt.terms,
Score: 0,
BySource: make(map[string]int),
}
// The seed doc belongs to this cluster too.
if seed := idx.documents[dt.docIdx]; seed.source != "" {
cluster.BySource[seed.source]++
cluster.Entries = append(cluster.Entries, seed.entry)
}
// Find all docs sharing terms with this seed
for _, other := range docTermList {
if assigned[other.docIdx] {
continue
}
shared := 0
for _, t1 := range dt.terms {
for _, t2 := range other.terms {
if t1 == t2 {
shared++
break
}
}
}
if shared >= minSharedTerms {
assigned[other.docIdx] = true
e := idx.documents[other.docIdx]
cluster.BySource[e.source]++
cluster.Entries = append(cluster.Entries, e.entry)
// Score is sum of TF-IDF of shared terms
for _, t := range dt.terms {
cluster.Score += idx.TFIDF(t, other.docIdx)
}
}
}
if len(cluster.BySource) > 0 {
// Name the topic by top scoring TF-IDF terms across all docs
allTerms := make(map[string]float64)
for _, t := range dt.terms {
for i := 0; i < idx.docCount; i++ {
allTerms[t] += idx.TFIDF(t, i)
}
}
type kv struct {
k string
v float64
}
var sorted []kv
for k, v := range allTerms {
sorted = append(sorted, kv{k, v})
}
sort.Slice(sorted, func(i, j int) bool { return sorted[i].v > sorted[j].v })
var nameParts []string
for _, kv := range sorted {
if len(nameParts) >= 3 {
break
}
nameParts = append(nameParts, kv.k)
}
cluster.Name = strings.Join(nameParts, " / ")
clusters = append(clusters, cluster)
if len(clusters) >= maxTopics {
break
}
}
}
return clusters
}
// Gaps finds topics with high browser activity but no skill coverage
func (idx *TFIDFIndex) Gaps(clusters []TopicCluster, skillKeywords map[string]bool) []TopicCluster {
var gaps []TopicCluster
for _, c := range clusters {
browserScore := c.BySource["browser-history"]
skillScore := c.BySource["skills-catalog"]
if browserScore > 0 && skillScore == 0 {
gaps = append(gaps, c)
}
}
sort.Slice(gaps, func(i, j int) bool {
return gaps[i].BySource["browser-history"] > gaps[j].BySource["browser-history"]
})
return gaps
}