package search import ( "context" "strconv" "strings" "github.com/barerepo/server/internal/gitx" "github.com/barerepo/server/internal/repocfg" "github.com/barerepo/server/internal/store" "github.com/barerepo/server/internal/thread" ) // maxIndexedBytes keeps one generated file out of the whole index. Chapter 17 indexes source. const maxIndexedBytes = 512 << 10 // Index is what the indexer needs from the database, an interface because a hook is its own process. type Index interface { PutDoc(ctx context.Context, d store.Doc, public bool, readers string) error PutDocs(ctx context.Context, repo, kind string, docs []store.Doc, public bool, readers string) error DeleteDoc(ctx context.Context, repo, kind, path string) error SetReadable(ctx context.Context, repo string, public bool, readers string) error HasDocs(ctx context.Context, repo string) (bool, error) } // Target names one repository to the indexer, with everything the read filter needs. type Target struct { Owner string Name string Dir string Ref string Config repocfg.Config } func (r Target) full() string { return r.Owner + "/" + r.Name } func (r Target) readers() string { return store.Readers(r.Owner, r.Config.Access.Push) } // IndexAll rebuilds every document for one repository, for a first push and for a reindex. func IndexAll(ctx context.Context, db Index, r Target) error { if err := indexCodeAll(ctx, db, r); err != nil { return err } if err := IndexThreads(ctx, db, r); err != nil { return err } return indexRepoRow(ctx, db, r) } // IndexMeta rewrites what any push can change without touching a file: the read set and the talk. func IndexMeta(ctx context.Context, db Index, r Target) error { if err := db.SetReadable(ctx, r.full(), r.Config.Public(), r.readers()); err != nil { return err } if err := indexRepoRow(ctx, db, r); err != nil { return err } return IndexThreads(ctx, db, r) } // IndexPush walks only the paths the push changed, which is what chapter 17 means by incremental. func IndexPush(ctx context.Context, db Index, r Target, old, new string) error { indexed, err := db.HasDocs(ctx, r.full()) if err != nil { return err } if !indexed || !gitx.ValidRev(old) || strings.Trim(old, "0") == "" { return indexCodeAll(ctx, db, r) } changed, err := gitx.Run(ctx, r.Dir, "diff", "--name-only", "--no-renames", old, new) if err != nil { return indexCodeAll(ctx, db, r) } var paths []string for _, p := range strings.Split(strings.TrimRight(changed, "\n"), "\n") { if p != "" { paths = append(paths, p) } } return indexCodePaths(ctx, db, r, paths) } // indexCodeAll reads every path at the tip, which is the one place a whole tree is walked. func indexCodeAll(ctx context.Context, db Index, r Target) error { if !gitx.ValidRev(r.Ref) { return nil } out, err := gitx.Run(ctx, r.Dir, "ls-tree", "-r", "--name-only", r.Ref) if err != nil { return nil } var paths []string for _, p := range strings.Split(strings.TrimRight(out, "\n"), "\n") { if p != "" { paths = append(paths, p) } } docs, err := readBlobs(ctx, r, paths) if err != nil { return err } return db.PutDocs(ctx, r.full(), store.Code, docs, r.Config.Public(), r.readers()) } // indexCodePaths updates the paths a push touched, and drops the ones it removed. func indexCodePaths(ctx context.Context, db Index, r Target, paths []string) error { docs, err := readBlobs(ctx, r, paths) if err != nil { return err } kept := make(map[string]bool, len(docs)) for _, d := range docs { kept[d.Path] = true if err := db.PutDoc(ctx, d, r.Config.Public(), r.readers()); err != nil { return err } } for _, p := range paths { if kept[p] { continue } if err := db.DeleteDoc(ctx, r.full(), store.Code, p); err != nil { return err } } return nil } // readBlobs reads the named paths at the tip through the object pool, skipping what is not source. func readBlobs(ctx context.Context, r Target, paths []string) ([]store.Doc, error) { if len(paths) == 0 || !gitx.ValidRev(r.Ref) { return nil, nil } specs := make([]string, 0, len(paths)) for _, p := range paths { specs = append(specs, r.Ref+":"+p) } objs, err := gitx.Batch(ctx, r.Dir, specs) if err != nil { return nil, err } docs := make([]store.Doc, 0, len(paths)) for i, p := range paths { obj := objs[specs[i]] if obj == nil || obj.Type != "blob" || obj.Size > maxIndexedBytes { continue } if strings.IndexByte(obj.Body, 0) >= 0 { continue } docs = append(docs, store.Doc{Repo: r.full(), Kind: store.Code, Path: p, Body: obj.Body}) } return docs, nil } // IndexThreads rewrites the discussion, which chapter 17 asks for on note write and not on query. func IndexThreads(ctx context.Context, db Index, r Target) error { list, err := thread.List(ctx, r.Dir) if err != nil { return nil } docs := make([]store.Doc, 0, len(list)) for _, sum := range list { _, comments, err := thread.Read(ctx, r.Dir, sum.N) if err != nil { continue } var body strings.Builder for _, c := range comments { body.WriteString(c.Body) body.WriteByte('\n') } docs = append(docs, store.Doc{Repo: r.full(), Kind: store.Thread, Path: strconv.Itoa(sum.N), Title: sum.Meta.Title, Body: body.String()}) } return db.PutDocs(ctx, r.full(), store.Thread, docs, r.Config.Public(), r.readers()) } // indexRepoRow is the repository itself, matched on its name and its description. func indexRepoRow(ctx context.Context, db Index, r Target) error { doc := store.Doc{Repo: r.full(), Kind: store.Repository, Title: r.full(), Body: r.Config.Repo.Description} return db.PutDoc(ctx, doc, r.Config.Public(), r.readers()) }