diff --git a/cmd/zoekt-merge-index/main.go b/cmd/zoekt-merge-index/main.go index a398616..8dd9b4b 100644 --- a/cmd/zoekt-merge-index/main.go +++ b/cmd/zoekt-merge-index/main.go @@ -2,6 +2,7 @@ package main import ( "bufio" + "fmt" "log" "os" "path/filepath" @@ -32,8 +33,7 @@ func merge(dstDir string, names []string) error { return err } -func main() { - paths := os.Args[1:] +func mergeCmd(paths []string) error { if paths[0] == "-" { paths = []string{} scanner := bufio.NewScanner(os.Stdin) @@ -41,12 +41,111 @@ func main() { paths = append(paths, strings.TrimSpace(scanner.Text())) } if err := scanner.Err(); err != nil { - log.Fatal(err) + return err } log.Printf("merging %d paths from stdin", len(paths)) } err := merge(filepath.Dir(paths[0]), paths) if err != nil { - log.Fatal(err) + return err + } + return nil +} + +// explode splits a shard into indiviual shards and places them in dstDir. +// If it returns without error, the input shard was deleted and the first +// result contains the list of all new shards. +// +// explode cleans up tmp files created in the process on a best effort basis. +func explode(dstDir string, inputShard string) error { + f, err := os.Open(inputShard) + if err != nil { + return err + } + defer f.Close() + + indexFile, err := zoekt.NewIndexFile(f) + if err != nil { + return err + } + defer indexFile.Close() + + exploded, err := zoekt.Explode(dstDir, indexFile) + defer func() { + // best effort removal of tmp files. If os.Remove failes, indexserver will delete + // the leftover tmp files during the next cleanup. + for tmpFn := range exploded { + os.Remove(tmpFn) + } + }() + if err != nil { + return fmt.Errorf("zoekt.Explode: %w", err) + } + var fns []string + for tmpFn, dstFn := range exploded { + err = os.Rename(tmpFn, dstFn) + if err != nil { + // clean up the shards we already renamed to avoid duplicate results. + for _, fn := range fns { + os.Remove(fn) + } + return fmt.Errorf("explode: rename failed: %w", err) + } + fns = append(fns, dstFn) + } + + // Don't remove the input shard if its name matches one of the destination + // shards. This can happen, for example, if the input shard is a simple shard. + for _, dstFn := range exploded { + if dstFn == inputShard { + return nil + } + } + + removeInputShard := func() (err error) { + defer func() { + if err != nil { + // delete the new shards to avoid duplicate results. + for _, fn := range fns { + os.Remove(fn) + } + } + }() + + paths, err := zoekt.IndexFilePaths(inputShard) + if err != nil { + return err + } + for _, path := range paths { + err = os.Remove(path) + if err != nil { + return err + } + } + return nil + } + + if err = removeInputShard(); err != nil { + return fmt.Errorf("explode: error removing input shard %s: %w", inputShard, err) + } + return nil +} + +func explodeCmd(path string) error { + return explode(filepath.Dir(path), path) +} + +func main() { + switch subCommand := os.Args[1]; subCommand { + case "merge": + if err := mergeCmd(os.Args[2:]); err != nil { + log.Fatal(err) + } + case "explode": + if err := explodeCmd(os.Args[2]); err != nil { + log.Fatal(err) + } + default: + log.Fatalf("unknown subcommand %s", subCommand) } } diff --git a/cmd/zoekt-merge-index/main_test.go b/cmd/zoekt-merge-index/main_test.go index e2171c0..b9fc128 100644 --- a/cmd/zoekt-merge-index/main_test.go +++ b/cmd/zoekt-merge-index/main_test.go @@ -49,3 +49,93 @@ func TestMerge(t *testing.T) { t.Errorf("got %v, want 2 files.", result.Files) } } + +// TODO (stefan): make zoekt-git-index deterministic to compare the simple shards +// byte by byte instead of by search results. + +// Merge 2 simple shards and then explode them. +func TestExplode(t *testing.T) { + dir := t.TempDir() + + v16Shards, err := filepath.Glob("../../testdata/shards/repo*_v16.*.zoekt") + if err != nil { + t.Fatal(err) + } + sort.Strings(v16Shards) + t.Log(v16Shards) + + err = merge(dir, v16Shards) + if err != nil { + t.Fatal(err) + } + + cs, err := filepath.Glob(filepath.Join(dir, "compound-*.zoekt")) + if err != nil { + t.Fatal(err) + } + err = explode(dir, cs[0]) + if err != nil { + t.Fatal(err) + } + + cs, err = filepath.Glob(filepath.Join(dir, "compound-*.zoekt")) + if err != nil { + t.Fatal(err) + } + + if len(cs) != 0 { + t.Fatalf("explode should have deleted the compound shard if it returned without error") + } + + exploded, err := filepath.Glob(filepath.Join(dir, "*.zoekt")) + if err != nil { + t.Fatal(err) + } + + if len(exploded) != len(v16Shards) { + t.Fatalf("the number of simple shards before %d and after %d should be the same", len(v16Shards), len(exploded)) + } + + ss, err := shards.NewDirectorySearcher(dir) + if err != nil { + t.Fatalf("NewDirectorySearcher(%s): %v", dir, err) + } + defer ss.Close() + + var sOpts zoekt.SearchOptions + ctx := context.Background() + + cases := []struct { + searchLiteral string + wantResults int + }{ + { + searchLiteral: "apple", + wantResults: 1, + }, + { + searchLiteral: "hello", + wantResults: 1, + }, + { + searchLiteral: "main", + wantResults: 2, + }, + } + + for _, c := range cases { + t.Run(c.searchLiteral, func(t *testing.T) { + q, err := query.Parse(c.searchLiteral) + if err != nil { + t.Fatalf("Parse(%s): %v", c.searchLiteral, err) + } + result, err := ss.Search(ctx, q, &sOpts) + if err != nil { + t.Fatalf("Search(%v): %v", q, err) + } + if got := len(result.Files); got != c.wantResults { + t.Fatalf("wanted %d results, got %d", c.wantResults, got) + } + }) + } +} diff --git a/cmd/zoekt-sourcegraph-indexserver/cleanup.go b/cmd/zoekt-sourcegraph-indexserver/cleanup.go index 7f7803e..5f91461 100644 --- a/cmd/zoekt-sourcegraph-indexserver/cleanup.go +++ b/cmd/zoekt-sourcegraph-indexserver/cleanup.go @@ -433,18 +433,34 @@ func (s *Server) vacuum() { } if info.Size() < s.minSizeBytes { - paths, err := zoekt.IndexFilePaths(path) - if err != nil { - debug.Printf("failed getting all file paths for %s", path) + // feature flag: place file EXPLODE in IndexDir + if _, err := os.Stat(filepath.Join(s.IndexDir, "EXPLODE")); err == nil { + cmd := exec.Command("zoekt-merge-index", "explode", path) + + s.muIndexDir.Lock() + b, err := cmd.CombinedOutput() + s.muIndexDir.Unlock() + + if err != nil { + debug.Printf("failed to explode compound shard %s: %s", path, string(b)) + } else { + shardsLog(s.IndexDir, "explode", []shard{{Path: path}}) + } + continue + } else { + paths, err := zoekt.IndexFilePaths(path) + if err != nil { + debug.Printf("failed getting all file paths for %s", path) + continue + } + s.muIndexDir.Lock() + for _, p := range paths { + os.Remove(p) + } + s.muIndexDir.Unlock() + shardsLog(s.IndexDir, "delete", []shard{{Path: path}}) continue } - s.muIndexDir.Lock() - for _, p := range paths { - os.Remove(p) - } - s.muIndexDir.Unlock() - shardsLog(s.IndexDir, "delete", []shard{{Path: path}}) - continue } s.muIndexDir.Lock() @@ -474,7 +490,7 @@ func removeTombstones(fn string) ([]*zoekt.Repository, error) { if mockMerger != nil { runMerge = mockMerger } else { - runMerge = exec.Command("zoekt-merge-index", fn).Run + runMerge = exec.Command("zoekt-merge-index", "merge", fn).Run } repos, _, err := zoekt.ReadMetadataPath(fn) diff --git a/cmd/zoekt-sourcegraph-indexserver/merge.go b/cmd/zoekt-sourcegraph-indexserver/merge.go index 6887703..89b1abf 100644 --- a/cmd/zoekt-sourcegraph-indexserver/merge.go +++ b/cmd/zoekt-sourcegraph-indexserver/merge.go @@ -213,7 +213,7 @@ func callMerge(shards []candidate) ([]byte, []byte, error) { return nil, nil, nil } - cmd := exec.Command("zoekt-merge-index", "-") + cmd := exec.Command("zoekt-merge-index", "merge", "-") outBuf := &bytes.Buffer{} errBuf := &bytes.Buffer{} diff --git a/merge.go b/merge.go index b02e283..84bd5c5 100644 --- a/merge.go +++ b/merge.go @@ -3,8 +3,10 @@ package zoekt import ( "crypto/sha1" "fmt" + "io" "io/ioutil" "log" + "net/url" "os" "path/filepath" "runtime" @@ -119,47 +121,132 @@ func merge(ds ...*indexData) (*IndexBuilder, error) { } } - doc := Document{ - Name: string(d.fileName(docID)), - // Content set below since it can return an error - // Branches set below since it requires lookups - SubRepositoryPath: d.subRepoPaths[repoID][d.subRepos[docID]], - Language: d.languageMap[d.getLanguage(docID)], - // SkipReason not set, will be part of content from original indexer. - } - - var err error - if doc.Content, err = d.readContents(docID); err != nil { + if err := addDocument(d, ib, repoID, docID); err != nil { return nil, err } + } + } - if doc.Symbols, _, err = d.readDocSections(docID, nil); err != nil { - return nil, err - } + return ib, nil +} + +// Explode takes an IndexFile f and creates 1 simple shard per repository +// contained in f. Explode returns a map of tmpName -> dstName. It is the +// responsibility of the caller to rename the temporary shard(s) and delete the +// input shard. +func Explode(dstDir string, f IndexFile) (map[string]string, error) { + searcher, err := NewSearcher(f) + if err != nil { + return nil, err + } + d := searcher.(*indexData) + + shardNames := make(map[string]string, len(d.repoMetaData)) - doc.SymbolsMetaData = make([]*Symbol, len(doc.Symbols)) - for i := range doc.SymbolsMetaData { - doc.SymbolsMetaData[i] = d.symbols.data(d.fileEndSymbol[docID] + uint32(i)) + writeShard := func(ib *IndexBuilder) error { + if len(ib.repoList) != 1 { + return fmt.Errorf("expected ib to contain exactly 1 repository") + } + fn := filepath.Join(dstDir, shardName(ib.repoList[0].Name, ib.indexFormatVersion, 0)) + fnTmp := fn + ".tmp" + shardNames[fnTmp] = fn + return builderWriteAll(fnTmp, ib) + } + + var ib *IndexBuilder + lastRepoID := -1 + for docID := uint32(0); int(docID) < len(d.fileBranchMasks); docID++ { + repoID := int(d.repos[docID]) + + if d.repoMetaData[repoID].Tombstone { + continue + } + + if repoID != lastRepoID { + if lastRepoID > repoID { + return shardNames, fmt.Errorf("non-contiguous repo ids in %s for document %d: old=%d current=%d", d.String(), docID, lastRepoID, repoID) } + lastRepoID = repoID - // calculate branches - { - mask := d.fileBranchMasks[docID] - id := uint32(1) - for mask != 0 { - if mask&0x1 != 0 { - doc.Branches = append(doc.Branches, d.branchNames[repoID][uint(id)]) - } - id <<= 1 - mask >>= 1 + if ib != nil { + if err := writeShard(ib); err != nil { + return shardNames, err } } - if err := ib.Add(doc); err != nil { - return nil, err + ib = newIndexBuilder() + ib.indexFormatVersion = IndexFormatVersion + if err := ib.setRepository(&d.repoMetaData[repoID]); err != nil { + return shardNames, err } } + + err := addDocument(d, ib, repoID, docID) + if err != nil { + return shardNames, err + } } - return ib, nil + if ib != nil { + if err := writeShard(ib); err != nil { + return shardNames, err + } + } + + return shardNames, nil +} + +func addDocument(d *indexData, ib *IndexBuilder, repoID int, docID uint32) error { + doc := Document{ + Name: string(d.fileName(docID)), + // Content set below since it can return an error + // Branches set below since it requires lookups + SubRepositoryPath: d.subRepoPaths[repoID][d.subRepos[docID]], + Language: d.languageMap[d.getLanguage(docID)], + // SkipReason not set, will be part of content from original indexer. + } + + var err error + if doc.Content, err = d.readContents(docID); err != nil { + return err + } + + if doc.Symbols, _, err = d.readDocSections(docID, nil); err != nil { + return err + } + + doc.SymbolsMetaData = make([]*Symbol, len(doc.Symbols)) + for i := range doc.SymbolsMetaData { + doc.SymbolsMetaData[i] = d.symbols.data(d.fileEndSymbol[docID] + uint32(i)) + } + + // calculate branches + { + mask := d.fileBranchMasks[docID] + id := uint32(1) + for mask != 0 { + if mask&0x1 != 0 { + doc.Branches = append(doc.Branches, d.branchNames[repoID][uint(id)]) + } + id <<= 1 + mask >>= 1 + } + } + return ib.Add(doc) +} + +// copied from builder package to avoid circular imports. +func hashString(s string) string { + h := sha1.New() + _, _ = io.WriteString(h, s) + return fmt.Sprintf("%x", h.Sum(nil)) +} + +// copied from builder package to avoid circular imports. +func shardName(name string, version, n int) string { + abs := url.QueryEscape(name) + if len(abs) > 200 { + abs = abs[:200] + hashString(abs)[:8] + } + return fmt.Sprintf("%s_v%d.%05d.zoekt", abs, version, n) } diff --git a/testdata/gen-shards.sh b/testdata/gen-shards.sh index 5221188..c4631de 100755 --- a/testdata/gen-shards.sh +++ b/testdata/gen-shards.sh @@ -2,12 +2,18 @@ set -ex +# generate repo17.v17.0000.zoekt cp -r repo repo17 go run ../cmd/zoekt-index -disable_ctags repo17 -go run ../cmd/zoekt-merge-index repo17_v16.00000.zoekt +go run ../cmd/zoekt-merge-index merge repo17_v16.00000.zoekt mv compound*zoekt repo17_v17.00000.zoekt rm -rf repo17 repo17_v16.00000.zoekt zoekt-builder-shard-log.tsv mv *.zoekt shards/ + +# generate repo2.v16.0000.zoekt +go run ../cmd/zoekt-index repo2 +rm zoekt-builder-shard-log.tsv +mv *.zoekt shards/ diff --git a/testdata/golden/TestReadSearch/repo2_v16.00000.golden b/testdata/golden/TestReadSearch/repo2_v16.00000.golden new file mode 100644 index 0000000..e4dd23d --- /dev/null +++ b/testdata/golden/TestReadSearch/repo2_v16.00000.golden @@ -0,0 +1,82 @@ +{ + "FormatVersion": 16, + "FeatureVersion": 12, + "FileMatches": [ + [ + { + "Score": 910, + "Debug": "", + "FileName": "main.go", + "Repository": "repo2", + "Branches": null, + "LineMatches": [ + { + "Line": "ZnVuYyBtYWluKCkgew==", + "LineStart": 33, + "LineEnd": 46, + "LineNumber": 7, + "Before": null, + "After": null, + "FileName": false, + "Score": 501, + "LineFragments": [ + { + "LineOffset": 0, + "Offset": 33, + "MatchLength": 9, + "SymbolInfo": null + } + ] + } + ], + "RepositoryID": 0, + "RepositoryPriority": 0, + "Content": null, + "Checksum": "Ju1TnQKZ6mE=", + "Language": "Go", + "SubRepositoryName": "", + "SubRepositoryPath": "", + "Version": "" + } + ], + [ + { + "Score": 710, + "Debug": "", + "FileName": "main.go", + "Repository": "repo2", + "Branches": null, + "LineMatches": [ + { + "Line": "cGFja2FnZSBtYWlu", + "LineStart": 0, + "LineEnd": 12, + "LineNumber": 1, + "Before": null, + "After": null, + "FileName": false, + "Score": 501, + "LineFragments": [ + { + "LineOffset": 0, + "Offset": 0, + "MatchLength": 7, + "SymbolInfo": null + } + ] + } + ], + "RepositoryID": 0, + "RepositoryPriority": 0, + "Content": null, + "Checksum": "Ju1TnQKZ6mE=", + "Language": "Go", + "SubRepositoryName": "", + "SubRepositoryPath": "", + "Version": "" + } + ], + null, + null + ] +} \ No newline at end of file diff --git a/testdata/repo2/main.go b/testdata/repo2/main.go new file mode 100644 index 0000000..c1c1a4b --- /dev/null +++ b/testdata/repo2/main.go @@ -0,0 +1,13 @@ +package main + +import ( + "fmt" +) + +func main() { + var b, c int = 1, 2 + fmt.Println(b, c) + + fruit := "apple" + fmt.Println(fruit) +} diff --git a/testdata/shards/repo2_v16.00000.zoekt b/testdata/shards/repo2_v16.00000.zoekt new file mode 100644 index 0000000000000000000000000000000000000000..7777c4d54239329a32fcdb25c06ceafeb30309ef GIT binary patch literal 2874 zcmXR&OwLYBPgTfG%*^BB%FHduFDg;c;NnzD%Pmpj(&XYwE6qy=%W7yURC95bB^D_p z=_n*CWagD9*eV$6C>U{ZrsbCC1r%lGmE`1UfFw1!xH!{_N;69otZWsO5(^4)Qk76u zfn_zhYPlE~7#MhY7&+OQ82Lnn8HGfc7#J8@K%xu`3``(bF)%R917WC3LGEE-U|7Y# zz_1aDp{l_YBba1h02u>ggN0Z)Sfp4O7#LnMG72#=Di$oaJj_zqwqiyL_Y4)EL#xD! zDyljIE-)~tFt9iU&X|Vqm zK;;)OuquGeTfhJ@e*pu}0fr(5h9w|*E+}0KqFERisu>tIFt9K%7(wY|28JCV%ZeBn zjxew`fXq1o(#Ofb@PYy2?k^zqN(>BO+7Kkk2MQm?76zUKhHM59{eU4AM1%do*uuca zz)%3<^EogmFfevN%w_BWnUl!?p>shrID8p<7pmYg{<^%bMaR~!o1A`_5;~G%- zr86*Y0O?l((YyyhiDC;#Jd=TO3&`F1V7h@p2}+U_!oYZi zf%gDIG6UlckUNSP7~g>W!zu2f7NIZZEFz>ACeI_Qz+0OrV&@oVV{8^|!7ithA(JE_ zsufTsXI`W2U@GgOX73cAVC$*I$igX?Y-l2Dt`fq;mM*EI7AdF5%EzT)DWn!t#hm7+ z#v#Wq<)+51>0hdB<)bFZkSdU3EO87BEb$Bs zEC~z@EQt&ZEJ+LuEXfQEEGY~OEU63(Ea?mkEEx<8ESU@pELjW;EIAAeEO`tJEcpx! zECmb_EOiVFEcFZw zEDa0{ER75dEKLjyEX@oIENu)7EFBCCES(GtEL{u?EZqzYEIkYiEWHd2EE5<_h89+)7J3H8My7^l z2Bt>Z28LD!21+_g0Xd18d5*!(o}NlpB}JvFI!Zo?dFiEz>8ZYn1xi-cO78heRtB{? zN>TZ#*$4+JDd{MAx+qyGC0iJp=VfOm=NDU+CY$DFl$aQ$E7ewmB@J|x{1U-#DM~HK zH&W733Jvm6vI6N1&Mz%W2DuU@<5ZNGmz7; zZe~enkWWZzZb43B2}rM!j*?quPAZ0=Pi9`KTTxViMoXTw%$!u`{JfIX zypm#AJE4IARDCiquyBH8LBS7_0;OICb_NEvBnAcs7O=+P)MT({K^k8&Fn~&OhIb%Y zUXVuDycCG1Kn|M5z`(%4Sq%21Z(;$cSpsU!fLbDpKvKLADeug@6yL-Gkb$6<4L1V= z!&U|c1|B|$G?M!n4l*z>C~)MZmgi*VrGgv?ZZ7dLFfdF4Ie{}Ty(lp^B(w5#&j> z$qWn(BHT&f^yZsboLvm^;zN)>IEo>O4kW_18q{{Q&Kzh41Pg&e zkcl7U443?5Se$_p#&%Ggv4Ro?NFB%z3=EtgmpSF+=jVd>pTM?5?SYGaWnf^CVqoAn J&A`Bv4giduNs9mg literal 0 HcmV?d00001