diff --git a/.goreleaser.yml b/.goreleaser.yml index db9bfd290..19b36f5a4 100644 --- a/.goreleaser.yml +++ b/.goreleaser.yml @@ -127,7 +127,7 @@ nfpms: vendor: "zzet" homepage: "https://github.com/zzet/gortex" maintainer: "zzet" - description: "Code intelligence engine that indexes repositories into an in-memory knowledge graph." + description: "Code intelligence engine that indexes repositories into a knowledge graph." license: "Custom" # NOTE: the homebrew cask is NOT generated here. goreleaser's `homebrew_casks` diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md index dbf087d11..f1b7b2a2a 100644 --- a/THIRD_PARTY_NOTICES.md +++ b/THIRD_PARTY_NOTICES.md @@ -248,29 +248,7 @@ For an automated license-aware report, run a tool such as - `github.com/aymanbagabas/go-osc52/v2` @ v2.0.1 - `github.com/aymanbagabas/go-udiff` @ v0.3.1 - `github.com/bits-and-blooms/bitset` @ v1.24.4 -- `github.com/blevesearch/bleve_index_api` @ v1.3.11 -- `github.com/blevesearch/bleve/v2` @ v2.6.0 -- `github.com/blevesearch/geo` @ v0.2.5 -- `github.com/blevesearch/go-faiss` @ v1.1.2 -- `github.com/blevesearch/go-metrics` @ v0.0.0-20201227073835-cf1acfcdf475 - `github.com/blevesearch/go-porterstemmer` @ v1.0.3 -- `github.com/blevesearch/goleveldb` @ v1.0.1 -- `github.com/blevesearch/gtreap` @ v0.1.1 -- `github.com/blevesearch/mmap-go` @ v1.2.0 -- `github.com/blevesearch/scorch_segment_api/v2` @ v2.4.7 -- `github.com/blevesearch/segment` @ v0.9.1 -- `github.com/blevesearch/snowball` @ v0.6.1 -- `github.com/blevesearch/snowballstem` @ v0.9.0 -- `github.com/blevesearch/stempel` @ v0.2.0 -- `github.com/blevesearch/upsidedown_store_api` @ v1.0.2 -- `github.com/blevesearch/vellum` @ v1.2.0 -- `github.com/blevesearch/zapx/v11` @ v11.4.3 -- `github.com/blevesearch/zapx/v12` @ v12.4.3 -- `github.com/blevesearch/zapx/v13` @ v13.4.3 -- `github.com/blevesearch/zapx/v14` @ v14.4.3 -- `github.com/blevesearch/zapx/v15` @ v15.4.3 -- `github.com/blevesearch/zapx/v16` @ v16.3.4 -- `github.com/blevesearch/zapx/v17` @ v17.1.3 - `github.com/campoy/embedmd` @ v1.0.0 - `github.com/charmbracelet/bubbles` @ v1.0.0 - `github.com/charmbracelet/bubbletea` @ v1.3.10 @@ -286,8 +264,6 @@ For an automated license-aware report, run a tool such as - `github.com/clipperhouse/stringish` @ v0.1.1 - `github.com/clipperhouse/uax29/v2` @ v2.7.0 - `github.com/coder/hnsw` @ v0.6.1 -- `github.com/couchbase/ghistogram` @ v0.1.0 -- `github.com/couchbase/moss` @ v0.2.0 - `github.com/cpuguy83/go-md2man/v2` @ v2.0.6 - `github.com/daulet/tokenizers` @ v1.27.0 - `github.com/davecgh/go-spew` @ v1.1.2-0.20180830191138-d8f796af33cc @@ -303,72 +279,84 @@ For an automated license-aware report, run a tool such as - `github.com/fsnotify/fsnotify` @ v1.10.1 - `github.com/fwcd/tree-sitter-kotlin` @ v0.0.0-20260411204054-55622a49bd59 - `github.com/go-errors/errors` @ v1.5.1 -- `github.com/go-logr/logr` @ v1.4.3 +- `github.com/go-logr/logr` @ v1.4.4 - `github.com/go-viper/mapstructure/v2` @ v2.5.0 - `github.com/gofrs/flock` @ v0.13.0 - `github.com/gofrs/uuid` @ v4.4.0+incompatible - `github.com/golang/freetype` @ v0.0.0-20170609003504-e2365dfdc4a0 - `github.com/golang/protobuf` @ v1.5.0 -- `github.com/golang/snappy` @ v1.0.0 - `github.com/gomlx/bsplines` @ v0.2.0 +- `github.com/gomlx/compute` @ v0.1.2 +- `github.com/gomlx/compute-onnx` @ v0.0.0-20260730095030-43334863f719 - `github.com/gomlx/exceptions` @ v0.0.3 -- `github.com/gomlx/go-huggingface` @ v0.3.5 -- `github.com/gomlx/go-xla` @ v0.2.2 -- `github.com/gomlx/gomlx` @ v0.27.3 -- `github.com/gomlx/onnx-gomlx` @ v0.4.2 +- `github.com/gomlx/go-huggingface` @ v0.4.1 +- `github.com/gomlx/go-xla` @ v0.4.1 +- `github.com/gomlx/gomlx` @ v0.28.2 +- `github.com/gomlx/onnx-gomlx` @ v0.5.2 - `github.com/google/go-cmp` @ v0.7.0 -- `github.com/google/gofuzz` @ v1.2.0 +- `github.com/google/go-github/v88` @ v88.0.0 +- `github.com/google/go-querystring` @ v1.2.0 - `github.com/google/jsonschema-go` @ v0.4.3 -- `github.com/google/pprof` @ v0.0.0-20240227163752-401108e1b7e7 +- `github.com/google/pprof` @ v0.0.0-20260802141513-ef3492d7dac3 - `github.com/google/renameio` @ v1.0.1 - `github.com/google/uuid` @ v1.6.0 +- `github.com/gortexhq/gcx-go` @ v0.1.0 +- `github.com/gortexhq/tree-sitter-dart` @ v0.1.0 +- `github.com/gortexhq/tree-sitter-dockerfile` @ v0.1.0 +- `github.com/gortexhq/tree-sitter-markdown` @ v0.1.0 +- `github.com/gortexhq/tree-sitter-org-mode` @ v0.1.0 +- `github.com/gortexhq/tree-sitter-protobuf` @ v0.1.0 +- `github.com/gortexhq/tree-sitter-sql` @ v0.1.0 +- `github.com/gortexhq/tree-sitter-swift` @ v0.1.1-0.20260424235305-8dde3a3327dd +- `github.com/hashicorp/golang-lru/v2` @ v2.0.7 - `github.com/inconshreveable/mousetrap` @ v1.1.0 +- `github.com/jackc/pgpassfile` @ v1.0.0 +- `github.com/jackc/pgservicefile` @ v0.0.0-20240606120523-5a60cdf6a761 +- `github.com/jackc/pgx/v5` @ v5.10.0 +- `github.com/jackc/puddle/v2` @ v2.2.2 - `github.com/janpfeifer/go-benchmarks` @ v0.1.1 - `github.com/janpfeifer/gonb` @ v0.11.3 - `github.com/janpfeifer/must` @ v0.2.0 -- `github.com/jedib0t/go-pretty/v6` @ v6.7.10 -- `github.com/json-iterator/go` @ v1.1.12 +- `github.com/jedib0t/go-pretty/v6` @ v6.8.3 - `github.com/klauspost/compress` @ v1.18.5 -- `github.com/klauspost/cpuid/v2` @ v2.3.0 -- `github.com/knights-analytics/hugot` @ v0.7.3 -- `github.com/knights-analytics/ortgenai` @ v0.3.1 +- `github.com/klauspost/cpuid/v2` @ v2.4.0 +- `github.com/knights-analytics/hugot` @ v0.7.7 +- `github.com/knights-analytics/ortgenai` @ v0.3.2 - `github.com/kr/pretty` @ v0.3.1 - `github.com/kr/text` @ v0.2.0 - `github.com/kylelemons/godebug` @ v1.1.0 -- `github.com/lucasb-eyer/go-colorful` @ v1.4.0 +- `github.com/ledongthuc/pdf` @ v0.0.0-20250511090121-5959a4027728 +- `github.com/lucasb-eyer/go-colorful` @ v1.4.1 - `github.com/MakeNowJust/heredoc` @ v1.0.0 -- `github.com/mark3labs/mcp-go` @ v0.54.0 -- `github.com/mattn/go-isatty` @ v0.0.22 +- `github.com/mark3labs/mcp-go` @ v0.57.0 +- `github.com/mattn/go-isatty` @ v0.0.24 - `github.com/mattn/go-localereader` @ v0.0.1 - `github.com/mattn/go-pointer` @ v0.0.1 -- `github.com/mattn/go-runewidth` @ v0.0.23 -- `github.com/mdempsky/unconvert` @ v0.0.0-20250216222326-4a038b3d31f5 +- `github.com/mattn/go-runewidth` @ v0.0.27 - `github.com/MetalBlueberry/go-plotly` @ v0.7.0 - `github.com/mitchellh/colorstring` @ v0.0.0-20190213212951-d06e56a500db -- `github.com/modern-go/concurrent` @ v0.0.0-20180306012644-bacd9c7ef1dd -- `github.com/modern-go/reflect2` @ v1.0.2 -- `github.com/mschoch/smat` @ v0.2.0 - `github.com/muesli/ansi` @ v0.0.0-20230316100256-276c6243b2f6 - `github.com/muesli/cancelreader` @ v0.2.2 - `github.com/muesli/termenv` @ v0.16.0 +- `github.com/ncruces/go-strftime` @ v1.0.0 - `github.com/parquet-go/bitpack` @ v1.0.0 - `github.com/parquet-go/jsonlite` @ v1.5.0 - `github.com/parquet-go/parquet-go` @ v0.29.0 - `github.com/pascaldekloe/name` @ v1.0.0 -- `github.com/pelletier/go-toml/v2` @ v2.3.1 +- `github.com/pelletier/go-toml/v2` @ v2.4.3 - `github.com/pierrec/lz4/v4` @ v4.1.26 - `github.com/pkg/errors` @ v0.9.1 - `github.com/pkg/profile` @ v1.7.0 - `github.com/pkoukk/tiktoken-go` @ v0.1.8 - `github.com/pkoukk/tiktoken-go-loader` @ v0.0.2 - `github.com/pmezard/go-difflib` @ v1.0.1-0.20181226105442-5d4384ee4fb2 +- `github.com/remyoudompheng/bigfft` @ v0.0.0-20230129092748-24d4a6f8daec - `github.com/rivo/uniseg` @ v0.4.7 -- `github.com/RoaringBitmap/roaring/v2` @ v2.18.0 - `github.com/rogpeppe/go-internal` @ v1.14.1 - `github.com/russross/blackfriday/v2` @ v2.1.0 - `github.com/sabhiram/go-gitignore` @ v0.0.0-20210923224102-525f6e181f06 - `github.com/sagikazarmark/locafero` @ v0.12.0 -- `github.com/sahilm/fuzzy` @ v0.1.2 +- `github.com/sahilm/fuzzy` @ v0.1.3 - `github.com/santhosh-tekuri/jsonschema/v6` @ v6.0.2 - `github.com/schollz/progressbar/v3` @ v3.19.0 - `github.com/sgtdi/fswatcher` @ v1.3.0 @@ -414,31 +402,28 @@ For an automated license-aware report, run a tool such as - `github.com/viant/xunsafe` @ v0.9.2 - `github.com/viterin/partial` @ v1.1.0 - `github.com/viterin/vek` @ v0.4.3 -- `github.com/x448/float16` @ v0.8.4 - `github.com/xo/terminfo` @ v0.0.0-20220910002029-abceb7e1c41e -- `github.com/yalue/onnxruntime_go` @ v1.30.1 +- `github.com/yalue/onnxruntime_go` @ v1.32.0 - `github.com/yosida95/uritemplate/v3` @ v3.0.2 - `github.com/yuin/goldmark` @ v1.4.13 -- `github.com/zeebo/assert` @ v1.1.0 +- `github.com/zeebo/assert` @ v1.3.0 - `github.com/zeebo/blake3` @ v0.2.4 - `github.com/zeebo/pcg` @ v1.0.1 -- `go.etcd.io/bbolt` @ v1.4.3 -- `go.etcd.io/gofail` @ v0.2.0 - `go.uber.org/goleak` @ v1.3.0 - `go.uber.org/multierr` @ v1.11.0 - `go.uber.org/zap` @ v1.28.0 -- `go.yaml.in/yaml/v3` @ v3.0.4 -- `golang.org/x/crypto` @ v0.52.0 -- `golang.org/x/exp` @ v0.0.0-20260508232706-74f9aab9d74a -- `golang.org/x/image` @ v0.41.0 -- `golang.org/x/mod` @ v0.36.0 -- `golang.org/x/net` @ v0.54.0 -- `golang.org/x/sync` @ v0.20.0 -- `golang.org/x/sys` @ v0.45.0 -- `golang.org/x/telemetry` @ v0.0.0-20260508192327-42602be52be6 -- `golang.org/x/term` @ v0.43.0 -- `golang.org/x/text` @ v0.37.0 -- `golang.org/x/tools` @ v0.45.0 +- `go.yaml.in/yaml/v3` @ v3.0.5 +- `golang.org/x/crypto` @ v0.54.0 +- `golang.org/x/exp` @ v0.0.0-20260727155853-b88d891fe743 +- `golang.org/x/image` @ v0.44.0 +- `golang.org/x/mod` @ v0.38.0 +- `golang.org/x/net` @ v0.57.0 +- `golang.org/x/sync` @ v0.22.0 +- `golang.org/x/sys` @ v0.47.0 +- `golang.org/x/telemetry` @ v0.0.0-20260708182218-49f421fb7959 +- `golang.org/x/term` @ v0.45.0 +- `golang.org/x/text` @ v0.40.0 +- `golang.org/x/tools` @ v0.48.0 - `golang.org/x/tools/go/expect` @ v0.1.1-deprecated - `golang.org/x/tools/go/packages/packagestest` @ v0.1.1-deprecated - `gonum.org/v1/gonum` @ v0.16.0 @@ -448,4 +433,18 @@ For an automated license-aware report, run a tool such as - `gopkg.in/yaml.v2` @ v2.4.0 - `gopkg.in/yaml.v3` @ v3.0.1 - `k8s.io/klog/v2` @ v2.140.0 -- `pgregory.net/rapid` @ v1.2.0 +- `modernc.org/cc/v4` @ v4.29.1 +- `modernc.org/ccgo/v4` @ v4.34.6 +- `modernc.org/fileutil` @ v1.4.0 +- `modernc.org/gc/v2` @ v2.6.5 +- `modernc.org/gc/v3` @ v3.1.4 +- `modernc.org/goabi0` @ v0.2.0 +- `modernc.org/libc` @ v1.74.4 +- `modernc.org/mathutil` @ v1.7.1 +- `modernc.org/memory` @ v1.11.0 +- `modernc.org/opt` @ v0.2.0 +- `modernc.org/sortutil` @ v1.2.1 +- `modernc.org/sqlite` @ v1.56.0 +- `modernc.org/strutil` @ v1.2.1 +- `modernc.org/token` @ v1.1.0 +- `pgregory.net/rapid` @ v1.3.0 diff --git a/bench/fixtures/retrieval.yaml b/bench/fixtures/retrieval.yaml index 75e191be7..2aa84265f 100644 --- a/bench/fixtures/retrieval.yaml +++ b/bench/fixtures/retrieval.yaml @@ -67,7 +67,7 @@ cases: - { id: exact-NewBM25, tier: exact, query: "NewBM25", expected: [internal/search/bm25.go::NewBM25] } - { id: exact-HybridBackend, tier: exact, query: "HybridBackend", expected: [internal/search/hybrid.go::HybridBackend] } - { id: exact-NewHybrid, tier: exact, query: "NewHybrid", expected: [internal/search/hybrid.go::NewHybrid] } - - { id: exact-rrfFuse, tier: exact, query: "rrfFuse", expected: [internal/search/hybrid.go::rrfFuse] } + - { id: exact-alphaFuse, tier: exact, query: "alphaFuse", expected: [internal/search/hybrid.go::alphaFuse] } - { id: exact-VectorBackend, tier: exact, query: "VectorBackend", expected: [internal/search/vector.go::VectorBackend] } - { id: exact-SearchBackend, tier: exact, query: "search.Backend", expected: [internal/search/search.go::Backend] } - { id: exact-SearchResult, tier: exact, query: "SearchResult", expected: [internal/search/search.go::SearchResult] } @@ -121,7 +121,7 @@ cases: - { id: concept-evict-repo, tier: concept, query: "drop every node for a repository prefix", expected: [internal/graph/graph.go::Graph.EvictRepo] } - { id: concept-bm25-index, tier: concept, query: "text search backend with TF-IDF ranking", expected: [internal/search/bm25.go::BM25Backend] } - { id: concept-hybrid-fuse, tier: concept, query: "combine text and vector search with RRF", expected: [internal/search/hybrid.go::HybridBackend] } - - { id: concept-rrf-algorithm, tier: concept, query: "reciprocal rank fusion", expected: [internal/search/hybrid.go::rrfFuse] } + - { id: concept-adaptive-alpha, tier: concept, query: "adaptive alpha weighted reciprocal rank fusion", expected: [internal/search/hybrid.go::alphaFuse] } - { id: concept-swap-backend, tier: concept, query: "hot-swap the in-memory search backend", expected: [internal/search/swappable.go::Swappable] } - { id: concept-glove-embed, tier: concept, query: "built-in GloVe word vector embedder", expected: [internal/embedding/static.go::StaticProvider] } - { id: concept-mcp-server, tier: concept, query: "MCP server type holding engine and graph", expected: [internal/mcp/server.go::Server] } @@ -230,7 +230,7 @@ cases: query: "search backend implementations" expected: - internal/search/bm25.go::BM25Backend - - internal/search/bleve.go::BleveBackend + - internal/search/symbolsearcher_backend.go::SymbolSearcherBackend - internal/search/hybrid.go::HybridBackend - internal/search/vector.go::VectorBackend - internal/search/swappable.go::Swappable diff --git a/bench/fixtures/retrieval_typo.yaml b/bench/fixtures/retrieval_typo.yaml index a12bc6405..d048c6d79 100644 --- a/bench/fixtures/retrieval_typo.yaml +++ b/bench/fixtures/retrieval_typo.yaml @@ -100,11 +100,11 @@ cases: query: NewHybid expected: - internal/search/hybrid.go::NewHybrid - - id: exact-rrfFuse-typo + - id: exact-alphaFuse-typo tier: exact - query: rrfFse + query: alphaFse expected: - - internal/search/hybrid.go::rrfFuse + - internal/search/hybrid.go::alphaFuse - id: exact-VectorBackend-typo tier: exact query: VectorBcakend @@ -345,11 +345,11 @@ cases: query: comlbine text and vector search with RRF expected: - internal/search/hybrid.go::HybridBackend - - id: concept-rrf-algorithm-typo + - id: concept-adaptive-alpha-typo tier: concept - query: reciprocal randk fusion + query: adaptive alpha weighted reciprocal randk fusion expected: - - internal/search/hybrid.go::rrfFuse + - internal/search/hybrid.go::alphaFuse - id: concept-swap-backend-typo tier: concept query: hot-swap the in-memory searich backend @@ -673,7 +673,7 @@ cases: query: search backend implementatsions expected: - internal/search/bm25.go::BM25Backend - - internal/search/bleve.go::BleveBackend + - internal/search/symbolsearcher_backend.go::SymbolSearcherBackend - internal/search/hybrid.go::HybridBackend - internal/search/vector.go::VectorBackend - internal/search/swappable.go::Swappable diff --git a/cmd/gortex/call_test.go b/cmd/gortex/call_test.go index 1dedf85bb..3d9f30042 100644 --- a/cmd/gortex/call_test.go +++ b/cmd/gortex/call_test.go @@ -234,7 +234,14 @@ func TestCall_AllCompactNamesUseFacadeRelay(t *testing.T) { callDaemonTool, callFacadeDaemonTool = origLegacy, origFacade }) - for _, name := range gortexmcp.FacadeToolNames() { + facadeNames := []string{ + "analyze", "ask", "capabilities", "change", "edit", "explore", + "overlay", "pr", "publish_review", "read", "recall", "refactor", + "relations", "remember", "response", "review", "search", "session", + "trace", "workspace", "workspace_admin", + } + for _, name := range facadeNames { + require.True(t, gortexmcp.IsFacadeToolName(name), "%q must remain a registered facade name", name) t.Run(name, func(t *testing.T) { var facadeTool string callDaemonTool = func(_ string, tool string, _ map[string]any) (json.RawMessage, error) { diff --git a/cmd/gortex/daemon.go b/cmd/gortex/daemon.go index 31cddbac1..36106c01c 100644 --- a/cmd/gortex/daemon.go +++ b/cmd/gortex/daemon.go @@ -55,9 +55,12 @@ var ( daemonHTTPConversationAllow []string daemonBackend string daemonBackendPath string - daemonBackendBufferPoolMB uint64 daemonTools string daemonToolsMode string + // daemonBackendBufferPoolMBIgnored is the sink for the retired + // --backend-buffer-pool-mb flag. Nothing reads it: SQLite sizes its page + // cache via a pragma, so there is no advisory cap to honour. + daemonBackendBufferPoolMBIgnored uint64 ) var daemonCmd = &cobra.Command{ @@ -144,8 +147,12 @@ func init() { "storage backend: sqlite (pure-Go embedded SQL, persists to --backend-path so warm restarts skip re-indexing). It is the only backend; point --backend-path at a throwaway file for a store that does not outlive the run") daemonStartCmd.Flags().StringVar(&daemonBackendPath, "backend-path", "", "path to the store file (its parent directory is created if absent). Defaults to ~/.gortex/store/store.sqlite") - daemonStartCmd.Flags().Uint64Var(&daemonBackendBufferPoolMB, "backend-buffer-pool-mb", 0, - "advisory page-cache cap (MiB) for the store. 0 reads $GORTEX_DAEMON_BUFFER_POOL_MB or lets the backend choose its own default; sqlite manages its own cache and ignores it") + daemonStartCmd.Flags().Uint64Var(&daemonBackendBufferPoolMBIgnored, "backend-buffer-pool-mb", 0, + "deprecated no-op; sqlite sizes its page cache via a pragma, so there is no advisory cap to set") + // Hidden rather than removed: cobra hard-errors on an unknown flag, so a + // deletion would break existing daemon-start scripts and the detach + // re-exec path. Nothing should learn it from --help. + _ = daemonStartCmd.Flags().MarkHidden("backend-buffer-pool-mb") daemonStartCmd.Flags().StringVar(&daemonTools, "tools", "", "restrict the published MCP tool surface to a preset: core (default)|full|readonly|edit|nav (optionally with ,+tool / ,-tool deltas). GORTEX_TOOLS overrides this") daemonStartCmd.Flags().StringVar(&daemonToolsMode, "tools-mode", "", @@ -1277,11 +1284,6 @@ func renderDaemonHeader(w io.Writer, st daemon.StatusResponse) { return fmt.Sprintf("docs=%d ", sb.DocCount) } switch { - case sb.DiskPath != "": - t.AppendRow(table.Row{"search", fmt.Sprintf( - "%s %sheap=%s disk=%s path=%s", - sb.Name, formatSearchDocs(sb), formatBytes(sb.Bytes), - formatBytes(sb.DiskBytes), sb.DiskPath)}) case sb.DiskResident: // No heap footprint to report — the index lives inside the // graph store's own file, not a separate in-memory @@ -1443,18 +1445,6 @@ func renderDaemonRepos(w io.Writer, st daemon.StatusResponse) { return rows[i].Memory.TotalBytes > rows[j].Memory.TotalBytes }) - // The disk_b column only appears when any repo actually has disk - // usage — i.e. Bleve is running in disk mode. Keeping it - // conditional stops the default in-memory output from carrying a - // dead column users would (rightly) ask about. - showDisk := false - for _, r := range rows { - if r.Memory.DiskBytes > 0 { - showDisk = true - break - } - } - fmt.Fprintln(w, "\ntracked repos:") t := table.NewWriter() t.SetOutputMirror(w) @@ -1500,10 +1490,6 @@ func renderDaemonRepos(w io.Writer, st daemon.StatusResponse) { for i := 0; i < 8; i++ { colConfigs = append(colConfigs, table.ColumnConfig{Number: len(colConfigs) + 1, Align: text.AlignRight}) } - if showDisk { - header = append(header, "disk_b") - colConfigs = append(colConfigs, table.ColumnConfig{Number: len(colConfigs) + 1, Align: text.AlignRight}) - } header = append(header, "path") colConfigs = append(colConfigs, table.ColumnConfig{Number: len(colConfigs) + 1, Align: text.AlignLeft}) t.AppendHeader(header) @@ -1533,9 +1519,6 @@ func renderDaemonRepos(w io.Writer, st daemon.StatusResponse) { formatBytes(r.Memory.SearchBytes), formatBytes(r.Memory.VectorsBytes), ) - if showDisk { - row = append(row, formatBytes(r.Memory.DiskBytes)) - } row = append(row, r.Path) t.AppendRow(row) } @@ -1550,9 +1533,6 @@ func renderDaemonRepos(w io.Writer, st daemon.StatusResponse) { footer = append(footer, "") } footer = append(footer, formatBytes(other), "", "", "", "", "", "", "") - if showDisk { - footer = append(footer, "") - } footer = append(footer, "embedder + runtime + caches (not attributed)") t.AppendFooter(footer) } @@ -1770,21 +1750,6 @@ func daemonControlClient() (*daemon.Client, error) { return c, nil } -// resolveDaemonBufferPoolMB returns the effective buffer-pool cap. -// Precedence: --backend-buffer-pool-mb flag > GORTEX_DAEMON_BUFFER_POOL_MB env > 0 -// (which Open then maps to DefaultBufferPoolMB inside the store). -func resolveDaemonBufferPoolMB() uint64 { - if daemonBackendBufferPoolMB != 0 { - return daemonBackendBufferPoolMB - } - if env := strings.TrimSpace(os.Getenv("GORTEX_DAEMON_BUFFER_POOL_MB")); env != "" { - if v, err := strconv.ParseUint(env, 10, 64); err == nil { - return v - } - } - return 0 -} - // killByPID is the fallback stop path for stale daemons that have a PID // file but don't respond on the socket. Asks the process to terminate, // waits, then force-kills. Silently returns nil if the PID no longer diff --git a/cmd/gortex/daemon_controller.go b/cmd/gortex/daemon_controller.go index 0f4d034f6..3cd7c02a2 100644 --- a/cmd/gortex/daemon_controller.go +++ b/cmd/gortex/daemon_controller.go @@ -619,16 +619,15 @@ type searchBackendInfo struct { // resolveSearchBackend inspects the live search backend and produces // the stats needed by status rendering: which backend is active, total -// document count, its heap footprint, and (for disk-backed Bleve) the -// on-disk size. +// document count, and its heap footprint. // // Real-world unwrap order: Swappable → HybridBackend → (text, vector). -// The text side is itself a concrete BM25/Bleve/SymbolSearcherBackend. -// Both layers have to be peeled; if we stop early we fall into the -// default branch and the status reports "unknown" — which was the bug -// users saw. When the store implements graph.SymbolSearcher, the -// indexer wires up a *search.SymbolSearcherBackend instead of building -// an in-process BM25/Bleve index at all (see initialSearchBackend in +// The text side is itself a concrete BM25/SymbolSearcherBackend. Both +// layers have to be peeled; if we stop early we fall into the default +// branch and the status reports "unknown" — which was the bug users +// saw. When the store implements graph.SymbolSearcher, the indexer +// wires up a *search.SymbolSearcherBackend instead of building an +// in-process BM25 index at all (see initialSearchBackend in // internal/indexer/indexer.go) — that case has to be matched // explicitly too, or it falls into the same "unknown" default. func resolveSearchBackend(b search.Backend) searchBackendInfo { @@ -637,35 +636,29 @@ func resolveSearchBackend(b search.Backend) searchBackendInfo { return out } - // 1) Unwrap Swappable so we see the currently-active inner. + // 1) Pin Swappable so the inspected backend cannot be retired while + // status derives its type, counts, and sizes. inner := b if sw, ok := inner.(*search.Swappable); ok { - inner = sw.Inner() + var release func() + inner, release = sw.AcquireBackend() + defer release() } // 2) If Hybrid is in play, split its text/vector sizes and keep // drilling into the text side for name/doc-count identification. if hyb, ok := inner.(*search.HybridBackend); ok { out.vectorBytes = hyb.VectorSizeBytes() inner = hyb.TextBackend() - // TextBackend() itself could be a Swappable in some setups — - // unlikely today but cheap to guard. + // TextBackend() itself could be a Swappable in some setups. Pin it + // too so a nested replacement cannot invalidate this inspection. if sw, ok := inner.(*search.Swappable); ok { - inner = sw.Inner() + var release func() + inner, release = sw.AcquireBackend() + defer release() } } switch back := inner.(type) { - case *search.BleveBackend: - if path := back.DiskPath(); path != "" { - out.Name = "bleve-disk" - out.DiskPath = path - out.DiskBytes = back.DiskBytes() - } else { - out.Name = "bleve-memory" - } - out.DocCount = back.Count() - out.DocCountKnown = true - out.Bytes = back.SizeBytes() case *search.BM25Backend: out.Name = "bm25" out.DocCount = back.Count() @@ -688,9 +681,9 @@ func resolveSearchBackend(b search.Backend) searchBackendInfo { out.DocCount, out.DocCountKnown = back.DocCount() default: out.Name = "unknown" - out.DocCount = b.Count() + out.DocCount = inner.Count() out.DocCountKnown = true - out.Bytes = search.BackendSize(b) + out.Bytes = search.BackendSize(inner) } return out } @@ -985,7 +978,6 @@ func (c *realController) Status(ctx context.Context) (daemon.StatusResponse, err share := float64(nodes) / float64(totalNodes) mem.SearchBytes = uint64(float64(backendStats.Bytes) * share) mem.VectorsBytes = uint64(float64(backendStats.vectorBytes) * share) - mem.DiskBytes = uint64(float64(backendStats.DiskBytes) * share) } mem.TotalBytes = mem.NodesBytes + mem.EdgesBytes + mem.SearchBytes + mem.VectorsBytes @@ -1076,10 +1068,11 @@ func (c *realController) Status(ctx context.Context) (daemon.StatusResponse, err // of Status. resp := daemon.StatusResponse{ - TrackedRepos: tracked, - MemoryBytes: mem.Alloc, - SearchBackend: searchBackendForResponse, - TrigramCache: trigramCacheForResponse(), + TrackedRepos: tracked, + MemoryBytes: mem.Alloc, + SearchBackend: searchBackendForResponse, + TrigramCache: trigramCacheForResponse(), + GraphIntegrity: daemon.GraphIntegrityStatusFor(g), Runtime: daemon.RuntimeStats{ Alloc: mem.Alloc, Sys: mem.Sys, diff --git a/cmd/gortex/daemon_enrichment_progress_test.go b/cmd/gortex/daemon_enrichment_progress_test.go index 9db8f1e15..a1a998539 100644 --- a/cmd/gortex/daemon_enrichment_progress_test.go +++ b/cmd/gortex/daemon_enrichment_progress_test.go @@ -108,9 +108,9 @@ func TestStatusResponse_EnrichmentJSONRoundTrip(t *testing.T) { assert.NotContains(t, string(rawEmpty), `"enrichment"`, "Enrichment must be omitted (omitempty) when nil") } -// TestSearchBackendStats_DiskResidentJSONRoundTrip locks in the new +// TestSearchBackendStats_DiskResidentJSONRoundTrip locks in that the // DiskResident flag survives the wire round trip and is omitted when -// false (the existing bleve/bm25 backends never set it). +// false (the in-process BM25 backend never sets it). func TestSearchBackendStats_DiskResidentJSONRoundTrip(t *testing.T) { sb := daemon.SearchBackendStats{Name: "sqlite-fts5", DocCount: 48572, DiskResident: true} raw, err := json.Marshal(sb) diff --git a/cmd/gortex/daemon_health_integrity_test.go b/cmd/gortex/daemon_health_integrity_test.go new file mode 100644 index 000000000..cd9f9a691 --- /dev/null +++ b/cmd/gortex/daemon_health_integrity_test.go @@ -0,0 +1,62 @@ +package main + +import ( + "encoding/json" + "strings" + "sync/atomic" + "testing" + "time" + + "github.com/zzet/gortex/internal/graph" +) + +type healthIntegritySnapshotStore struct { + graph.Store + calls atomic.Int32 + snapshot graph.StructuralIntegritySnapshot +} + +func (s *healthIntegritySnapshotStore) StructuralIntegritySnapshot(graph.StructuralIntegritySnapshotOptions) graph.StructuralIntegritySnapshot { + s.calls.Add(1) + return s.snapshot +} + +func TestBuildDaemonHealthSnapshotUsesQueryFreeIntegrityCapability(t *testing.T) { + store := &healthIntegritySnapshotStore{ + Store: graph.New(), + snapshot: graph.StructuralIntegritySnapshot{ + Totals: graph.StructuralIntegrityTotals{WriteRejected: 4, ReadSuppressed: 2}, + }, + } + out := buildDaemonHealthSnapshot(time.Now(), nil, &daemonState{graph: store}, nil) + if store.calls.Load() != 1 { + t.Fatalf("health must read exactly one in-memory snapshot, got %d calls", store.calls.Load()) + } + integrity, ok := out["graph_integrity"] + if !ok || integrity == nil { + t.Fatalf("health omitted graph_integrity: %+v", out) + } + encoded, err := json.Marshal(out) + if err != nil { + t.Fatal(err) + } + text := string(encoded) + for _, want := range []string{`"graph_integrity"`, `"status":"warn"`, `"degraded":true`, `"write_rejected":4`, `"read_suppressed":2`} { + if !strings.Contains(text, want) { + t.Fatalf("health JSON missing %s: %s", want, text) + } + } + for _, forbidden := range []string{`"samples"`, `"attribution"`, `"file_path"`} { + if strings.Contains(text, forbidden) { + t.Fatalf("routine health exposed raw audit detail %s: %s", forbidden, text) + } + } +} + +func TestBuildDaemonHealthSnapshotOmitsZeroIntegrity(t *testing.T) { + store := &healthIntegritySnapshotStore{Store: graph.New()} + out := buildDaemonHealthSnapshot(time.Now(), nil, &daemonState{graph: store}, nil) + if _, exists := out["graph_integrity"]; exists { + t.Fatalf("zero integrity telemetry must be omitted: %+v", out) + } +} diff --git a/cmd/gortex/daemon_health_snapshot.go b/cmd/gortex/daemon_health_snapshot.go index 5963d5994..e0ef95fb2 100644 --- a/cmd/gortex/daemon_health_snapshot.go +++ b/cmd/gortex/daemon_health_snapshot.go @@ -116,6 +116,9 @@ func buildDaemonHealthSnapshot( } } if state != nil && state.graph != nil { + if integrity := daemon.GraphIntegrityStatusFor(state.graph); integrity != nil { + out["graph_integrity"] = integrity + } if reporter, ok := state.graph.(graph.DBStatReporter); ok { dbBytes, walBytes := reporter.DBStats() if dbBytes > 0 || walBytes > 0 { diff --git a/cmd/gortex/daemon_state.go b/cmd/gortex/daemon_state.go index 0deb2c16f..f812fd710 100644 --- a/cmd/gortex/daemon_state.go +++ b/cmd/gortex/daemon_state.go @@ -73,24 +73,6 @@ type daemonState struct { backendPath string } -// lspDisabledSet builds the set of LSP spec names that should NOT be -// auto-registered by Router.RegisterAvailable. Two inputs are merged: -// -// 1. Per-spec config overrides — any entry in `semantic.providers` -// with `enabled: false` whose name matches a known LSP spec. -// Already-disabled-by-config users keep their opt-out without -// having to also set the env var. -// 2. The GORTEX_LSP_DISABLE env var — comma-separated spec names. -// The literal value "all" or "*" disables auto-registration -// entirely (the explicit-config loop above still runs). -// -// The special key "__all__" in the returned map signals -// "skip auto-register everywhere" and is checked separately by -// callers; per-spec keys carry the spec.Name. -func lspDisabledSet(providers []config.SemanticProviderConfig, envVar string) map[string]bool { - return serverstack.LspDisabledSet(providers, envVar) -} - // buildDaemonState builds the daemon's stack through the shared // serverstack constructor and returns the long-lived daemonState the // warmup loop and controller share. @@ -106,14 +88,13 @@ func buildDaemonState(logger *zap.Logger) (*daemonState, error) { applyToolPresetFlags(cfg, daemonTools, daemonToolsMode) ss, err := serverstack.NewSharedServer(serverstack.SharedServerConfig{ - Lifecycle: serverstack.LifecycleDaemon, - Backend: daemonBackend, - BackendPath: daemonBackendPath, - BufferPoolMB: resolveDaemonBufferPoolMB(), - Config: cfg, - Global: gc, - Logger: logger, - Version: version, + Lifecycle: serverstack.LifecycleDaemon, + Backend: daemonBackend, + BackendPath: daemonBackendPath, + Config: cfg, + Global: gc, + Logger: logger, + Version: version, Embedder: serverstack.EmbedderRequest{ FlagChanged: daemonEmbeddingsChanged, FlagEnabled: daemonEmbeddings, diff --git a/cmd/gortex/daemon_state_test.go b/cmd/gortex/daemon_state_test.go index 048285620..2549bdc02 100644 --- a/cmd/gortex/daemon_state_test.go +++ b/cmd/gortex/daemon_state_test.go @@ -8,6 +8,7 @@ import ( "github.com/zzet/gortex/internal/config" "github.com/zzet/gortex/internal/intern" + "github.com/zzet/gortex/internal/serverstack" ) // TestLSPDisabledSet_ConfigOnly — a `semantic.providers` entry with @@ -16,7 +17,7 @@ import ( // `enabled: false` for a custom non-registry daemon doesn't shadow // a same-named LSP). func TestLSPDisabledSet_ConfigOnly(t *testing.T) { - got := lspDisabledSet([]config.SemanticProviderConfig{ + got := serverstack.LspDisabledSet([]config.SemanticProviderConfig{ {Name: "gopls", Enabled: false}, {Name: "tsserver", Enabled: true}, // explicitly enabled — must NOT land in disabled {Name: "not-a-real-lsp", Enabled: false}, @@ -30,7 +31,7 @@ func TestLSPDisabledSet_ConfigOnly(t *testing.T) { // TestLSPDisabledSet_EnvOnly — comma-separated names land in the // disabled set. Whitespace is trimmed; empty entries are skipped. func TestLSPDisabledSet_EnvOnly(t *testing.T) { - got := lspDisabledSet(nil, "gopls, tsserver,, ,pyright") + got := serverstack.LspDisabledSet(nil, "gopls, tsserver,, ,pyright") want := map[string]bool{"gopls": true, "tsserver": true, "pyright": true} if !reflect.DeepEqual(got, want) { t.Fatalf("got %v, want %v", got, want) @@ -42,7 +43,7 @@ func TestLSPDisabledSet_EnvOnly(t *testing.T) { // auto-registration entirely. func TestLSPDisabledSet_EnvAllKillSwitch(t *testing.T) { for _, env := range []string{"all", "ALL", "*", " all "} { - got := lspDisabledSet(nil, env) + got := serverstack.LspDisabledSet(nil, env) if !got["__all__"] { t.Fatalf("env=%q: expected __all__ kill switch, got %v", env, got) } @@ -52,7 +53,7 @@ func TestLSPDisabledSet_EnvAllKillSwitch(t *testing.T) { // TestLSPDisabledSet_ConfigAndEnvMerge — disables from both sources // merge cleanly into one map. func TestLSPDisabledSet_ConfigAndEnvMerge(t *testing.T) { - got := lspDisabledSet([]config.SemanticProviderConfig{ + got := serverstack.LspDisabledSet([]config.SemanticProviderConfig{ {Name: "gopls", Enabled: false}, }, "tsserver,pyright") want := map[string]bool{ @@ -68,7 +69,7 @@ func TestLSPDisabledSet_ConfigAndEnvMerge(t *testing.T) { // TestLSPDisabledSet_Empty — no providers, empty env yields an empty // map (not nil — callers index into it). func TestLSPDisabledSet_Empty(t *testing.T) { - got := lspDisabledSet(nil, "") + got := serverstack.LspDisabledSet(nil, "") if got == nil { t.Fatal("expected non-nil empty map") } diff --git a/cmd/gortex/daemon_status_render_test.go b/cmd/gortex/daemon_status_render_test.go index dfb5107da..b437e6257 100644 --- a/cmd/gortex/daemon_status_render_test.go +++ b/cmd/gortex/daemon_status_render_test.go @@ -79,39 +79,41 @@ func TestRenderDaemonRepos_NoRepos(t *testing.T) { assert.Contains(t, buf.String(), "tracked repos: (none)") } -func TestRenderDaemonRepos_DiskColumnAppearsWhenDiskMode(t *testing.T) { +func TestRenderDaemonHeader_SearchBackendRow(t *testing.T) { st := sampleStatus() - // Flip the biggest repo into disk mode. - st.TrackedRepos[0].Memory.DiskBytes = 500_000_000 + // What resolveSearchBackend emits for the store-native FTS index: + // no heap figure of its own, so the row says disk-resident instead + // of printing a fabricated "heap=0 B". + st.SearchBackend = daemon.SearchBackendStats{ + Name: "sqlite-fts5", + DocCount: 65000, + DocCountKnown: true, + DiskResident: true, + } var buf bytes.Buffer - renderDaemonRepos(&buf, st) + renderDaemonHeader(&buf, st) out := buf.String() - assert.Contains(t, out, "disk_b", "disk_b column must appear when any repo has DiskBytes > 0") -} - -func TestRenderDaemonRepos_NoDiskColumnInMemoryMode(t *testing.T) { - var buf bytes.Buffer - renderDaemonRepos(&buf, sampleStatus()) - assert.NotContains(t, buf.String(), "disk_b", - "disk_b column should be hidden when all repos are in-memory") + assert.Contains(t, out, "sqlite-fts5") + assert.Contains(t, out, "65000") + assert.Contains(t, out, "disk-resident") + assert.NotContains(t, out, "heap=") } -func TestRenderDaemonHeader_SearchBackendRow(t *testing.T) { +func TestRenderDaemonHeader_SearchBackendRow_HeapBackend(t *testing.T) { st := sampleStatus() + // The in-process BM25 index does have a heap footprint to report. st.SearchBackend = daemon.SearchBackendStats{ - Name: "bleve-disk", - DocCount: 65000, + Name: "bm25", + DocCount: 12000, DocCountKnown: true, - Bytes: 200 * 1024 * 1024, - DiskPath: "/tmp/gortex/bleve.scorch", - DiskBytes: 800 * 1024 * 1024, + Bytes: 200 * 1024 * 1024, } var buf bytes.Buffer renderDaemonHeader(&buf, st) out := buf.String() - assert.Contains(t, out, "bleve-disk") - assert.Contains(t, out, "65000") - assert.Contains(t, out, "/tmp/gortex/bleve.scorch") + assert.Contains(t, out, "bm25") + assert.Contains(t, out, "12000") + assert.Contains(t, out, "heap=") } func TestRenderDaemonHeader_WarmupLabel(t *testing.T) { diff --git a/cmd/gortex/eval_embedders.go b/cmd/gortex/eval_embedders.go index 599246693..a2a70ae74 100644 --- a/cmd/gortex/eval_embedders.go +++ b/cmd/gortex/eval_embedders.go @@ -280,7 +280,9 @@ func benchVariant(name string, probeTexts []string, fixture recall.Fixture, cfg // Pull the hybrid backend's vector side and run semantic-only. inner := idx.Search() if sw, ok := inner.(*search.Swappable); ok { - inner = sw.Inner() + var release func() + inner, release = sw.AcquireBackend() + defer release() } hybrid, _ := inner.(*search.HybridBackend) if hybrid == nil || hybrid.VectorIndex() == nil || hybrid.VectorIndex().Count() == 0 { diff --git a/cmd/gortex/eval_recall.go b/cmd/gortex/eval_recall.go index 8177967a2..414d8cd0f 100644 --- a/cmd/gortex/eval_recall.go +++ b/cmd/gortex/eval_recall.go @@ -156,7 +156,9 @@ func runEvalRecall(_ *cobra.Command, _ []string) error { // HybridBackend is what RRF queries. inner := idx.Search() if sw, ok := inner.(*search.Swappable); ok { - inner = sw.Inner() + var release func() + inner, release = sw.AcquireBackend() + defer release() } hybrid, _ := inner.(*search.HybridBackend) var textBackend search.Backend diff --git a/cmd/gortex/eval_server.go b/cmd/gortex/eval_server.go index 3b97069f5..ed38acfc9 100644 --- a/cmd/gortex/eval_server.go +++ b/cmd/gortex/eval_server.go @@ -6,7 +6,6 @@ import ( "net/http" "os" "os/signal" - "path/filepath" "strconv" "github.com/spf13/cobra" @@ -23,9 +22,8 @@ import ( ) var ( - evalPort int - evalIndex string - evalCacheDir string + evalPort int + evalIndex string ) var evalServerCmd = &cobra.Command{ @@ -45,7 +43,6 @@ func init() { evalServerCmd.Flags().StringVar(&evalBind, "bind", "127.0.0.1", "bind address; a non-loopback bind requires --auth-token") evalServerCmd.Flags().StringVar(&evalAuthToken, "auth-token", "", "bearer token required for every request (fallback: $GORTEX_EVAL_TOKEN)") evalServerCmd.Flags().StringVar(&evalIndex, "index", "", "repository path to index on startup") - evalServerCmd.Flags().StringVar(&evalCacheDir, "cache-dir", "", "index cache directory (default ~/.gortex-eval-cache)") rootCmd.AddCommand(evalServerCmd) } @@ -77,48 +74,15 @@ func runEvalServer(cmd *cobra.Command, args []string) error { srv.SetArtifacts(cfg.Artifacts) srv.SetNamedQueries(cfg.Queries) - // Index the repository if --index is provided, with cache support. + // Index the repository if --index is provided. if evalIndex != "" { - cache, err := eval.NewCache(evalCacheDir, version) + fmt.Fprintf(os.Stderr, "[gortex] eval-server: indexing %s...\n", evalIndex) + result, err := idx.Index(evalIndex) if err != nil { - return fmt.Errorf("creating index cache: %w", err) - } - - // Derive repo name from directory name and get commit hash. - repoName := filepath.Base(evalIndex) - commitHash := gitCommitHash(evalIndex) - - cached := false - if commitHash != "" { - if cache.Check(repoName, commitHash) && cache.Validate(repoName, commitHash) { - cachePath, err := cache.Load(repoName, commitHash) - if err == nil { - fmt.Fprintf(os.Stderr, "[gortex] eval-server: loaded cached index from %s\n", cachePath) - cached = true - } else { - fmt.Fprintf(os.Stderr, "[gortex] eval-server: cache load failed, will re-index: %v\n", err) - } - } - } - - if !cached { - fmt.Fprintf(os.Stderr, "[gortex] eval-server: indexing %s...\n", evalIndex) - result, err := idx.Index(evalIndex) - if err != nil { - return fmt.Errorf("indexing %s: %w", evalIndex, err) - } - fmt.Fprintf(os.Stderr, "[gortex] eval-server: indexed %d files (%d nodes, %d edges) in %dms\n", - result.FileCount, result.NodeCount, result.EdgeCount, result.DurationMs) - - // Store to cache for future runs. - if commitHash != "" { - if err := cache.Store(repoName, commitHash, evalIndex); err != nil { - fmt.Fprintf(os.Stderr, "[gortex] eval-server: cache store warning: %v\n", err) - } else { - fmt.Fprintf(os.Stderr, "[gortex] eval-server: cached index for %s@%s\n", repoName, commitHash[:8]) - } - } + return fmt.Errorf("indexing %s: %w", evalIndex, err) } + fmt.Fprintf(os.Stderr, "[gortex] eval-server: indexed %d files (%d nodes, %d edges) in %dms\n", + result.FileCount, result.NodeCount, result.EdgeCount, result.DurationMs) } // Run analysis (communities, processes) after indexing. @@ -170,5 +134,3 @@ func runEvalServer(cmd *cobra.Command, args []string) error { return httpServer.Close() } } - -// gitCommitHash is defined in git.go diff --git a/docs/04-evaluation/task-set.md b/docs/04-evaluation/task-set.md index bb7b091bc..b966a449f 100644 --- a/docs/04-evaluation/task-set.md +++ b/docs/04-evaluation/task-set.md @@ -33,8 +33,9 @@ order of operations." `parser.ExtractionResult`; `Indexer.processExtraction` writes them into the `graph.Graph` and accumulates incoming-edge tracking for the next phase. -4. `Indexer.buildSearchIndex` (BM25 / Bleve) + `idx.embedder` - (if set) populate the search backends. +4. `Indexer.buildSearchIndex` (in-process BM25 in tests and evals, + store-native FTS in production) + `idx.embedder` (if set) + populate the search backends. 5. Semantic enrichment (`internal/semantic`) runs LSP / SCIP providers in parallel; resolved edges get `Origin=lsp_resolved` for tier filtering. diff --git a/docs/multi-repo.md b/docs/multi-repo.md index c4dbefaf9..7975eb4aa 100644 --- a/docs/multi-repo.md +++ b/docs/multi-repo.md @@ -249,4 +249,3 @@ Scoped tool responses carry a `scope_applied` meta field plus a one-line widen h - **Cross-repo edges** — the resolver links symbols across repo boundaries with same-repo preference. Cross-repo edges carry a `cross_repo: true` flag. - **Impact analysis** — `explain_change_impact`, `verify_change`, and `get_test_targets` follow cross-repo edges automatically, grouping results by repository. - **Shared repos** — the same repo can appear in multiple projects with different reference tags. It's indexed once and shared across projects. -- **Auto-detection** — set `workspace.auto_detect: true` in `.gortex.yaml` to auto-discover Git repos in a parent directory. diff --git a/eval/environments/gortex_docker.py b/eval/environments/gortex_docker.py index 3241419f7..a5545770c 100644 --- a/eval/environments/gortex_docker.py +++ b/eval/environments/gortex_docker.py @@ -5,12 +5,11 @@ 2. Copy gortex binary and tool bridge scripts (native/native_augment modes) 3. Start eval-server inside container, health-check with configurable timeout 4. Extract patch (git diff) before teardown - 5. Mount/copy cached indexes when available - 6. Record setup failures gracefully — never raise, return failure result + 5. Record setup failures gracefully — never raise, return failure result Architecture: Agent bash cmd → /usr/local/bin/gortex-search → curl localhost:4747/tool/search_symbols - → eval-server → in-memory graph + → eval-server → graph store Fallback: → gortex CLI (cold path) """ @@ -29,7 +28,6 @@ DEFAULT_EVAL_SERVER_PORT = 4747 DEFAULT_GORTEX_TIMEOUT = 120 -DEFAULT_CACHE_DIR = Path.home() / ".gortex-eval-cache" HEALTH_CHECK_INTERVAL = 2.0 CONTAINER_WORKDIR = "/testbed" GORTEX_BINARY_CONTAINER_PATH = "/usr/local/bin/gortex" @@ -46,12 +44,6 @@ ] -def _make_cache_key(repo_name: str, commit_hash: str) -> str: - """Build a deterministic cache directory name from repo and commit.""" - safe_repo = repo_name.replace("/", "__") - return f"{safe_repo}_{commit_hash}" - - class GortexDockerEnvironment: """Docker environment managing the full container lifecycle for Gortex eval. @@ -71,7 +63,6 @@ def __init__( gortex_binary: str | Path | None = None, gortex_timeout: int = DEFAULT_GORTEX_TIMEOUT, eval_server_port: int = DEFAULT_EVAL_SERVER_PORT, - cache_dir: str | Path | None = None, instance_id: str = "", ) -> None: self.image = image @@ -80,7 +71,6 @@ def __init__( self.gortex_binary = Path(gortex_binary) if gortex_binary else None self.gortex_timeout = gortex_timeout self.eval_server_port = eval_server_port - self.cache_dir = Path(cache_dir) if cache_dir else DEFAULT_CACHE_DIR self.instance_id = instance_id self._client: docker.DockerClient | None = None @@ -111,7 +101,6 @@ def setup(self) -> dict[str, Any] | None: start = time.time() self._copy_gortex_binary() self._copy_bridge_scripts() - self._restore_or_skip_cache() self._start_eval_server() self._wait_for_health() self.index_time = time.time() - start @@ -263,45 +252,12 @@ def _copy_bridge_scripts(self) -> None: copied += 1 logger.info("Copied %d bridge scripts into container", copied) - def _restore_or_skip_cache(self) -> None: - """Mount/copy a cached index into the container if one exists.""" - repo_name, commit_hash = self._get_repo_identity() - cache_key = _make_cache_key(repo_name, commit_hash) - cache_path = self.cache_dir / cache_key - - if not cache_path.is_dir(): - logger.info("No cached index for %s, eval-server will index fresh", cache_key) - return - - tarball = cache_path / "index.tar.gz" - if not tarball.is_file(): - logger.info("Cache dir exists but no tarball for %s, skipping", cache_key) - return - - logger.info("Restoring cached index %s into container", cache_key) - try: - cache_dest = "/root/.gortex-cache" - self._container.exec_run(["mkdir", "-p", cache_dest]) - with open(tarball, "rb") as f: - self._container.put_archive(cache_dest, f.read()) - logger.info("Cached index restored to %s", cache_dest) - except Exception as exc: - logger.warning("Cache restore failed, will index fresh: %s", exc) - def _start_eval_server(self) -> None: """Start ``gortex eval-server`` as a background process in the container.""" - cache_flag = "" - cache_dest = "/root/.gortex-cache" - # Check if cache was restored - exit_code, _ = self._container.exec_run(["test", "-d", cache_dest]) - if exit_code == 0: - cache_flag = f"--cache-dir {cache_dest}" - cmd = ( f"nohup /usr/local/bin/gortex eval-server " f"--port {self.eval_server_port} " f"--index {self.repo_path} " - f"{cache_flag} " f"> /tmp/gortex-eval-server.log 2>&1 &" ) logger.info("Starting eval-server on port %d", self.eval_server_port) @@ -349,20 +305,6 @@ def _wait_for_health(self) -> None: f"for instance {self.instance_id}. Server log tail:\n{log_tail}" ) - def _get_repo_identity(self) -> tuple[str, str]: - """Extract (repo_name, commit_hash) from the container's /testbed repo.""" - _, repo_out = self._container.exec_run( - ["bash", "-c", "basename $(git remote get-url origin 2>/dev/null || basename $(pwd)) .git"], - workdir=self.repo_path, - ) - _, commit_out = self._container.exec_run( - ["bash", "-c", "git rev-parse HEAD 2>/dev/null || echo unknown"], - workdir=self.repo_path, - ) - repo_name = repo_out.decode("utf-8", errors="replace").strip() or "unknown" - commit_hash = commit_out.decode("utf-8", errors="replace").strip() or "unknown" - return repo_name, commit_hash - def _put_file_in_container( self, local_path: Path, diff --git a/eval/prompts/system_native.jinja b/eval/prompts/system_native.jinja index 053c60913..0c4f969e4 100644 --- a/eval/prompts/system_native.jinja +++ b/eval/prompts/system_native.jinja @@ -16,7 +16,7 @@ Failure to follow these rules will cause your response to be rejected. ## Code Intelligence -You have **Gortex** — a code intelligence engine that indexes this codebase into an in-memory knowledge graph. It knows every function call chain, class hierarchy, execution flow, and symbol relationship. These are fast bash commands (~100ms). Use them when useful, skip them when a simple grep suffices. +You have **Gortex** — a code intelligence engine that indexes this codebase into a knowledge graph. It knows every function call chain, class hierarchy, execution flow, and symbol relationship. These are fast bash commands (~100ms). Use them when useful, skip them when a simple grep suffices. ### Gortex Commands diff --git a/eval/prompts/system_native_augment.jinja b/eval/prompts/system_native_augment.jinja index ef56ab8b1..bfb284c51 100644 --- a/eval/prompts/system_native_augment.jinja +++ b/eval/prompts/system_native_augment.jinja @@ -16,7 +16,7 @@ Failure to follow these rules will cause your response to be rejected. ## Code Intelligence -You have **Gortex** — a code intelligence engine that indexes this codebase into an in-memory knowledge graph. It knows every function call chain, class hierarchy, execution flow, and symbol relationship. These are fast bash commands (~100ms). Use them when useful, skip them when a simple grep suffices. +You have **Gortex** — a code intelligence engine that indexes this codebase into a knowledge graph. It knows every function call chain, class hierarchy, execution flow, and symbol relationship. These are fast bash commands (~100ms). Use them when useful, skip them when a simple grep suffices. Your `grep` and `rg` results are also automatically enriched with `[Gortex]` annotations showing callers, callees, and execution flows for matched symbols. Pay attention to these — they often point you to the right code without extra tool calls. diff --git a/eval/run_eval.py b/eval/run_eval.py index f75fe11ee..8bb3663a5 100644 --- a/eval/run_eval.py +++ b/eval/run_eval.py @@ -250,7 +250,6 @@ def process_instance( gortex_binary=env_cfg.get("gortex_binary"), gortex_timeout=int(env_cfg.get("gortex_timeout", 120)), eval_server_port=int(env_cfg.get("eval_server_port", 4747)), - cache_dir=env_cfg.get("cache_dir"), instance_id=instance_id, ) env.setup() diff --git a/eval/tests/test_docker_integration.py b/eval/tests/test_docker_integration.py index 7c481780e..40a4768a6 100644 --- a/eval/tests/test_docker_integration.py +++ b/eval/tests/test_docker_integration.py @@ -21,7 +21,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) -from environments.gortex_docker import GortexDockerEnvironment, _make_cache_key +from environments.gortex_docker import GortexDockerEnvironment # --- Docker availability check --- @@ -152,15 +152,3 @@ def test_setup_failure_records_error(self, mock_docker) -> None: assert result is not None assert result["exit_status"] == "setup_failure" assert "fail-test" in result["instance_id"] - - def test_cache_key_determinism(self) -> None: - """Cache key for same inputs is always the same.""" - k1 = _make_cache_key("repo", "abc123") - k2 = _make_cache_key("repo", "abc123") - assert k1 == k2 - - def test_cache_key_uniqueness(self) -> None: - """Different inputs produce different cache keys.""" - k1 = _make_cache_key("repo_a", "commit1") - k2 = _make_cache_key("repo_b", "commit2") - assert k1 != k2 diff --git a/eval/tests/test_gortex_docker.py b/eval/tests/test_gortex_docker.py index 6cb5e8f9c..932d6085b 100644 --- a/eval/tests/test_gortex_docker.py +++ b/eval/tests/test_gortex_docker.py @@ -1,7 +1,7 @@ """Unit tests for eval/environments/gortex_docker.py. -Tests focus on pure logic (cache key, failure recording, properties) -and mock Docker interactions to avoid requiring a running Docker daemon. +Tests focus on pure logic (failure recording, properties) and mock +Docker interactions to avoid requiring a running Docker daemon. """ from __future__ import annotations @@ -20,30 +20,9 @@ DEFAULT_EVAL_SERVER_PORT, DEFAULT_GORTEX_TIMEOUT, GortexDockerEnvironment, - _make_cache_key, ) -# -- _make_cache_key --------------------------------------------------------- - -class TestMakeCacheKey: - def test_basic(self): - assert _make_cache_key("django", "abc123") == "django_abc123" - - def test_slash_in_repo_name(self): - assert _make_cache_key("django/django", "abc123") == "django__django_abc123" - - def test_deterministic(self): - k1 = _make_cache_key("repo", "commit") - k2 = _make_cache_key("repo", "commit") - assert k1 == k2 - - def test_different_inputs_different_keys(self): - k1 = _make_cache_key("repo_a", "commit1") - k2 = _make_cache_key("repo_b", "commit2") - assert k1 != k2 - - # -- GortexDockerEnvironment init ------------------------------------------- class TestInit: @@ -66,13 +45,11 @@ def test_custom_params(self): gortex_binary="/tmp/gortex", gortex_timeout=60, eval_server_port=9999, - cache_dir="/tmp/cache", instance_id="django__django-1234", ) assert env.gortex_binary == Path("/tmp/gortex") assert env.gortex_timeout == 60 assert env.eval_server_port == 9999 - assert env.cache_dir == Path("/tmp/cache") assert env.instance_id == "django__django-1234" diff --git a/go.mod b/go.mod index 9226a0a14..504b5c75d 100644 --- a/go.mod +++ b/go.mod @@ -215,7 +215,6 @@ require ( github.com/alexaandru/go-sitter-forest/zig v1.9.4 github.com/alexaandru/go-sitter-forest/ziggy v1.9.1 github.com/alexaandru/go-sitter-forest/ziggy_schema v1.9.1 - github.com/blevesearch/bleve/v2 v2.6.0 github.com/blevesearch/go-porterstemmer v1.0.3 github.com/charmbracelet/bubbles v1.0.0 github.com/charmbracelet/bubbletea v1.3.10 @@ -291,27 +290,8 @@ require ( ) require ( - github.com/RoaringBitmap/roaring/v2 v2.24.0 // indirect github.com/atotto/clipboard v0.1.4 // indirect github.com/aymanbagabas/go-osc52/v2 v2.0.1 // indirect - github.com/bits-and-blooms/bitset v1.25.0 // indirect - github.com/blevesearch/bleve_index_api v1.3.12 // indirect - github.com/blevesearch/geo v0.2.5 // indirect - github.com/blevesearch/go-faiss v1.1.5 // indirect - github.com/blevesearch/gtreap v0.1.1 // indirect - github.com/blevesearch/mmap-go v1.2.0 // indirect - github.com/blevesearch/scorch_segment_api/v2 v2.4.7 // indirect - github.com/blevesearch/segment v0.9.1 // indirect - github.com/blevesearch/snowballstem v0.9.0 // indirect - github.com/blevesearch/upsidedown_store_api v1.0.2 // indirect - github.com/blevesearch/vellum v1.2.0 // indirect - github.com/blevesearch/zapx/v11 v11.4.3 // indirect - github.com/blevesearch/zapx/v12 v12.4.3 // indirect - github.com/blevesearch/zapx/v13 v13.4.3 // indirect - github.com/blevesearch/zapx/v14 v14.4.3 // indirect - github.com/blevesearch/zapx/v15 v15.4.3 // indirect - github.com/blevesearch/zapx/v16 v16.3.4 // indirect - github.com/blevesearch/zapx/v17 v17.1.9 // indirect github.com/charmbracelet/colorprofile v0.4.3 // indirect github.com/charmbracelet/x/cellbuf v0.0.15 // indirect github.com/charmbracelet/x/term v0.2.2 // indirect @@ -326,7 +306,6 @@ require ( github.com/go-errors/errors v1.5.1 // indirect github.com/go-logr/logr v1.4.4 // indirect github.com/go-viper/mapstructure/v2 v2.5.0 // indirect - github.com/golang/snappy v1.0.0 // indirect github.com/gomlx/compute v0.1.2 // indirect github.com/gomlx/exceptions v0.0.3 // indirect github.com/gomlx/go-xla v0.4.1 // indirect @@ -338,7 +317,6 @@ require ( github.com/inconshreveable/mousetrap v1.1.0 // indirect github.com/jackc/pgpassfile v1.0.0 // indirect github.com/jackc/pgservicefile v0.0.0-20240606120523-5a60cdf6a761 // indirect - github.com/json-iterator/go v1.1.12 // indirect github.com/klauspost/cpuid/v2 v2.4.0 // indirect github.com/knights-analytics/ortgenai v0.3.2 // indirect github.com/lucasb-eyer/go-colorful v1.4.1 // indirect @@ -346,9 +324,6 @@ require ( github.com/mattn/go-localereader v0.0.1 // indirect github.com/mattn/go-pointer v0.0.1 // indirect github.com/mattn/go-runewidth v0.0.27 // indirect - github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd // indirect - github.com/modern-go/reflect2 v1.0.2 // indirect - github.com/mschoch/smat v0.2.0 // indirect github.com/muesli/ansi v0.0.0-20230316100256-276c6243b2f6 // indirect github.com/muesli/cancelreader v0.2.2 // indirect github.com/ncruces/go-strftime v1.0.0 // indirect @@ -367,7 +342,6 @@ require ( github.com/xo/terminfo v1.0.0 // indirect github.com/yosida95/uritemplate/v3 v3.0.2 // indirect github.com/zeebo/assert v1.3.0 // indirect - go.etcd.io/bbolt v1.5.0 // indirect go.uber.org/multierr v1.11.0 // indirect go.yaml.in/yaml/v3 v3.0.5 // indirect golang.org/x/crypto v0.55.0 // indirect diff --git a/go.sum b/go.sum index 4bb9c8fdd..c57fded8b 100644 --- a/go.sum +++ b/go.sum @@ -6,8 +6,6 @@ codeberg.org/go-pdf/fpdf v0.10.0 h1:u+w669foDDx5Ds43mpiiayp40Ov6sZalgcPMDBcZRd4= codeberg.org/go-pdf/fpdf v0.10.0/go.mod h1:Y0DGRAdZ0OmnZPvjbMp/1bYxmIPxm0ws4tfoPOc4LjU= git.sr.ht/~sbinet/gg v0.6.0 h1:RIzgkizAk+9r7uPzf/VfbJHBMKUr0F5hRFxTUGMnt38= git.sr.ht/~sbinet/gg v0.6.0/go.mod h1:uucygbfC9wVPQIfrmwM2et0imr8L7KQWywX0xpFMm94= -github.com/RoaringBitmap/roaring/v2 v2.24.0 h1:zQkkBZtG3WRP4j+P3A5DO221SvL1Br88TJkhyqEQRZo= -github.com/RoaringBitmap/roaring/v2 v2.24.0/go.mod h1:SfT3of9nYh3vis1dIbCj4Yw6KQGujTN+f345nrN/0JA= github.com/ajstarks/svgo v0.0.0-20211024235047-1546f124cd8b h1:slYM766cy2nI3BwyRiyQj/Ud48djTMtMebDqepE95rw= github.com/ajstarks/svgo v0.0.0-20211024235047-1546f124cd8b/go.mod h1:1KcenG0jGWcpt8ov532z81sp/kMMUG485J2InIOyADM= github.com/alexaandru/go-sitter-forest/ada v1.9.0 h1:hV0rMiYCssJD6rRTya4HD1w9LnvgJUoq2QAJAQM7kzs= @@ -440,48 +438,8 @@ github.com/aymanbagabas/go-osc52/v2 v2.0.1 h1:HwpRHbFMcZLEVr42D4p7XBqjyuxQH5SMiE github.com/aymanbagabas/go-osc52/v2 v2.0.1/go.mod h1:uYgXzlJ7ZpABp8OJ+exZzJJhRNQ2ASbcXHWsFqH8hp8= github.com/aymanbagabas/go-udiff v0.3.1 h1:LV+qyBQ2pqe0u42ZsUEtPiCaUoqgA9gYRDs3vj1nolY= github.com/aymanbagabas/go-udiff v0.3.1/go.mod h1:G0fsKmG+P6ylD0r6N/KgQD/nWzgfnl8ZBcNLgcbrw8E= -github.com/bits-and-blooms/bitset v1.24.6 h1:qcrftZUVBIwfs+m+nhoCBAPT+ZPZZjti8SbHbDQQkZ4= -github.com/bits-and-blooms/bitset v1.24.6/go.mod h1:7hO7Gc7Pp1vODcmWvKMRA9BNmbv6a/7QIWpPxHddWR8= -github.com/bits-and-blooms/bitset v1.25.0 h1:0Ro0qF4abCkM6SqWPVj29sFhAbMPAZpaDD7xhJ10beM= -github.com/bits-and-blooms/bitset v1.25.0/go.mod h1:7hO7Gc7Pp1vODcmWvKMRA9BNmbv6a/7QIWpPxHddWR8= -github.com/blevesearch/bleve/v2 v2.6.0 h1:Cyd3dd4q5tCbOV8MnKUVRUDYMHOir9xn12NZzXVSEd4= -github.com/blevesearch/bleve/v2 v2.6.0/go.mod h1:gLmI8lWgHgrIYf7UpUX7JISI1CaqC6VScu46mHThuAY= -github.com/blevesearch/bleve_index_api v1.3.12 h1:MirVNltwGq8z0PhOgiQp+bKL5qq8OvCxEwOOC7NnHNE= -github.com/blevesearch/bleve_index_api v1.3.12/go.mod h1:xvd48t5XMeeioWQ5/jZvgLrV98flT2rdvEJ3l/ki4Ko= -github.com/blevesearch/geo v0.2.5 h1:yJg9FX1oRwLnjXSXF+ECHfXFTF4diF02Ca/qUGVjJhE= -github.com/blevesearch/geo v0.2.5/go.mod h1:Jhq7WE2K6mJTx1xS44M2pUO6Io+wjCSHh1+co3YOgH4= -github.com/blevesearch/go-faiss v1.1.5 h1:/IU5lkOahH9Ghfk9n3F6N0XD7PYVXZJWmNDc9TtXuco= -github.com/blevesearch/go-faiss v1.1.5/go.mod h1:w3W9AiWsFRGVaMG+/cmJi7iHEAuGyC6blsgO1EzCK/M= github.com/blevesearch/go-porterstemmer v1.0.3 h1:GtmsqID0aZdCSNiY8SkuPJ12pD4jI+DdXTAn4YRcHCo= github.com/blevesearch/go-porterstemmer v1.0.3/go.mod h1:angGc5Ht+k2xhJdZi511LtmxuEf0OVpvUUNrwmM1P7M= -github.com/blevesearch/gtreap v0.1.1 h1:2JWigFrzDMR+42WGIN/V2p0cUvn4UP3C4Q5nmaZGW8Y= -github.com/blevesearch/gtreap v0.1.1/go.mod h1:QaQyDRAT51sotthUWAH4Sj08awFSSWzgYICSZ3w0tYk= -github.com/blevesearch/mmap-go v1.2.0 h1:l33nNKPFcBjJUMwem6sAYJPUzhUCABoK9FxZDGiFNBI= -github.com/blevesearch/mmap-go v1.2.0/go.mod h1:Vd6+20GBhEdwJnU1Xohgt88XCD/CTWcqbCNxkZpyBo0= -github.com/blevesearch/scorch_segment_api/v2 v2.4.7 h1:GlMzW08hcsM3DnLUxhyF/1PcDal1qtvvIuytuph5djw= -github.com/blevesearch/scorch_segment_api/v2 v2.4.7/go.mod h1://IJ7tG3QCf0cWW/aVSXqy77tc1AvLu3fcJLYEvOAFs= -github.com/blevesearch/segment v0.9.1 h1:+dThDy+Lvgj5JMxhmOVlgFfkUtZV2kw49xax4+jTfSU= -github.com/blevesearch/segment v0.9.1/go.mod h1:zN21iLm7+GnBHWTao9I+Au/7MBiL8pPFtJBJTsk6kQw= -github.com/blevesearch/snowballstem v0.9.0 h1:lMQ189YspGP6sXvZQ4WZ+MLawfV8wOmPoD/iWeNXm8s= -github.com/blevesearch/snowballstem v0.9.0/go.mod h1:PivSj3JMc8WuaFkTSRDW2SlrulNWPl4ABg1tC/hlgLs= -github.com/blevesearch/upsidedown_store_api v1.0.2 h1:U53Q6YoWEARVLd1OYNc9kvhBMGZzVrdmaozG2MfoB+A= -github.com/blevesearch/upsidedown_store_api v1.0.2/go.mod h1:M01mh3Gpfy56Ps/UXHjEO/knbqyQ1Oamg8If49gRwrQ= -github.com/blevesearch/vellum v1.2.0 h1:xkDiOEsHc2t3Cp0NsNZZ36pvc130sCzcGKOPMzXe+e0= -github.com/blevesearch/vellum v1.2.0/go.mod h1:uEcfBJz7mAOf0Kvq6qoEKQQkLODBF46SINYNkZNae4k= -github.com/blevesearch/zapx/v11 v11.4.3 h1:PTZOO5loKpHC/x/GzmPZNa9cw7GZIQxd5qRjwij9tHY= -github.com/blevesearch/zapx/v11 v11.4.3/go.mod h1:4gdeyy9oGa/lLa6D34R9daXNUvfMPZqUYjPwiLmekwc= -github.com/blevesearch/zapx/v12 v12.4.3 h1:eElXvAaAX4m04t//CGBQAtHNPA+Q6A1hHZVrN3LSFYo= -github.com/blevesearch/zapx/v12 v12.4.3/go.mod h1:TdFmr7afSz1hFh/SIBCCZvcLfzYvievIH6aEISCte58= -github.com/blevesearch/zapx/v13 v13.4.3 h1:qsdhRhaSpVnqDFlRiH9vG5+KJ+dE7KAW9WyZz/KXAiE= -github.com/blevesearch/zapx/v13 v13.4.3/go.mod h1:knK8z2NdQHlb5ot/uj8wuvOq5PhDGjNYQQy0QDnopZk= -github.com/blevesearch/zapx/v14 v14.4.3 h1:GY4Hecx0C6UTmiNC2pKdeA2rOKiLR5/rwpU9WR51dgM= -github.com/blevesearch/zapx/v14 v14.4.3/go.mod h1:rz0XNb/OZSMjNorufDGSpFpjoFKhXmppH9Hi7a877D8= -github.com/blevesearch/zapx/v15 v15.4.3 h1:iJiMJOHrz216jyO6lS0m9RTCEkprUnzvqAI2lc/0/CU= -github.com/blevesearch/zapx/v15 v15.4.3/go.mod h1:1pssev/59FsuWcgSnTa0OeEpOzmhtmr/0/11H0Z8+Nw= -github.com/blevesearch/zapx/v16 v16.3.4 h1:hDAqA8qusZTNbPEL7//w5P65UZ2de6yhSeUaTbp0Po0= -github.com/blevesearch/zapx/v16 v16.3.4/go.mod h1:zqkPPqs9GS9FzVWzCO3Wf1X044yWAV17+4zb+FTiEHg= -github.com/blevesearch/zapx/v17 v17.1.9 h1:K5MsArRyuwfylDTN1+cU7plGVIFz4gPoH4HD5M7I8ik= -github.com/blevesearch/zapx/v17 v17.1.9/go.mod h1:34TIaJmdo5hMh2IBLoE4Day65j7DJ++8s5trz1yrsGY= github.com/campoy/embedmd v1.0.0 h1:V4kI2qTJJLf4J29RzI/MAt2c3Bl4dQSYPuflzwFH2hY= github.com/campoy/embedmd v1.0.0/go.mod h1:oxyr9RCiSXg0M3VJ3ks0UGfp98BpSSGr0kpiX3MzVl8= github.com/charmbracelet/bubbles v1.0.0 h1:12J8/ak/uCZEMQ6KU7pcfwceyjLlWsDLAxB5fXonfvc= @@ -514,7 +472,6 @@ github.com/cpuguy83/go-md2man/v2 v2.0.6/go.mod h1:oOW0eioCTA6cOiMLiUPZOpcVxMig6N github.com/daulet/tokenizers v1.27.0 h1:MmFYAEDFz69s/nNQfHg59DWqHz3v94m99kEZ/JbL+s4= github.com/daulet/tokenizers v1.27.0/go.mod h1:YjFY1o1HGMyWkQgbXJDghhvke/yFDp2vGdIO2hYs4MQ= github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= -github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc h1:U9qPSI2PIWSS1VwoXQT9A3Wy9MM3WgvqSxFWenqJduM= github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= github.com/dlclark/regexp2 v1.12.0 h1:0j4c5qQmnC6XOWNjP3PIXURXN2gWx76rd3KvgdPkCz8= @@ -541,8 +498,6 @@ github.com/gofrs/flock v0.13.0 h1:95JolYOvGMqeH31+FC7D2+uULf6mG61mEZ/A8dRYMzw= github.com/gofrs/flock v0.13.0/go.mod h1:jxeyy9R1auM5S6JYDBhDt+E2TCo7DkratH4Pgi8P+Z0= github.com/golang/freetype v0.0.0-20170609003504-e2365dfdc4a0 h1:DACJavvAHhabrF08vX0COfcOBJRhZ8lUbR+ZWIs0Y5g= github.com/golang/freetype v0.0.0-20170609003504-e2365dfdc4a0/go.mod h1:E/TSTwGwJL78qG/PmXZO1EjYhfJinVAhrmmHX6Z8B9k= -github.com/golang/snappy v1.0.0 h1:Oy607GVXHs7RtbggtPBnr2RmDArIsAefDwvrdWvRhGs= -github.com/golang/snappy v1.0.0/go.mod h1:/XxbfmMg8lxefKM7IXC3fBNl/7bRcc72aCRzEWrmP2Q= github.com/gomlx/compute v0.1.2 h1:YPH+lLIYk6qDm4JjbMNL8oz1I4jeeeUqqrO7fURd0/U= github.com/gomlx/compute v0.1.2/go.mod h1:MDTT683Wvq7IMSKXDGLvV3Co7HDpx/Xsgp2RbF5lhNM= github.com/gomlx/compute-onnx v0.0.0-20260730095030-43334863f719 h1:ulZbGdTaaa7V0GEfQAHEAfCPLd5KWHEO489zAIzRoDw= @@ -564,7 +519,6 @@ github.com/google/go-github/v88 v88.0.0 h1:dZA9IKkPK1eXZj4ypngnpRj5FwdpTv4whix2P github.com/google/go-github/v88 v88.0.0/go.mod h1:rufTDgn2N45wjhukLTyxmvc9nilSp3mr3Rgtt6b1MPw= github.com/google/go-querystring v1.2.0 h1:yhqkPbu2/OH+V9BfpCVPZkNmUXhb2gBxJArfhIxNtP0= github.com/google/go-querystring v1.2.0/go.mod h1:8IFJqpSRITyJ8QhQ13bmbeMBDfmeEJZD5A0egEOmkqU= -github.com/google/gofuzz v1.0.0/go.mod h1:dBl0BpW6vV/+mYPU4Po3pmUjxk6FQPldtuIdl/M65Eg= github.com/google/jsonschema-go v0.4.3 h1:/DBOLZTfDow7pe2GmaJNhltueGTtDKICi8V8p+DQPd0= github.com/google/jsonschema-go v0.4.3/go.mod h1:r5quNTdLOYEz95Ru18zA0ydNbBuYoo9tgaYcxEYhJVE= github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3 h1:LMLX+LgTNWpfvCBdFebv6EsYotImrt/Ppc5cXIriCSo= @@ -603,8 +557,6 @@ github.com/janpfeifer/go-benchmarks v0.1.1 h1:gLLy07/JrOKSnMWeUxSnjTdhkglgmrNR2I github.com/janpfeifer/go-benchmarks v0.1.1/go.mod h1:5AagXCOUzevvmYFQalcgoa4oWPyH1IkZNckolGWfiSM= github.com/jedib0t/go-pretty/v6 v6.8.3 h1:yVSk5aemoYHCvcrtqyXklwqcgHQIQzmy/oUzFlmffSQ= github.com/jedib0t/go-pretty/v6 v6.8.3/go.mod h1:YwC5CE4fJ1HFUDeivSV1r//AmANFHyqczZk+U6BDALU= -github.com/json-iterator/go v1.1.12 h1:PV8peI4a0ysnczrg+LtxykD8LfKY9ML6u2jnxaEnrnM= -github.com/json-iterator/go v1.1.12/go.mod h1:e30LSqwooZae/UwlEbR2852Gd8hjQvJoHmT4TnhNGBo= github.com/klauspost/cpuid/v2 v2.4.0 h1:S6Hrbc7+ywsr0r+RLapfGBHfyefhCTwEh3A0tV913Dw= github.com/klauspost/cpuid/v2 v2.4.0/go.mod h1:19jmZ9mjzoF//ddRSUsv0zfBTJWh3QJh9FNxZTMrGxU= github.com/knights-analytics/hugot v0.7.7 h1:+gVc2p8Q2VLwgE1XVZcBidpYj4pNnTjUiVE5w7RIUSU= @@ -633,13 +585,6 @@ github.com/mattn/go-runewidth v0.0.27 h1:Feg/Oou5zI/wnpgDF6omIU0OokC9GxLC/WRknhV github.com/mattn/go-runewidth v0.0.27/go.mod h1:3qAiGCV4Koz/yuveO58qUefmUTRm8r0IGEXZ9jeHp/8= github.com/mitchellh/colorstring v0.0.0-20190213212951-d06e56a500db h1:62I3jR2EmQ4l5rM/4FEfDWcRD+abF5XlKShorW5LRoQ= github.com/mitchellh/colorstring v0.0.0-20190213212951-d06e56a500db/go.mod h1:l0dey0ia/Uv7NcFFVbCLtqEBQbrT4OCwCSKTEv6enCw= -github.com/modern-go/concurrent v0.0.0-20180228061459-e0a39a4cb421/go.mod h1:6dJC0mAP4ikYIbvyc7fijjWJddQyLn8Ig3JB5CqoB9Q= -github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd h1:TRLaZ9cD/w8PVh93nsPXa1VrQ6jlwL5oN8l14QlcNfg= -github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd/go.mod h1:6dJC0mAP4ikYIbvyc7fijjWJddQyLn8Ig3JB5CqoB9Q= -github.com/modern-go/reflect2 v1.0.2 h1:xBagoLtFs94CBntxluKeaWgTMpvLxC4ur3nMaC9Gz0M= -github.com/modern-go/reflect2 v1.0.2/go.mod h1:yWuevngMOJpCy52FWWMvUC8ws7m/LJsjYzDa0/r8luk= -github.com/mschoch/smat v0.2.0 h1:8imxQsjDm8yFEAVBe7azKmKSgzSkZXDuKkSq9374khM= -github.com/mschoch/smat v0.2.0/go.mod h1:kc9mz7DoBKqDyiRL7VZN8KvXQMWeTaVnttLRXOlotKw= github.com/muesli/ansi v0.0.0-20230316100256-276c6243b2f6 h1:ZK8zHtRHOkbHy6Mmr5D264iyp3TiX5OmNcI5cIARiQI= github.com/muesli/ansi v0.0.0-20230316100256-276c6243b2f6/go.mod h1:CJlz5H+gyd6CUWT45Oy4q24RdLyn7Md9Vj2/ldJBSIo= github.com/muesli/cancelreader v0.2.2 h1:3I4Kt4BQjOR54NavqnDogx/MIoWBFa0StPA8ELUXHmA= @@ -773,8 +718,6 @@ github.com/zeebo/blake3 v0.2.4 h1:KYQPkhpRtcqh0ssGYcKLG1JYvddkEA8QwCM/yBqhaZI= github.com/zeebo/blake3 v0.2.4/go.mod h1:7eeQ6d2iXWRGF6npfaxl2CU+xy2Fjo2gxeyZGCRUjcE= github.com/zeebo/pcg v1.0.1 h1:lyqfGeWiv4ahac6ttHs+I5hwtH/+1mrhlCtVNQM2kHo= github.com/zeebo/pcg v1.0.1/go.mod h1:09F0S9iiKrwn9rlI5yjLkmrug154/YRW6KnnXVDM/l4= -go.etcd.io/bbolt v1.5.0 h1:S7GAl7Fxv12yohbwFfIbQCGDWbQbtDGPET4P/bD4lxU= -go.etcd.io/bbolt v1.5.0/go.mod h1:mkltfYE5aUHQxUct9N9V+Kp7aSjFqjgrhcXIS70Lrdk= go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto= go.uber.org/goleak v1.3.0/go.mod h1:CoHD4mav9JJNrW/WLlf7HGZPjdw8EucARQHekz1X6bE= go.uber.org/multierr v1.11.0 h1:blXXJkSxSSfBVBlC76pxqeO+LN3aDfLQo+309xJstO0= diff --git a/internal/agents/agentstest/harness.go b/internal/agents/agentstest/harness.go index 4a221cd69..a674bc0db 100644 --- a/internal/agents/agentstest/harness.go +++ b/internal/agents/agentstest/harness.go @@ -19,7 +19,6 @@ import ( "encoding/json" "os" "path/filepath" - "sort" "testing" "github.com/zzet/gortex/internal/agents" @@ -168,15 +167,3 @@ func WriteYAML(t *testing.T, path string, obj map[string]any) { t.Fatalf("write: %v", err) } } - -// SortedFilePaths returns the Paths from a result's Files in sorted -// order. Makes golden assertions invariant to adapter iteration -// order over maps. -func SortedFilePaths(res *agents.Result) []string { - out := make([]string, 0, len(res.Files)) - for _, f := range res.Files { - out = append(out, f.Path) - } - sort.Strings(out) - return out -} diff --git a/internal/agents/aider/adapter.go b/internal/agents/aider/adapter.go index 509cdb3d7..e670dae98 100644 --- a/internal/agents/aider/adapter.go +++ b/internal/agents/aider/adapter.go @@ -37,7 +37,7 @@ func (a *Adapter) DocsURL() string { return DocsURL } // aiderIgnoreLines is the set of cache paths Aider should never // ingest as source. Keeping them out of the chat avoids wasting -// tokens on Gortex's own binary index and Bleve scorer data. +// tokens on Gortex's own graph store and cache artifacts. var aiderIgnoreLines = []string{ "# Added by `gortex init` — Gortex cache artifacts are not source", ".gortex/", diff --git a/internal/agents/registry.go b/internal/agents/registry.go index e84255c1e..e2504a1e7 100644 --- a/internal/agents/registry.go +++ b/internal/agents/registry.go @@ -49,27 +49,6 @@ func (r *Registry) All() []Adapter { return out } -// Names returns the sorted list of registered adapter names. Used by -// the --agents flag's help text and the unknown-name error message. -func (r *Registry) Names() []string { - r.mu.RLock() - defer r.mu.RUnlock() - out := make([]string, 0, len(r.byName)) - for name := range r.byName { - out = append(out, name) - } - sort.Strings(out) - return out -} - -// Lookup returns the adapter with the given name, or nil when no -// adapter is registered under that name. -func (r *Registry) Lookup(name string) Adapter { - r.mu.RLock() - defer r.mu.RUnlock() - return r.byName[name] -} - // Filter returns the subset of adapters selected by allow/skip lists. // allowCSV: // - "" (empty) or "auto" selects every registered adapter (the diff --git a/internal/analysis/rule_family.go b/internal/analysis/rule_family.go index 946d9bd5b..8739fc7ca 100644 --- a/internal/analysis/rule_family.go +++ b/internal/analysis/rule_family.go @@ -18,20 +18,6 @@ type RuleFamily interface { Evaluate(g graph.Store, changedSet []string) []GuardViolation } -// GuardsFamily adapts the flat guards: list (co-change / boundary rules). -type GuardsFamily struct { - Rules []config.GuardRule -} - -func (f GuardsFamily) Name() string { return "guards" } - -func (f GuardsFamily) Evaluate(g graph.Store, changedSet []string) []GuardViolation { - if len(f.Rules) == 0 { - return nil - } - return EvaluateGuards(g, f.Rules, changedSet) -} - // ArchitectureFamily adapts the declarative architecture: layer DSL. type ArchitectureFamily struct { Config config.ArchitectureConfig diff --git a/internal/analyzer/temporal_verify.go b/internal/analyzer/temporal_verify.go index 1f0016888..f03b61c94 100644 --- a/internal/analyzer/temporal_verify.go +++ b/internal/analyzer/temporal_verify.go @@ -2,14 +2,13 @@ package analyzer // LLM-backed adapter for the Temporal dispatch verification pass. // -// PURPOSE — wire the deterministic verification core in -// internal/resolver/temporal_verify.go to (a) a real LLM provider, (b) on-disk -// source grounding, (c) a reproducibility cache, and (d) the canonical -// map[string]any output shape. Keeps the resolver core free of any LLM / I/O -// dependency; all the "actions" live here. -// RATIONALE — the verifier and source provider are injected interfaces, so this -// file holds the only LLM + filesystem coupling. The cache makes re-runs cheap -// and deterministic (same code + model → cached verdict). +// PURPOSE — adapt the deterministic verification core in +// internal/resolver/temporal_verify.go to a real LLM provider, a reproducibility +// cache, and the canonical map[string]any output shape. Source grounding remains +// an injected resolver.TemporalSourceProvider supplied by the host that owns +// repository path resolution. +// RATIONALE — the cache keeps re-runs cheap and deterministic (same code + model +// → cached verdict) without coupling the resolver core to LLM or filesystem I/O. // KEYWORDS — temporal, verify, llm, source, cache, adapter import ( @@ -22,9 +21,7 @@ import ( "path/filepath" "strings" - "github.com/zzet/gortex/internal/graph" "github.com/zzet/gortex/internal/llm" - "github.com/zzet/gortex/internal/llm/provider" "github.com/zzet/gortex/internal/resolver" ) @@ -61,94 +58,6 @@ func VerifyReportToMap(rep resolver.TemporalVerifyReport) map[string]any { } } -// --- File-backed source provider ------------------------------------------ - -// maxNodeSourceBytes caps the per-node source handed to the LLM so a giant -// function body can't blow the prompt budget. -const maxNodeSourceBytes = 6000 - -// FileSourceProvider reads a graph node's source from disk, slicing the file by -// the node's [StartLine, EndLine]. Files are cached in-memory for the run. -type FileSourceProvider struct { - root string - cache map[string]string -} - -// NewFileSourceProvider returns a source provider rooted at the indexed repo. -func NewFileSourceProvider(root string) *FileSourceProvider { - return &FileSourceProvider{root: root, cache: map[string]string{}} -} - -// NodeSource returns the source text of n's declaration, or ("", false). -func (p *FileSourceProvider) NodeSource(n *graph.Node) (string, bool) { - if n == nil || n.FilePath == "" { - return "", false - } - body, ok := p.fileBody(n.FilePath) - if !ok { - return "", false - } - lines := strings.Split(body, "\n") - start, end := n.StartLine, n.EndLine - if start < 1 { - start = 1 - } - if start > len(lines) { - return "", false - } - if end < start || end > len(lines) { - end = len(lines) - } - src := strings.Join(lines[start-1:end], "\n") - if len(src) > maxNodeSourceBytes { - src = src[:maxNodeSourceBytes] + "\n// …truncated" - } - return src, true -} - -func (p *FileSourceProvider) fileBody(rel string) (string, bool) { - if b, ok := p.cache[rel]; ok { - return b, b != "" - } - abs, ok := p.resolveWithinRoot(rel) - if !ok { - p.cache[rel] = "" - return "", false - } - raw, err := os.ReadFile(abs) - if err != nil { - p.cache[rel] = "" - return "", false - } - p.cache[rel] = string(raw) - return string(raw), true -} - -// resolveWithinRoot resolves rel — relative to p.root, or absolute — and -// confirms the result stays inside p.root. A node FilePath that escapes the -// indexed tree (via "..", or an absolute path elsewhere) is refused, so a -// crafted graph node can't make the verifier read arbitrary files off disk and -// ship them to the LLM. An empty root refuses everything (nothing to bound to). -func (p *FileSourceProvider) resolveWithinRoot(rel string) (string, bool) { - if p.root == "" || rel == "" { - return "", false - } - rootAbs, err := filepath.Abs(p.root) - if err != nil { - return "", false - } - cand := rel - if !filepath.IsAbs(cand) { - cand = filepath.Join(rootAbs, cand) - } - cand = filepath.Clean(cand) - relToRoot, err := filepath.Rel(rootAbs, cand) - if err != nil || relToRoot == ".." || strings.HasPrefix(relToRoot, ".."+string(filepath.Separator)) { - return "", false - } - return cand, true -} - // --- LLM verifier ---------------------------------------------------------- const temporalVerifySystemPrompt = `You verify guesses made by a static analyzer for the Temporal workflow engine. @@ -164,26 +73,11 @@ type llmTemporalVerifier struct { p llm.Provider } -// NewLLMTemporalVerifier builds a verifier from resolved LLM config. Returns a -// close func and (nil, nil, err) when no provider can be constructed — the -// caller treats that as "LLM unavailable" and skips verification. -func NewLLMTemporalVerifier(cfg llm.Config) (resolver.TemporalVerifier, func() error, error) { - cfg = cfg.ApplyDefaults() - if !cfg.IsEnabled() { - return nil, nil, fmt.Errorf("llm provider not enabled (set llm.provider)") - } - p, err := provider.New(cfg) - if err != nil { - return nil, nil, err - } - return &llmTemporalVerifier{p: p}, p.Close, nil -} - // NewLLMTemporalVerifierFromProvider wraps an already-constructed provider. // // PURPOSE — let a host that already owns a live llm.Provider (e.g. the MCP // server's shared LLM service) reuse it for temporal verification instead of -// spinning up a second provider from raw config via NewLLMTemporalVerifier. +// spinning up a second provider from raw config. // RATIONALE — the verifier is a thin Verify(req)→verdict adapter over // Provider.Complete; binding it to an existing provider avoids duplicate model // loads / API clients and respects the caller's provider lifecycle (no Close diff --git a/internal/analyzer/temporal_verify_test.go b/internal/analyzer/temporal_verify_test.go index 45330f750..6e80d033f 100644 --- a/internal/analyzer/temporal_verify_test.go +++ b/internal/analyzer/temporal_verify_test.go @@ -2,14 +2,12 @@ package analyzer import ( "context" - "os" "path/filepath" "testing" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" - "github.com/zzet/gortex/internal/graph" "github.com/zzet/gortex/internal/resolver" ) @@ -50,27 +48,6 @@ func TestParseTemporalVerdict(t *testing.T) { } } -func TestFileSourceProvider_SlicesByLine(t *testing.T) { - dir := t.TempDir() - rel := "pkg/a.go" - require.NoError(t, os.MkdirAll(filepath.Join(dir, "pkg"), 0o755)) - require.NoError(t, os.WriteFile(filepath.Join(dir, rel), - []byte("package pkg\n\nfunc A() {}\nfunc B() {\n\treturn\n}\n"), 0o644)) - - p := NewFileSourceProvider(dir) - // Node B spans lines 4..6. - n := &graph.Node{FilePath: rel, StartLine: 4, EndLine: 6} - src, ok := p.NodeSource(n) - require.True(t, ok) - assert.Contains(t, src, "func B() {") - assert.Contains(t, src, "return") - assert.NotContains(t, src, "func A()") - - // Missing file → not ok. - _, ok = p.NodeSource(&graph.Node{FilePath: "nope.go", StartLine: 1, EndLine: 1}) - assert.False(t, ok) -} - type countingVerifier struct { calls int res resolver.TemporalVerifyResult @@ -109,26 +86,3 @@ func TestCachingVerifier_HitsCacheAndPersists(t *testing.T) { assert.Equal(t, resolver.TemporalVerdictConfirmed, r3.Verdict, "loaded from disk, not the new delegate") assert.Equal(t, 0, inner2.calls) } - -func TestFileSourceProvider_RefusesPathEscape(t *testing.T) { - dir := t.TempDir() - // A secret file OUTSIDE the indexed root. - outside := filepath.Join(filepath.Dir(dir), "secret.txt") - require.NoError(t, os.WriteFile(outside, []byte("TOPSECRET"), 0o644)) - t.Cleanup(func() { _ = os.Remove(outside) }) - - p := NewFileSourceProvider(dir) - - // Relative traversal via "..". - _, ok := p.NodeSource(&graph.Node{FilePath: "../secret.txt", StartLine: 1, EndLine: 1}) - assert.False(t, ok, "must refuse a ../ escape out of the indexed root") - - // Absolute path outside the root. - _, ok = p.NodeSource(&graph.Node{FilePath: outside, StartLine: 1, EndLine: 1}) - assert.False(t, ok, "must refuse an absolute path outside the indexed root") - - // A legit relative path inside the root still resolves. - require.NoError(t, os.WriteFile(filepath.Join(dir, "in.go"), []byte("package p"), 0o644)) - _, ok = p.NodeSource(&graph.Node{FilePath: "in.go", StartLine: 1, EndLine: 1}) - assert.True(t, ok, "a path inside the root must still resolve") -} diff --git a/internal/config/config.go b/internal/config/config.go index 095d0f049..59436fb6e 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -162,20 +162,6 @@ func (a ArchitectureConfig) IsEmpty() bool { return len(a.Layers) == 0 && len(a.Rules) == 0 } -// MultiRepoConfig holds workspace-discovery settings used by the -// multi-repo bootstrapper. Carries the (formerly `workspace.auto_detect`) -// flag — moved out from under `workspace:` because that key is now -// reclaimed for the workspace-identity slug. -type MultiRepoConfig struct { - // AutoDetect — when true, tracking a parent directory walks its - // immediate subdirectories looking for `.git/`, treating each - // match as a tracked repo. The legacy YAML key - // `workspace.auto_detect: true` is still accepted by the custom - // Config unmarshaller for one release; the canonical key going - // forward is `multi.auto_detect`. - AutoDetect bool `mapstructure:"auto_detect" yaml:"auto_detect,omitempty"` -} - // ProjectGlob declares a project's path-globs inside a monorepo. // // projects: @@ -264,15 +250,14 @@ type SemanticConfig struct { SkipEmbed []SkipEmbedRule `mapstructure:"skip_embed" yaml:"skip_embed,omitempty"` // SkipSearch lists (language, kind) combinations that should be - // kept in the graph but excluded from the text search index - // (BM25/Bleve). Same shape as SkipEmbed but targets a different - // index. The motivating case: a big monorepo with ~135k JSON - // `variable` nodes (package.json keys, tsconfig entries, etc.) - // pushed total symbol count over search.AutoThreshold and - // triggered an auto-upgrade from BM25 (~900 B/doc) to Bleve - // (~32 KiB/doc). Those config-key nodes aren't useful search - // targets — users who want to find them by name still can via - // graph queries. Defaults are a superset of SkipEmbed because + // kept in the graph but excluded from the text search index. + // Same shape as SkipEmbed but targets a different index. The + // motivating case: a big monorepo with ~135k JSON `variable` + // nodes (package.json keys, tsconfig entries, etc.) inflated the + // symbol corpus without ever being searched by name. Those + // config-key nodes aren't useful search targets — users who want + // to find them by name still can via graph queries. Defaults are + // a superset of SkipEmbed because // anything that isn't worth embedding usually isn't worth // full-text-indexing either. See DefaultSkipSearch. SkipSearch []SkipEmbedRule `mapstructure:"skip_search" yaml:"skip_search,omitempty"` @@ -343,9 +328,9 @@ func DefaultSkipEmbed() []SkipEmbedRule { // DefaultSkipSearch returns the baseline (language, kind) pairs that // are kept out of the text search index. Superset of DefaultSkipEmbed: -// if a node isn't worth a vector slot it generally isn't worth a BM25/ -// Bleve slot either, and on big monorepos these config-key nodes are -// what pushes the backend into its Bleve auto-upgrade (~32 KiB/doc). +// if a node isn't worth a vector slot it generally isn't worth a +// full-text slot either, and on big monorepos these config-key nodes +// dominate the corpus without ever being searched by name. // JSON is the heaviest of the additions — tsconfig / package.json / // lockfile keys alone can account for >100k variable nodes. func DefaultSkipSearch() []SkipEmbedRule { @@ -511,9 +496,8 @@ type Config struct { Artifacts []ArtifactEntry `mapstructure:"artifacts" yaml:"artifacts,omitempty"` // Queries are named, reusable detector bundles runnable via // `analyze kind=named`. - Queries []NamedQuery `mapstructure:"queries" yaml:"queries,omitempty"` - Multi MultiRepoConfig `mapstructure:"multi" yaml:"multi,omitempty"` - Semantic SemanticConfig `mapstructure:"semantic" yaml:"semantic,omitempty"` + Queries []NamedQuery `mapstructure:"queries" yaml:"queries,omitempty"` + Semantic SemanticConfig `mapstructure:"semantic" yaml:"semantic,omitempty"` // LLM configures the LLM service that backs the `ask` MCP tool and // the search-assist passes. Empty by default — daemon skips LLM // wiring entirely when the active provider has no model configured. @@ -605,8 +589,10 @@ type IndexConfig struct { // SkipSearch is the effective text-index skip rules resolved from // Semantic.SkipSearch, same propagation pattern as SkipEmbed. // Users configure this under semantic.skip_search; the indexer - // reads it here. Controls what goes into BM25/Bleve — unlike - // SkipEmbed it doesn't affect the graph or vector index. + // reads it here. Controls what goes into the text search index + // (in-process BM25 in tests and evals, store-native FTS in + // production) — unlike SkipEmbed it doesn't affect the graph or + // vector index. SkipSearch []SkipEmbedRule `mapstructure:"-" yaml:"-"` // IndexProse is the effective prose-indexing toggle resolved from // Search.IndexProse -- same `-` (not on-disk) propagation pattern @@ -1802,9 +1788,6 @@ func Default() *Config { Mode: "defer", }, }, - Multi: MultiRepoConfig{ - AutoDetect: false, - }, Semantic: SemanticConfig{ Enabled: true, TimeoutSeconds: 120, @@ -1818,13 +1801,6 @@ func Default() *Config { // Load reads config from file, environment, and returns a merged Config. // configPath may be empty; in that case only default locations are searched. -// -// Legacy-shape handling: previously the `workspace:` key held a struct -// (`workspace: { auto_detect: true }`). The new schema -// reclaims `workspace:` as a scalar slug. Existing configs are migrated -// in place — `workspace.auto_detect` lifts into `multi.auto_detect`, -// and the loader emits a one-line deprecation note via the returned -// error chain (callers can choose whether to surface or swallow it). func Load(configPath string) (*Config, error) { v := viper.New() v.SetConfigName(".gortex") @@ -1850,13 +1826,6 @@ func Load(configPath string) (*Config, error) { // No config file found — use defaults + env. } - // Migrate legacy `workspace:` mapping shape (held a struct with - // `auto_detect`) into the new `multi:` block so the v.Unmarshal - // below decodes the new schema cleanly. We do the migration on the - // viper key map so env-var overrides and viper's own merge logic - // stay consistent. - migrateLegacyWorkspaceKey(v) - if err := v.Unmarshal(cfg); err != nil { return nil, err } @@ -1872,50 +1841,6 @@ func Load(configPath string) (*Config, error) { return cfg, nil } -// migrateLegacyWorkspaceKey rewrites `workspace.auto_detect` → `multi.auto_detect` -// in the viper key store before unmarshal, so a `.gortex.yaml` written -// against the legacy schema still produces a working Config without the -// caller seeing a parse error. The migration is silent — there's no -// global logger here — but the audit step (`gortex audit_agent_config`, -// reserved for a follow-up) can flag the deprecated key. -// -// Only the documented legacy field is migrated. Any other map under -// `workspace:` is rejected by `validateWorkspaceSchema` so unknown -// shapes don't get silently ignored. -func migrateLegacyWorkspaceKey(v *viper.Viper) { - raw := v.Get("workspace") - if raw == nil { - return - } - switch t := raw.(type) { - case string: - // Already in new shape; nothing to do. - case map[string]interface{}: - if ad, ok := t["auto_detect"]; ok { - // Move to the new home unless `multi.auto_detect` - // is already set explicitly (caller wins). - if v.Get("multi.auto_detect") == nil { - v.Set("multi.auto_detect", ad) - } - } - // The old shape never carried a workspace identity slug, - // so we clear the polymorphic key so v.Unmarshal doesn't - // fail trying to coerce a map into a string. - v.Set("workspace", "") - case map[interface{}]interface{}: - // yaml.v2 / older path — same semantics. - if ad, ok := t["auto_detect"]; ok { - if v.Get("multi.auto_detect") == nil { - v.Set("multi.auto_detect", ad) - } - } - v.Set("workspace", "") - default: - // Unrecognised shape; downstream coercion will surface - // a precise error rather than us silently dropping it. - } -} - // validateWorkspaceSchema enforces the defaults / boundaries that // can't be expressed via struct tags alone: // diff --git a/internal/config/manager.go b/internal/config/manager.go index 3ba7bf3c8..68894778e 100644 --- a/internal/config/manager.go +++ b/internal/config/manager.go @@ -299,7 +299,8 @@ func (cm *ConfigManager) GetRepoConfig(repoPrefix string) *Config { out.Index.SkipEmbed = DefaultSkipEmbed() } // Same plumbing for semantic.skip_search — controls what goes into - // the BM25/Bleve text index. Separate from SkipEmbed so users can + // the text search index (in-process BM25 in tests and evals, + // store-native FTS in production). Separate from SkipEmbed so users can // tune the two filters independently (e.g. a tiny-repo user who // doesn't care about text-index memory can clear SkipSearch while // keeping SkipEmbed's embedding-cost savings). diff --git a/internal/config/temporal_allowlist_test.go b/internal/config/temporal_allowlist_test.go index 46f6bc184..3363d09ab 100644 --- a/internal/config/temporal_allowlist_test.go +++ b/internal/config/temporal_allowlist_test.go @@ -3,55 +3,50 @@ package config import ( "os" "path/filepath" - "sort" + "reflect" "testing" ) -func writeTemporalAllowlist(t *testing.T, repoPath, body string) { +func writeTemporalAllowlist(t *testing.T, root, content string) { t.Helper() - dir := filepath.Join(repoPath, ".gortex") + dir := filepath.Join(root, ".gortex") if err := os.MkdirAll(dir, 0o755); err != nil { - t.Fatalf("mkdir: %v", err) + t.Fatal(err) } - if err := os.WriteFile(filepath.Join(dir, "temporal-allowlist.yaml"), []byte(body), 0o644); err != nil { - t.Fatalf("write: %v", err) + if err := os.WriteFile(filepath.Join(dir, "temporal-allowlist.yaml"), []byte(content), 0o600); err != nil { + t.Fatal(err) } } -func TestLoadLocalTemporalEnvHelpers_GateOff(t *testing.T) { - dir := t.TempDir() - writeTemporalAllowlist(t, dir, "env_helpers:\n - FetchActivityName\n") - t.Setenv(LocalTemporalOptInEnv, "") // not opted in - if got := LoadLocalTemporalEnvHelpers(dir); got != nil { - t.Fatalf("expected nil without opt-in, got %v", got) +func TestLoadLocalTemporalEnvHelpersGateAndNormalization(t *testing.T) { + root := t.TempDir() + writeTemporalAllowlist(t, root, "env_helpers:\n - ' HelperA '\n - ''\n - HelperB\n") + + t.Setenv(LocalTemporalOptInEnv, "") + if got := LoadLocalTemporalEnvHelpers(root); got != nil { + t.Fatalf("gate off returned %#v", got) } -} -func TestLoadLocalTemporalEnvHelpers_GateOnReadsFile(t *testing.T) { - dir := t.TempDir() - writeTemporalAllowlist(t, dir, "env_helpers:\n - FetchActivityName\n - GetActivity\n - \"\"\n") - t.Setenv(LocalTemporalOptInEnv, "1") - got := LoadLocalTemporalEnvHelpers(dir) - sort.Strings(got) - want := []string{"FetchActivityName", "GetActivity"} - if len(got) != len(want) || got[0] != want[0] || got[1] != want[1] { - t.Fatalf("got %v, want %v (blank entries dropped)", got, want) + for _, truthy := range []string{"1", " 1 ", "true", " TrUe "} { + t.Run(truthy, func(t *testing.T) { + t.Setenv(LocalTemporalOptInEnv, truthy) + want := []string{"HelperA", "HelperB"} + if got := LoadLocalTemporalEnvHelpers(root); !reflect.DeepEqual(got, want) { + t.Fatalf("got %#v, want %#v", got, want) + } + }) } } -func TestLoadLocalTemporalEnvHelpers_GateOnNoFile(t *testing.T) { - dir := t.TempDir() +func TestLoadLocalTemporalEnvHelpersFailSoft(t *testing.T) { t.Setenv(LocalTemporalOptInEnv, "true") - if got := LoadLocalTemporalEnvHelpers(dir); got != nil { - t.Fatalf("expected nil when file absent, got %v", got) + if got := LoadLocalTemporalEnvHelpers(t.TempDir()); got != nil { + t.Fatalf("missing file returned %#v", got) } -} -func TestLoadLocalTemporalEnvHelpers_GateOnMalformed(t *testing.T) { - dir := t.TempDir() - writeTemporalAllowlist(t, dir, "env_helpers: : not yaml :::\n") - t.Setenv(LocalTemporalOptInEnv, "1") - if got := LoadLocalTemporalEnvHelpers(dir); got != nil { - t.Fatalf("expected nil on malformed file (fail-soft), got %v", got) + root := t.TempDir() + writeTemporalAllowlist(t, root, "env_helpers: [") + if got := LoadLocalTemporalEnvHelpers(root); got != nil { + t.Fatalf("malformed file returned %#v", got) } } diff --git a/internal/daemon/graph_integrity.go b/internal/daemon/graph_integrity.go new file mode 100644 index 000000000..3a342707d --- /dev/null +++ b/internal/daemon/graph_integrity.go @@ -0,0 +1,41 @@ +package daemon + +import "github.com/zzet/gortex/internal/graph" + +// GraphIntegrityStatus is cheap, since-open telemetry for structural graph +// violations contained by the store backstop. A non-nil value is degraded but +// does not affect daemon readiness because the invalid edge was rejected. +type GraphIntegrityStatus struct { + Status string `json:"status"` + Degraded bool `json:"degraded"` + WriteRejected uint64 `json:"write_rejected,omitempty"` + ReadSuppressed uint64 `json:"read_suppressed,omitempty"` + AttributionOverflow uint64 `json:"attribution_overflow,omitempty"` + SampleOverflow uint64 `json:"sample_overflow,omitempty"` + RepoTotalsOverflow uint64 `json:"repo_totals_overflow,omitempty"` +} + +// GraphIntegrityStatusFor reads only the optional store-scoped in-memory +// snapshot capability. It never scans the graph or durable SQLite rows. +func GraphIntegrityStatusFor(store graph.Store) *GraphIntegrityStatus { + if store == nil { + return nil + } + snapshotter, ok := store.(graph.StructuralIntegritySnapshotter) + if !ok { + return nil + } + snapshot := snapshotter.StructuralIntegritySnapshot(graph.StructuralIntegritySnapshotOptions{}) + if snapshot.Totals.Empty() && snapshot.AttributionOverflow == 0 && snapshot.SampleOverflow == 0 && snapshot.RepoTotalsOverflow == 0 { + return nil + } + return &GraphIntegrityStatus{ + Status: "warn", + Degraded: true, + WriteRejected: snapshot.Totals.WriteRejected, + ReadSuppressed: snapshot.Totals.ReadSuppressed, + AttributionOverflow: snapshot.AttributionOverflow, + SampleOverflow: snapshot.SampleOverflow, + RepoTotalsOverflow: snapshot.RepoTotalsOverflow, + } +} diff --git a/internal/daemon/graph_integrity_test.go b/internal/daemon/graph_integrity_test.go new file mode 100644 index 000000000..59b2eda4b --- /dev/null +++ b/internal/daemon/graph_integrity_test.go @@ -0,0 +1,71 @@ +package daemon + +import ( + "encoding/json" + "strings" + "testing" + + "github.com/zzet/gortex/internal/graph" +) + +type storeWithoutIntegrity struct { + graph.Store +} + +func TestGraphIntegrityStatusOptionalZeroAndWarning(t *testing.T) { + if got := GraphIntegrityStatusFor(nil); got != nil { + t.Fatalf("nil store returned status: %+v", got) + } + if got := GraphIntegrityStatusFor(storeWithoutIntegrity{Store: graph.New()}); got != nil { + t.Fatalf("store without optional capability returned status: %+v", got) + } + + store := graph.New() + if got := GraphIntegrityStatusFor(store); got != nil { + t.Fatalf("zero telemetry must be omitted: %+v", got) + } + store.AddNode(&graph.Node{ID: "source", Kind: graph.KindFunction, RepoPrefix: "repo-a"}) + store.AddEdge(&graph.Edge{From: "source", To: "target#param:x", Kind: graph.EdgeImplements, Origin: "lsp"}) + got := GraphIntegrityStatusFor(store) + if got == nil || got.Status != "warn" || !got.Degraded || got.WriteRejected != 1 || got.ReadSuppressed != 0 { + t.Fatalf("unexpected warning status: %+v", got) + } +} + +func TestStatusResponseGraphIntegrityJSONAndReadinessIndependence(t *testing.T) { + without, err := json.Marshal(StatusResponse{Ready: true}) + if err != nil { + t.Fatal(err) + } + if strings.Contains(string(without), "graph_integrity") { + t.Fatalf("zero integrity status must be omitted: %s", without) + } + + response := StatusResponse{ + Ready: true, + GraphIntegrity: &GraphIntegrityStatus{ + Status: "warn", Degraded: true, WriteRejected: 2, ReadSuppressed: 3, + }, + } + encoded, err := json.Marshal(response) + if err != nil { + t.Fatal(err) + } + var decoded map[string]any + if err := json.Unmarshal(encoded, &decoded); err != nil { + t.Fatal(err) + } + if ready, ok := decoded["ready"].(bool); !ok || !ready { + t.Fatalf("integrity warning changed readiness: %s", encoded) + } + integrity, ok := decoded["graph_integrity"].(map[string]any) + if !ok || integrity["status"] != "warn" || integrity["degraded"] != true { + t.Fatalf("missing integrity warning JSON: %s", encoded) + } + text := string(encoded) + for _, forbidden := range []string{"samples", "attribution", "file_path", "from", "to"} { + if strings.Contains(text, `"`+forbidden+`"`) { + t.Fatalf("routine status exposed audit detail %q: %s", forbidden, encoded) + } + } +} diff --git a/internal/daemon/overlay.go b/internal/daemon/overlay.go index 8445c7054..22776c9dd 100644 --- a/internal/daemon/overlay.go +++ b/internal/daemon/overlay.go @@ -311,17 +311,6 @@ func (m *OverlayManager) Touch(sessionID string) error { return nil } -// IdleTTL returns the configured idle expiry duration. Exposed so the -// overlay_list tool can compute and surface an `expires_at` hint to -// editor extensions that want to schedule a keepalive proactively. -// Zero means "no expiry" (test-mode). -func (m *OverlayManager) IdleTTL() time.Duration { - if m == nil { - return 0 - } - return m.idleTTL -} - // SessionStatus is the per-session liveness snapshot reported through // overlay_list. Callers compare `IdleSeconds` to `IdleTTLSeconds` to // decide when to push a keepalive; `ExpiresAt` (RFC3339) is the @@ -489,18 +478,6 @@ func (m *OverlayManager) Files(sessionID string) (map[string]OverlayFile, error) return out, nil } -// SessionWorkspace returns the workspace slug captured at Register. -// ErrSessionNotFound when the session doesn't exist. -func (m *OverlayManager) SessionWorkspace(sessionID string) (string, error) { - m.mu.RLock() - defer m.mu.RUnlock() - sess, ok := m.sessions[sessionID] - if !ok { - return "", ErrSessionNotFound - } - return sess.WorkspaceID, nil -} - // SweepIdle drops sessions whose LastUsed is older than IdleTTL. // Returns the count of dropped sessions for telemetry. Safe to call // from a single janitor goroutine on a ticker. @@ -966,24 +943,6 @@ func (m *OverlayManager) PushToBranch(sessionID, branchName string, overlay Over return nil } -// DeleteFromBranch removes one overlay file from a specific branch. -// Companion to PushToBranch; does not change the active pointer. -func (m *OverlayManager) DeleteFromBranch(sessionID, branchName, path string) error { - m.mu.Lock() - defer m.mu.Unlock() - sess, ok := m.sessions[sessionID] - if !ok { - return ErrSessionNotFound - } - br, ok := sess.branches[branchName] - if !ok { - return ErrBranchNotFound - } - delete(br.files, path) - sess.LastUsed = time.Now() - return nil -} - // FilesForBranch returns the file map for a specific branch. Used by // compare_branches (which has to read both branches in one shot to // compute the delta). The returned slice never aliases the manager's diff --git a/internal/daemon/proto.go b/internal/daemon/proto.go index 2397f8534..a097fe83b 100644 --- a/internal/daemon/proto.go +++ b/internal/daemon/proto.go @@ -295,6 +295,10 @@ type StatusResponse struct { // the daemon — while SearchBackend above may simultaneously report // the symbol index as disk-resident. Nil when no indexer is wired. TrigramCache *TrigramCacheStats `json:"trigram_cache,omitempty"` + // GraphIntegrity is present only after the store has contained at least + // one structural edge violation. It is warning telemetry and does not + // change Ready because the invalid edge was rejected or suppressed. + GraphIntegrity *GraphIntegrityStatus `json:"graph_integrity,omitempty"` // CountsUnknown marks a response assembled without the aggregate pass — // the per-repo and whole-store counters are zero because they were never // computed, not because the graph is empty. Set when a caller fell back @@ -464,10 +468,11 @@ type TrigramCacheStats struct { // SearchBackendStats identifies which search backend is currently // serving queries, so users can read the `search_b` column in the -// repo breakdown with the right mental model. Bleve with the default -// gtreap KV store costs ~32 KiB per document; BM25 costs ~2 KiB. +// repo breakdown with the right mental model. The in-process BM25 +// index costs ~2 KiB of heap per document; the store-native FTS index +// lives inside the graph store's own file and costs no heap of its own. type SearchBackendStats struct { - Name string `json:"name"` // "bm25" | "bleve-memory" | "bleve-disk" | "sqlite-fts5" + Name string `json:"name"` // "bm25" | "sqlite-fts5" | "unknown" DocCount int `json:"doc_count"` // indexed documents across all repos // DocCountKnown distinguishes "the index holds zero documents" from // "this backend cannot report a document count". Backends whose only @@ -475,9 +480,7 @@ type SearchBackendStats struct { // false so renderers omit the number instead of presenting the delta // as a corpus size. DocCountKnown bool `json:"doc_count_known,omitempty"` - Bytes uint64 `json:"bytes"` // approximate heap footprint - DiskPath string `json:"disk_path,omitempty"` // set only when Name == "bleve-disk" - DiskBytes uint64 `json:"disk_bytes,omitempty"` // current on-disk size for "bleve-disk" + Bytes uint64 `json:"bytes"` // approximate heap footprint // DiskResident marks a backend (e.g. "sqlite-fts5") that has no // meaningful heap footprint of its own — its index lives inside the // graph store's own file — and no cheap byte count is available @@ -703,19 +706,14 @@ type ConfiguredServerStatus struct { // data structures that dominate the daemon's footprint. All values // are approximate — exact accounting would require walking Go's // heap, which is too expensive for a status call. See the individual -// estimators (graph.RepoMemoryEstimate, search.BleveBackend.SizeBytes, +// estimators (graph.RepoMemoryEstimate, search.BackendSize, // search.VectorBackend.SizeBytes) for methodology. type MemoryBreakdown struct { NodesBytes uint64 `json:"nodes_bytes"` EdgesBytes uint64 `json:"edges_bytes"` SearchBytes uint64 `json:"search_bytes"` VectorsBytes uint64 `json:"vectors_bytes"` - // DiskBytes is populated only when the Bleve backend is running in - // disk mode (GORTEX_BLEVE_DISK_DIR set). Each repo gets a - // node-proportional share of the on-disk index size. Zero in - // memory-only mode. - DiskBytes uint64 `json:"disk_bytes,omitempty"` - TotalBytes uint64 `json:"total_bytes"` + TotalBytes uint64 `json:"total_bytes"` } // WriteJSONLine writes v as one JSON object followed by a newline. The diff --git a/internal/daemon/router.go b/internal/daemon/router.go index 55461e1e6..e7a9f5c93 100644 --- a/internal/daemon/router.go +++ b/internal/daemon/router.go @@ -412,18 +412,6 @@ func (r *Router) EffectiveEnabledRemotes(sess *Session) []ServerEntry { return out } -// EncodeJSON is a small helper for callers that want to marshal a -// scope override + tool args into the body bytes RouteToolCall -// expects. Round-trippable on the proxy side because the local -// server's POST /v1/tools/ handler is the same Mux route the -// MCP client originally hit. -func EncodeJSON(v any) ([]byte, error) { - if v == nil { - return []byte("{}"), nil - } - return json.Marshal(v) -} - // LookupForCwd exposes RouteForCwd against the router's own // servers.toml + roster cache + cwd resolver. Callers (notably the // daemon's MCP dispatcher) use this to decide whether a session's diff --git a/internal/daemon/server.go b/internal/daemon/server.go index 39f492785..e3b42d649 100644 --- a/internal/daemon/server.go +++ b/internal/daemon/server.go @@ -940,9 +940,6 @@ func (s *Server) untrackConn(c net.Conn) { // Sessions exposes the registry for inspection (status command, tests). func (s *Server) Sessions() *SessionRegistry { return s.sessions } -// StartedAt returns the time Listen() completed — used for uptime math. -func (s *Server) StartedAt() time.Time { return s.started } - // unmarshalParams decodes RawMessage into a typed struct, treating empty // or null params as an empty struct (zero value) so callers don't need // to special-case missing params. diff --git a/internal/daemon/servers.go b/internal/daemon/servers.go index c85c23e12..5c944afc8 100644 --- a/internal/daemon/servers.go +++ b/internal/daemon/servers.go @@ -631,9 +631,8 @@ func scanWorkspaceField(data []byte) string { // Strip quotes. v = strings.Trim(v, `"' `) if v == "" { - // Could be a struct shape (e.g. `workspace:\n auto_detect: true`) - // — that's the legacy config.WorkspaceConfig shape, now - // migrated to `multi:` instead. Skip. + // Empty or mapping-valued workspace declarations do not identify + // a workspace slug. continue } // A workspace slug may never collide with the reserved local diff --git a/internal/docs/docs.go b/internal/docs/docs.go index 1ee1514a7..0540174ec 100644 --- a/internal/docs/docs.go +++ b/internal/docs/docs.go @@ -1,5 +1,5 @@ // Package docs generates a "living changelog + ownership + blame + -// stale code" bundle from the in-memory graph. Output is markdown or +// stale code" bundle from the graph store. Output is markdown or // JSON; the CLI verb `gortex docs` and the MCP tool `generate_docs` // both call into here. // @@ -287,8 +287,8 @@ func walkNodes(g graph.Store, opts Options, now time.Time) ([]OwnershipRow, []St return ownerRows, stale } -// tsFromMeta normalises int64 (in-process) vs float64 (gob-decoded -// snapshot) timestamps so this package works on both code paths — +// tsFromMeta normalises int64 (in-process) vs float64 (decoded from the +// store's JSON meta) timestamps so this package works on both code paths — // same trick the MCP handlers use today. func tsFromMeta(v any) int64 { switch x := v.(type) { diff --git a/internal/elide/elide.go b/internal/elide/elide.go index 495619b75..9a3660e24 100644 --- a/internal/elide/elide.go +++ b/internal/elide/elide.go @@ -508,20 +508,6 @@ func getSpec(lang string) *languageSpec { return specs[normalizeLang(lang)] } -// Languages reports the canonical language codes elide knows how to -// compress. The returned slice is sorted and safe for the caller to -// retain; it is recomputed on every call so test-only manipulations -// don't bleed across goroutines. -func Languages() []string { - specsOnce.Do(initSpecs) - out := make([]string, 0, len(specs)) - for k := range specs { - out = append(out, k) - } - sort.Strings(out) - return out -} - func normalizeLang(lang string) string { switch strings.ToLower(lang) { case "c++", "cpp", "cxx", "cc": diff --git a/internal/eval/cache.go b/internal/eval/cache.go deleted file mode 100644 index c21b531b9..000000000 --- a/internal/eval/cache.go +++ /dev/null @@ -1,186 +0,0 @@ -package eval - -import ( - "fmt" - "io" - "os" - "path/filepath" - "strings" - - "github.com/zzet/gortex/internal/platform" -) - -// Cache provides filesystem-based index caching keyed by (repo_name, commit_hash). -// Cache entries are stored under {cacheDir}/{repo_name}_{commit_hash}/ and contain -// a .version file, graph.bin, and search.bleve/ directory. -type Cache struct { - dir string // root cache directory - version string // current gortex version for compatibility checks -} - -// NewCache creates a Cache rooted at dir with the given gortex version string. -// -// If dir is empty the location is resolved by env: when $XDG_CACHE_HOME is -// set it is honoured ($XDG_CACHE_HOME/gortex/eval-cache); otherwise the -// historical default ~/.gortex-eval-cache/ is kept so an existing eval -// cache is not orphaned. -func NewCache(dir, version string) (*Cache, error) { - if dir == "" { - if v := os.Getenv("XDG_CACHE_HOME"); v != "" && filepath.IsAbs(v) { - dir = filepath.Join(platform.CacheDir(), "eval-cache") - } else { - home, err := os.UserHomeDir() - if err != nil { - return nil, fmt.Errorf("cache: resolve home dir: %w", err) - } - dir = filepath.Join(home, ".gortex-eval-cache") - } - } - return &Cache{dir: dir, version: version}, nil -} - -// CacheKey generates the cache directory name for a (repo, commit) pair. -// The key format is {repo}_{commit}. -func CacheKey(repo, commit string) string { - return repo + "_" + commit -} - -// entryDir returns the full path to the cache entry directory. -func (c *Cache) entryDir(repo, commit string) string { - return filepath.Join(c.dir, CacheKey(repo, commit)) -} - -// versionFile returns the path to the .version file inside a cache entry. -func (c *Cache) versionFile(repo, commit string) string { - return filepath.Join(c.entryDir(repo, commit), ".version") -} - -// Check returns true if a cached index exists for the (repo, commit) pair. -// It verifies the entry directory exists and contains the expected files. -func (c *Cache) Check(repo, commit string) bool { - dir := c.entryDir(repo, commit) - info, err := os.Stat(dir) - if err != nil || !info.IsDir() { - return false - } - // Verify .version file exists. - if _, err := os.Stat(c.versionFile(repo, commit)); err != nil { - return false - } - return true -} - -// Load returns the path to the cached index directory for the (repo, commit) pair. -// Returns an error if the cache entry does not exist. -func (c *Cache) Load(repo, commit string) (string, error) { - dir := c.entryDir(repo, commit) - info, err := os.Stat(dir) - if err != nil { - return "", fmt.Errorf("cache: entry not found for %s: %w", CacheKey(repo, commit), err) - } - if !info.IsDir() { - return "", fmt.Errorf("cache: entry %s is not a directory", CacheKey(repo, commit)) - } - return dir, nil -} - -// Store persists an index directory into the cache for the (repo, commit) pair. -// It copies the contents of indexPath (graph.bin, search.bleve/) into the cache -// entry directory and writes a .version file with the current gortex version. -func (c *Cache) Store(repo, commit, indexPath string) error { - dir := c.entryDir(repo, commit) - - // Remove any existing entry to ensure a clean store. - if err := os.RemoveAll(dir); err != nil { - return fmt.Errorf("cache: remove existing entry: %w", err) - } - - if err := os.MkdirAll(dir, 0o755); err != nil { - return fmt.Errorf("cache: create entry dir: %w", err) - } - - // Copy contents from indexPath into the cache entry. - if err := copyDir(indexPath, dir); err != nil { - // Clean up on failure. - _ = os.RemoveAll(dir) - return fmt.Errorf("cache: copy index: %w", err) - } - - // Write .version file. - if err := os.WriteFile(c.versionFile(repo, commit), []byte(c.version), 0o644); err != nil { - _ = os.RemoveAll(dir) - return fmt.Errorf("cache: write version file: %w", err) - } - - return nil -} - -// Validate checks whether the cached index for (repo, commit) is compatible -// with the current gortex version by comparing the .version file contents. -func (c *Cache) Validate(repo, commit string) bool { - data, err := os.ReadFile(c.versionFile(repo, commit)) - if err != nil { - return false - } - return strings.TrimSpace(string(data)) == c.version -} - -// Evict removes the cached index entry for the (repo, commit) pair. -func (c *Cache) Evict(repo, commit string) error { - dir := c.entryDir(repo, commit) - if err := os.RemoveAll(dir); err != nil { - return fmt.Errorf("cache: evict %s: %w", CacheKey(repo, commit), err) - } - return nil -} - -// copyDir recursively copies the contents of src into dst. -// dst must already exist. Only regular files and directories are copied. -func copyDir(src, dst string) error { - entries, err := os.ReadDir(src) - if err != nil { - return err - } - - for _, entry := range entries { - srcPath := filepath.Join(src, entry.Name()) - dstPath := filepath.Join(dst, entry.Name()) - - if entry.IsDir() { - if err := os.MkdirAll(dstPath, 0o755); err != nil { - return err - } - if err := copyDir(srcPath, dstPath); err != nil { - return err - } - } else { - if err := copyFile(srcPath, dstPath); err != nil { - return err - } - } - } - return nil -} - -// copyFile copies a single file from src to dst, preserving permissions. -func copyFile(src, dst string) error { - srcFile, err := os.Open(src) - if err != nil { - return err - } - defer srcFile.Close() - - srcInfo, err := srcFile.Stat() - if err != nil { - return err - } - - dstFile, err := os.OpenFile(dst, os.O_CREATE|os.O_WRONLY|os.O_TRUNC, srcInfo.Mode()) - if err != nil { - return err - } - defer dstFile.Close() - - _, err = io.Copy(dstFile, srcFile) - return err -} diff --git a/internal/eval/cache_property_test.go b/internal/eval/cache_property_test.go deleted file mode 100644 index 82c653ad9..000000000 --- a/internal/eval/cache_property_test.go +++ /dev/null @@ -1,75 +0,0 @@ -package eval - -import ( - "testing" - - "github.com/stretchr/testify/assert" - "pgregory.net/rapid" -) - -// Feature: eval-framework, Property 6: Cache key determinism and uniqueness - -// --- Generators --- - -// genRepoCommitPair generates a (repo, commit) pair with non-empty strings -// that don't contain underscores (to avoid ambiguity in the key format). -func genRepoCommitPair() *rapid.Generator[[2]string] { - return rapid.Custom(func(t *rapid.T) [2]string { - repo := rapid.StringMatching(`[a-zA-Z0-9\-\.\/]{1,50}`).Draw(t, "repo") - commit := rapid.StringMatching(`[a-f0-9]{7,40}`).Draw(t, "commit") - return [2]string{repo, commit} - }) -} - -// genDistinctRepoCommitPairs generates two distinct (repo, commit) pairs. -func genDistinctRepoCommitPairs() *rapid.Generator[[2][2]string] { - return rapid.Custom(func(t *rapid.T) [2][2]string { - pair1 := genRepoCommitPair().Draw(t, "pair1") - pair2 := genRepoCommitPair().Draw(t, "pair2") - - // Ensure the pairs are actually distinct - for pair1[0] == pair2[0] && pair1[1] == pair2[1] { - pair2 = genRepoCommitPair().Draw(t, "pair2_retry") - } - - return [2][2]string{pair1, pair2} - }) -} - -// --- Property Tests --- - -// TestProperty6_CacheKeyDeterminism verifies that CacheKey always returns the -// same result for the same (repo, commit) inputs. -// **Validates: Requirements 4.4** -func TestProperty6_CacheKeyDeterminism(t *testing.T) { - rapid.Check(t, func(t *rapid.T) { - pair := genRepoCommitPair().Draw(t, "pair") - repo, commit := pair[0], pair[1] - - key1 := CacheKey(repo, commit) - key2 := CacheKey(repo, commit) - key3 := CacheKey(repo, commit) - - assert.Equal(t, key1, key2, "CacheKey must be deterministic: first and second calls differ") - assert.Equal(t, key1, key3, "CacheKey must be deterministic: first and third calls differ") - assert.NotEmpty(t, key1, "CacheKey must produce a non-empty string") - }) -} - -// TestProperty6_CacheKeyUniqueness verifies that two distinct (repo, commit) -// pairs always produce different cache keys. -// **Validates: Requirements 4.4** -func TestProperty6_CacheKeyUniqueness(t *testing.T) { - rapid.Check(t, func(t *rapid.T) { - pairs := genDistinctRepoCommitPairs().Draw(t, "pairs") - repo1, commit1 := pairs[0][0], pairs[0][1] - repo2, commit2 := pairs[1][0], pairs[1][1] - - key1 := CacheKey(repo1, commit1) - key2 := CacheKey(repo2, commit2) - - assert.NotEqual(t, key1, key2, - "CacheKey must produce different keys for distinct pairs: (%q, %q) vs (%q, %q)", - repo1, commit1, repo2, commit2) - }) -} diff --git a/internal/eval/cache_test.go b/internal/eval/cache_test.go deleted file mode 100644 index 8cab969cb..000000000 --- a/internal/eval/cache_test.go +++ /dev/null @@ -1,159 +0,0 @@ -package eval - -import ( - "os" - "path/filepath" - "testing" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" -) - -func TestNewCache_EmptyDirDefaultsToHome(t *testing.T) { - c, err := NewCache("", "v1.0.0") - require.NoError(t, err) - - home, err := os.UserHomeDir() - require.NoError(t, err) - - expected := filepath.Join(home, ".gortex-eval-cache") - assert.Equal(t, expected, c.dir) -} - -func TestCheck_ReturnsFalseForNonExistentEntry(t *testing.T) { - c, err := NewCache(t.TempDir(), "v1.0.0") - require.NoError(t, err) - - assert.False(t, c.Check("myrepo", "abc123")) -} - -func TestStoreAndCheck(t *testing.T) { - cacheDir := t.TempDir() - c, err := NewCache(cacheDir, "v1.0.0") - require.NoError(t, err) - - // Create a fake index directory with some content. - indexDir := t.TempDir() - require.NoError(t, os.WriteFile(filepath.Join(indexDir, "graph.bin"), []byte("graph-data"), 0o644)) - - require.NoError(t, c.Store("myrepo", "abc123", indexDir)) - assert.True(t, c.Check("myrepo", "abc123")) -} - -func TestStoreAndLoad(t *testing.T) { - cacheDir := t.TempDir() - c, err := NewCache(cacheDir, "v1.0.0") - require.NoError(t, err) - - indexDir := t.TempDir() - require.NoError(t, os.WriteFile(filepath.Join(indexDir, "graph.bin"), []byte("graph-data"), 0o644)) - - require.NoError(t, c.Store("myrepo", "abc123", indexDir)) - - path, err := c.Load("myrepo", "abc123") - require.NoError(t, err) - - expected := filepath.Join(cacheDir, "myrepo_abc123") - assert.Equal(t, expected, path) - - // Verify the copied content is intact. - data, err := os.ReadFile(filepath.Join(path, "graph.bin")) - require.NoError(t, err) - assert.Equal(t, "graph-data", string(data)) -} - -func TestStoreAndValidate_MatchingVersion(t *testing.T) { - cacheDir := t.TempDir() - c, err := NewCache(cacheDir, "v1.0.0") - require.NoError(t, err) - - indexDir := t.TempDir() - require.NoError(t, os.WriteFile(filepath.Join(indexDir, "graph.bin"), []byte("data"), 0o644)) - - require.NoError(t, c.Store("myrepo", "abc123", indexDir)) - assert.True(t, c.Validate("myrepo", "abc123")) -} - -func TestValidate_ReturnsFalseForMismatchedVersion(t *testing.T) { - cacheDir := t.TempDir() - c1, err := NewCache(cacheDir, "v1.0.0") - require.NoError(t, err) - - indexDir := t.TempDir() - require.NoError(t, os.WriteFile(filepath.Join(indexDir, "graph.bin"), []byte("data"), 0o644)) - - require.NoError(t, c1.Store("myrepo", "abc123", indexDir)) - - // Create a new cache instance with a different version. - c2, err := NewCache(cacheDir, "v2.0.0") - require.NoError(t, err) - - assert.False(t, c2.Validate("myrepo", "abc123")) -} - -func TestEvict_RemovesEntry(t *testing.T) { - cacheDir := t.TempDir() - c, err := NewCache(cacheDir, "v1.0.0") - require.NoError(t, err) - - indexDir := t.TempDir() - require.NoError(t, os.WriteFile(filepath.Join(indexDir, "graph.bin"), []byte("data"), 0o644)) - - require.NoError(t, c.Store("myrepo", "abc123", indexDir)) - assert.True(t, c.Check("myrepo", "abc123")) - - require.NoError(t, c.Evict("myrepo", "abc123")) - assert.False(t, c.Check("myrepo", "abc123")) -} - -func TestStore_OverwritesExistingEntry(t *testing.T) { - cacheDir := t.TempDir() - c, err := NewCache(cacheDir, "v1.0.0") - require.NoError(t, err) - - // Store first version. - indexDir1 := t.TempDir() - require.NoError(t, os.WriteFile(filepath.Join(indexDir1, "graph.bin"), []byte("old-data"), 0o644)) - require.NoError(t, c.Store("myrepo", "abc123", indexDir1)) - - // Store second version — should overwrite. - indexDir2 := t.TempDir() - require.NoError(t, os.WriteFile(filepath.Join(indexDir2, "graph.bin"), []byte("new-data"), 0o644)) - require.NoError(t, c.Store("myrepo", "abc123", indexDir2)) - - path, err := c.Load("myrepo", "abc123") - require.NoError(t, err) - - data, err := os.ReadFile(filepath.Join(path, "graph.bin")) - require.NoError(t, err) - assert.Equal(t, "new-data", string(data)) -} - -func TestVersionMismatch_StoreV1_ValidateWithV2(t *testing.T) { - cacheDir := t.TempDir() - - // Store with v1. - c1, err := NewCache(cacheDir, "v1.0.0") - require.NoError(t, err) - - indexDir := t.TempDir() - require.NoError(t, os.WriteFile(filepath.Join(indexDir, "graph.bin"), []byte("data"), 0o644)) - require.NoError(t, c1.Store("myrepo", "abc123", indexDir)) - - // Create cache with v2 — validate should return false. - c2, err := NewCache(cacheDir, "v2.0.0") - require.NoError(t, err) - - assert.False(t, c2.Validate("myrepo", "abc123")) - - // Entry still exists (Check is version-agnostic). - assert.True(t, c2.Check("myrepo", "abc123")) - - // Evict the stale entry. - require.NoError(t, c2.Evict("myrepo", "abc123")) - assert.False(t, c2.Check("myrepo", "abc123")) - - // Re-store with v2. - require.NoError(t, c2.Store("myrepo", "abc123", indexDir)) - assert.True(t, c2.Validate("myrepo", "abc123")) -} diff --git a/internal/eval/integration_test.go b/internal/eval/integration_test.go index 5a15d65fa..6aa12df46 100644 --- a/internal/eval/integration_test.go +++ b/internal/eval/integration_test.go @@ -5,8 +5,6 @@ import ( "encoding/json" "net/http" "net/http/httptest" - "os" - "path/filepath" "strings" "testing" @@ -147,95 +145,3 @@ func TestEvalServerLifecycle(t *testing.T) { // --- Shutdown is implicit: ts.Close() in defer --- } - -// TestIndexCacheRoundTrip is an integration test that exercises the full -// cache lifecycle: create files → store → load → verify integrity → evict. -func TestIndexCacheRoundTrip(t *testing.T) { - cacheDir := t.TempDir() - version := "0.2.0-test" - - cache, err := NewCache(cacheDir, version) - require.NoError(t, err) - - // --- Step 1: Create a realistic index directory with multiple files --- - indexDir := t.TempDir() - require.NoError(t, os.WriteFile( - filepath.Join(indexDir, "graph.bin"), - []byte("serialized-graph-data-with-nodes-and-edges"), - 0o644, - )) - // Create a nested directory simulating search.bleve/ - bleveDir := filepath.Join(indexDir, "search.bleve") - require.NoError(t, os.MkdirAll(bleveDir, 0o755)) - require.NoError(t, os.WriteFile( - filepath.Join(bleveDir, "store"), - []byte("bleve-store-data"), - 0o644, - )) - require.NoError(t, os.WriteFile( - filepath.Join(bleveDir, "index_meta.json"), - []byte(`{"version":1}`), - 0o644, - )) - - repo := "django/django" - commit := "a1b2c3d4e5f6" - - // --- Step 2: Verify cache is empty initially --- - assert.False(t, cache.Check(repo, commit), "cache should be empty initially") - - // --- Step 3: Store the index --- - require.NoError(t, cache.Store(repo, commit, indexDir)) - - // --- Step 4: Verify cache entry exists --- - assert.True(t, cache.Check(repo, commit), "cache should have entry after store") - - // --- Step 5: Load and verify path --- - loadedPath, err := cache.Load(repo, commit) - require.NoError(t, err) - expectedPath := filepath.Join(cacheDir, CacheKey(repo, commit)) - assert.Equal(t, expectedPath, loadedPath) - - // --- Step 6: Verify all files are intact --- - graphData, err := os.ReadFile(filepath.Join(loadedPath, "graph.bin")) - require.NoError(t, err) - assert.Equal(t, "serialized-graph-data-with-nodes-and-edges", string(graphData)) - - bleveStore, err := os.ReadFile(filepath.Join(loadedPath, "search.bleve", "store")) - require.NoError(t, err) - assert.Equal(t, "bleve-store-data", string(bleveStore)) - - bleveMeta, err := os.ReadFile(filepath.Join(loadedPath, "search.bleve", "index_meta.json")) - require.NoError(t, err) - assert.Equal(t, `{"version":1}`, string(bleveMeta)) - - // --- Step 7: Validate version compatibility --- - assert.True(t, cache.Validate(repo, commit), "version should match") - - // --- Step 8: Version mismatch detection --- - cacheV2, err := NewCache(cacheDir, "0.3.0-different") - require.NoError(t, err) - assert.False(t, cacheV2.Validate(repo, commit), "different version should fail validation") - - // --- Step 9: Re-store with updated content --- - indexDir2 := t.TempDir() - require.NoError(t, os.WriteFile( - filepath.Join(indexDir2, "graph.bin"), - []byte("updated-graph-data"), - 0o644, - )) - require.NoError(t, cache.Store(repo, commit, indexDir2)) - - loadedPath2, err := cache.Load(repo, commit) - require.NoError(t, err) - updatedData, err := os.ReadFile(filepath.Join(loadedPath2, "graph.bin")) - require.NoError(t, err) - assert.Equal(t, "updated-graph-data", string(updatedData)) - - // --- Step 10: Evict and verify removal --- - require.NoError(t, cache.Evict(repo, commit)) - assert.False(t, cache.Check(repo, commit), "cache should be empty after eviction") - - _, err = cache.Load(repo, commit) - assert.Error(t, err, "load after eviction should fail") -} diff --git a/internal/eval/recall/rankers.go b/internal/eval/recall/rankers.go index 2088df90b..de11fcaf8 100644 --- a/internal/eval/recall/rankers.go +++ b/internal/eval/recall/rankers.go @@ -15,8 +15,9 @@ import ( ) // BM25Ranker adapts a plain search.Backend to the Ranker shape. Works -// for either a raw BM25/Bleve backend or a HybridBackend's text side -// extracted via HybridBackend.TextBackend(). +// for either a raw text backend — the in-process BM25 index the evals +// build, or the store-native FTS adapter production wires up — or a +// HybridBackend's text side extracted via HybridBackend.TextBackend(). // // Note: Gortex's indexer tokenizes symbol names at ingest time // (Tokenize — camelCase-aware), but the query side (TokenizeQuery) does @@ -41,7 +42,7 @@ func BM25Ranker(name string, backend search.Backend) Ranker { } // EngineRanker measures what a real MCP caller sees via -// Engine.SearchSymbols — BM25/Bleve results + camelCase-friendly +// Engine.SearchSymbols — text-backend results + camelCase-friendly // substring fallback. This is the recommended default for "bm25"- // style evaluation; it reflects production behaviour. func EngineRanker(name string, searchFn func(query string, limit int) []string) Ranker { diff --git a/internal/exporter/exporter.go b/internal/exporter/exporter.go index 2b2d474ef..0662fe15e 100644 --- a/internal/exporter/exporter.go +++ b/internal/exporter/exporter.go @@ -1,9 +1,9 @@ -// Package exporter writes the in-memory graph to portable formats so users -// can load it into external visualization and query tools (Neo4j, Memgraph -// via Cypher; yEd, Gephi, Cytoscape via GraphML). +// Package exporter walks a graph store and writes it to portable formats so +// users can load it into external visualization and query tools (Neo4j, +// Memgraph via Cypher; yEd, Gephi, Cytoscape via GraphML). // -// The exporter is read-only and operates on a snapshot — it never mutates -// the graph. Filters (repo, kinds) are applied during emission. +// The exporter is read-only — it never mutates the graph. Filters (repo, +// kinds) are applied during emission. package exporter import ( diff --git a/internal/githooks/install.go b/internal/githooks/install.go index 1ee6fb98a..413c2ebc0 100644 --- a/internal/githooks/install.go +++ b/internal/githooks/install.go @@ -152,13 +152,6 @@ func hookCommands(hook string, opts InstallOpts) []string { return cmds } -// HookPath resolves the absolute path of the post-commit hook for the -// repository rooted at repoRoot. Honours core.hooksPath when set. -// Thin wrapper over HookPathFor — preserved for backwards compatibility. -func HookPath(repoRoot string) (string, error) { - return HookPathFor(repoRoot, "post-commit") -} - // HookPathFor resolves the absolute path of the named hook file in // the repository rooted at repoRoot. Honours core.hooksPath when set. // hook is a bare hook name from SupportedHooks ("post-commit", @@ -192,46 +185,6 @@ func HookPathFor(repoRoot, hook string) (string, error) { return filepath.Join(hooksDir, hook), nil } -// StatusReport describes the current state of the post-commit hook. -type StatusReport struct { - HookPath string `json:"hook_path"` - Exists bool `json:"exists"` - Managed bool `json:"managed"` // true iff our marker block is present - Body string `json:"body,omitempty"` -} - -// Status reports the current state of the post-commit hook. Never -// modifies anything. -func Status(repoRoot string) (StatusReport, error) { - path, err := HookPath(repoRoot) - if err != nil { - return StatusReport{}, err - } - body, err := os.ReadFile(path) - if err != nil { - if os.IsNotExist(err) { - return StatusReport{HookPath: path}, nil - } - return StatusReport{}, fmt.Errorf("githooks: read %q: %w", path, err) - } - rep := StatusReport{ - HookPath: path, - Exists: true, - Body: string(body), - } - if bytes.Contains(body, []byte(MarkerBegin)) && bytes.Contains(body, []byte(MarkerEnd)) { - rep.Managed = true - } - return rep, nil -} - -// InstallPostCommit is a backwards-compatible wrapper over InstallHook -// that installs the post-commit hook. New callers should reach for -// InstallHook directly so they can install post-merge too. -func InstallPostCommit(repoRoot string, opts InstallOpts) (string, error) { - return InstallHook(repoRoot, "post-commit", opts) -} - // InstallHook writes the named hook with the configured commands // inside a hook-specific marker block. Idempotent: re-running replaces // just the gortex block, leaving any other content intact. @@ -300,11 +253,6 @@ func InstallHook(repoRoot, hook string, opts InstallOpts) (string, error) { return hookPath, nil } -// UninstallPostCommit is a backwards-compatible wrapper. -func UninstallPostCommit(repoRoot string) (string, bool, error) { - return UninstallHook(repoRoot, "post-commit") -} - // UninstallHook removes the gortex-managed block from the named hook. // If the file then contains nothing but the shebang and our installer // comment, the file is deleted entirely. Otherwise we leave the diff --git a/internal/githooks/install_test.go b/internal/githooks/install_test.go index 0de5217f0..01ea3e489 100644 --- a/internal/githooks/install_test.go +++ b/internal/githooks/install_test.go @@ -26,9 +26,9 @@ func initRepo(t *testing.T) string { return tmp } -func TestInstallPostCommit_FreshFile(t *testing.T) { +func TestInstallHookPostCommit_FreshFile(t *testing.T) { repo := initRepo(t) - path, err := InstallPostCommit(repo, InstallOpts{RegenMermaid: true, RegenWiki: true, Binary: "gortex"}) + path, err := InstallHook(repo, "post-commit", InstallOpts{RegenMermaid: true, RegenWiki: true, Binary: "gortex"}) if err != nil { t.Fatalf("Install: %v", err) } @@ -57,33 +57,35 @@ func TestInstallPostCommit_FreshFile(t *testing.T) { } } -func TestInstallPostCommit_Idempotent(t *testing.T) { +func TestInstallHookPostCommit_Idempotent(t *testing.T) { repo := initRepo(t) for i := range 3 { - if _, err := InstallPostCommit(repo, InstallOpts{RegenMermaid: true}); err != nil { + if _, err := InstallHook(repo, "post-commit", InstallOpts{RegenMermaid: true}); err != nil { t.Fatalf("install %d: %v", i, err) } } - rep, err := Status(repo) + hookPath, err := HookPathFor(repo, "post-commit") if err != nil { - t.Fatalf("Status: %v", err) + t.Fatalf("HookPathFor: %v", err) } - if !rep.Managed { - t.Error("after install, status should report managed") + body, err := os.ReadFile(hookPath) + if err != nil { + t.Fatalf("read hook: %v", err) } - if c := strings.Count(rep.Body, MarkerBegin); c != 1 { + got := string(body) + if c := strings.Count(got, MarkerBegin); c != 1 { t.Errorf("expected one MarkerBegin, got %d", c) } - if c := strings.Count(rep.Body, MarkerEnd); c != 1 { + if c := strings.Count(got, MarkerEnd); c != 1 { t.Errorf("expected one MarkerEnd, got %d", c) } } -func TestInstallPostCommit_PreservesUserContent(t *testing.T) { +func TestInstallHookPostCommit_PreservesUserContent(t *testing.T) { repo := initRepo(t) - hookPath, err := HookPath(repo) + hookPath, err := HookPathFor(repo, "post-commit") if err != nil { - t.Fatalf("HookPath: %v", err) + t.Fatalf("HookPathFor: %v", err) } preexisting := `#!/bin/sh # my custom hook @@ -92,7 +94,7 @@ echo "hello from user hook" if err := os.WriteFile(hookPath, []byte(preexisting), 0o755); err != nil { t.Fatalf("write preexisting: %v", err) } - if _, err := InstallPostCommit(repo, InstallOpts{RegenMermaid: true}); err != nil { + if _, err := InstallHook(repo, "post-commit", InstallOpts{RegenMermaid: true}); err != nil { t.Fatalf("Install: %v", err) } body, err := os.ReadFile(hookPath) @@ -108,11 +110,11 @@ echo "hello from user hook" } } -func TestUninstallPostCommit_RemovesBlock(t *testing.T) { +func TestUninstallHookPostCommit_RemovesBlock(t *testing.T) { repo := initRepo(t) - hookPath, err := HookPath(repo) + hookPath, err := HookPathFor(repo, "post-commit") if err != nil { - t.Fatalf("HookPath: %v", err) + t.Fatalf("HookPathFor: %v", err) } preexisting := `#!/bin/sh # my custom hook @@ -121,10 +123,10 @@ echo "hello" if err := os.WriteFile(hookPath, []byte(preexisting), 0o755); err != nil { t.Fatalf("write preexisting: %v", err) } - if _, err := InstallPostCommit(repo, InstallOpts{RegenWiki: true}); err != nil { + if _, err := InstallHook(repo, "post-commit", InstallOpts{RegenWiki: true}); err != nil { t.Fatalf("Install: %v", err) } - path, removed, err := UninstallPostCommit(repo) + path, removed, err := UninstallHook(repo, "post-commit") if err != nil { t.Fatalf("Uninstall: %v", err) } @@ -147,12 +149,12 @@ echo "hello" } } -func TestUninstallPostCommit_RemovesFileWhenStubOnly(t *testing.T) { +func TestUninstallHookPostCommit_RemovesFileWhenStubOnly(t *testing.T) { repo := initRepo(t) - if _, err := InstallPostCommit(repo, InstallOpts{RegenMermaid: true}); err != nil { + if _, err := InstallHook(repo, "post-commit", InstallOpts{RegenMermaid: true}); err != nil { t.Fatalf("Install: %v", err) } - path, removed, err := UninstallPostCommit(repo) + path, removed, err := UninstallHook(repo, "post-commit") if err != nil { t.Fatalf("Uninstall: %v", err) } @@ -164,9 +166,9 @@ func TestUninstallPostCommit_RemovesFileWhenStubOnly(t *testing.T) { } } -func TestUninstallPostCommit_Noop(t *testing.T) { +func TestUninstallHookPostCommit_Noop(t *testing.T) { repo := initRepo(t) - path, removed, err := UninstallPostCommit(repo) + path, removed, err := UninstallHook(repo, "post-commit") if err != nil { t.Fatalf("Uninstall: %v", err) } @@ -178,20 +180,6 @@ func TestUninstallPostCommit_Noop(t *testing.T) { } } -func TestStatus_NewRepo(t *testing.T) { - repo := initRepo(t) - rep, err := Status(repo) - if err != nil { - t.Fatalf("Status: %v", err) - } - if rep.Exists { - t.Error("fresh repo shouldn't have a hook") - } - if rep.Managed { - t.Error("fresh repo shouldn't be managed") - } -} - func TestInstallHook_PostMergeAndChurn(t *testing.T) { repo := initRepo(t) path, err := InstallHook(repo, "post-merge", InstallOpts{RegenChurn: true, ChurnBranch: "origin/main"}) @@ -264,7 +252,7 @@ func TestInstallHook_RejectsUnsupportedHook(t *testing.T) { } } -func TestHookPath_HonoursCoreHooksPath(t *testing.T) { +func TestHookPathFor_HonoursCoreHooksPath(t *testing.T) { repo := initRepo(t) customHooks := filepath.Join(repo, "custom-hooks") if err := os.MkdirAll(customHooks, 0o755); err != nil { @@ -275,12 +263,12 @@ func TestHookPath_HonoursCoreHooksPath(t *testing.T) { if out, err := cmd.CombinedOutput(); err != nil { t.Fatalf("config core.hooksPath: %v: %s", err, out) } - path, err := HookPath(repo) + path, err := HookPathFor(repo, "post-commit") if err != nil { - t.Fatalf("HookPath: %v", err) + t.Fatalf("HookPathFor: %v", err) } if filepath.Dir(path) != customHooks { - t.Errorf("HookPath should honour core.hooksPath, got %q under %q (want %q)", + t.Errorf("HookPathFor should honour core.hooksPath, got %q under %q (want %q)", path, filepath.Dir(path), customHooks) } } diff --git a/internal/graph/config_node_evict.go b/internal/graph/config_node_evict.go index 5f72f3e7b..e2fca171b 100644 --- a/internal/graph/config_node_evict.go +++ b/internal/graph/config_node_evict.go @@ -7,21 +7,6 @@ type ConfigNodeBatchEvicter interface { EvictConfigNodesByIDs(ids []string) (nodesRemoved, edgesRemoved int) } -// EvictConfigNodesByIDs dispatches only to the set-oriented capability. -// Production Graph and SQLite stores implement it; adapters report unsupported -// instead of hiding a point-delete loop. -func EvictConfigNodesByIDs(store Store, ids []string) (nodesRemoved, edgesRemoved int, supported bool) { - if store == nil || len(ids) == 0 { - return 0, 0, true - } - evicter, ok := store.(ConfigNodeBatchEvicter) - if !ok { - return 0, 0, false - } - nodesRemoved, edgesRemoved = evicter.EvictConfigNodesByIDs(ids) - return nodesRemoved, edgesRemoved, true -} - // EvictConfigNodesByIDs implements the bounded in-memory capability while // holding all shard locks once. Unknown and non-config-key IDs are ignored. func (g *Graph) EvictConfigNodesByIDs(ids []string) (nodesRemoved, edgesRemoved int) { diff --git a/internal/graph/graph.go b/internal/graph/graph.go index 67a283543..737bdde7a 100644 --- a/internal/graph/graph.go +++ b/internal/graph/graph.go @@ -555,6 +555,15 @@ type Graph struct { // absent from disk stores until they can provide the same completeness // guarantee. mutationReceipts mutationReceiptState + + // structuralIntegrity is the store-scoped since-open recorder. Shadows may + // forward through structuralIntegritySink so rejected attempts survive the + // throwaway graph; the explicit repo and fixed path prevent attribution + // from guessing ownership from unprefixed node IDs. + structuralIntegrity StructuralIntegrityMeter + structuralIntegritySink StructuralIntegrityEventRecorder + structuralIntegrityRepo string + structuralIntegrityPath StructuralDropPath } // cloneShingleEntry is one in-memory clone_shingles row: the owning @@ -1695,43 +1704,6 @@ func (g *Graph) AllEdgesLight(kinds ...EdgeKind) []*Edge { return out } -// EdgesForKindsLight returns the edges of the given kinds (an empty kinds list -// means every kind) for a whole-graph scan, preferring the meta-less -// LightEdgeScanner capability when the store implements it (skips the per-edge -// Meta decode on disk backends) and otherwise falling back to AllEdges() with a -// Go-side kind filter. See LightEdgeScanner for the Meta-presence contract: -// callers must read only the promoted edge fields, never arbitrary Meta. -func EdgesForKindsLight(g Store, kinds ...EdgeKind) []*Edge { - if g == nil { - return nil - } - if sc, ok := g.(LightEdgeScanner); ok { - return sc.AllEdgesLight(kinds...) - } - all := g.AllEdges() - if len(kinds) == 0 { - return all - } - want := make(map[EdgeKind]struct{}, len(kinds)) - for _, k := range kinds { - if k != "" { - want[k] = struct{}{} - } - } - if len(want) == 0 { - return nil - } - out := make([]*Edge, 0, len(all)) - for _, e := range all { - if e != nil { - if _, ok := want[e.Kind]; ok { - out = append(out, e) - } - } - } - return out -} - // DeadCodeCandidates is the in-memory reference implementation of // DeadCodeCandidator. Iterates the requested node kinds and filters // out anything whose incoming-edge bucket contains an allowlist match @@ -2620,8 +2592,10 @@ func (g *Graph) AddBatch(nodes []*Node, edges []*Edge) { if len(nodes) == 0 && len(edges) == 0 { return } - // Structural-shape backstop: see StructuralEdgeTargetInvalid. - edges, _ = FilterStructuralEdgeViolations(edges) + // Structural-shape backstop: the first rejecting boundary owns the event. + var rejected []*Edge + edges, rejected = FilterStructuralEdgeViolations(edges) + g.recordStructuralRejections(StructuralPathGraphAddBatch, rejected, nodes) // Lazy builtin-sentinel materialization: see BuiltinStubNodes. The // per-store seen-set keeps it one upsert per stub per store lifetime. if stubs := BuiltinStubNodes(edges); len(stubs) > 0 { @@ -2749,9 +2723,12 @@ func (g *Graph) AddBatch(nodes []*Node, edges []*Edge) { // adjacency-list length is unchanged. Drops the double-edge problem // that used to surface after daemon restarts (bug B1). func (g *Graph) AddEdge(e *Edge) { - // Structural-shape backstop: see StructuralEdgeTargetInvalid. - if e != nil && StructuralEdgeTargetInvalid(e.Kind, e.To) { - structuralWriteDrops.Add(1) + if e == nil { + return + } + // Structural-shape backstop: the first rejecting boundary owns the event. + if StructuralEdgeTargetInvalid(e.Kind, e.To) { + g.recordStructuralRejections(StructuralPathGraphAddEdge, []*Edge{e}, nil) return } receiptActive := g.beginReceiptMutation() diff --git a/internal/graph/newfence_test.go b/internal/graph/newfence_test.go index 57e995e97..51a14254e 100644 --- a/internal/graph/newfence_test.go +++ b/internal/graph/newfence_test.go @@ -22,9 +22,7 @@ import ( // stagingCallers lists the non-test files allowed to construct it. Keeping the // list this short is the point: a new entry means some production path is about // to hold graph data that no restart can recover. -var stagingCallers = []string{ - "internal/indexer/indexer.go", -} +var stagingCallers []string // graphImportPath is the package under fence. The scan resolves whatever local // name each file binds it to, so an alias or a dot import is caught the same as diff --git a/internal/graph/store.go b/internal/graph/store.go index d512798e9..4e6efa801 100644 --- a/internal/graph/store.go +++ b/internal/graph/store.go @@ -514,7 +514,7 @@ type SymbolFTSItem struct { // expose engine-native full-text search over the graph's symbol // names. When the backing store implements it, the daemon's // search_symbols path routes through the backend FTS instead of -// building a parallel in-process Bleve/BM25 index — saving ~100MB +// building a parallel in-process BM25 index — saving ~100MB // of heap on a vscode-scale repo and putting the search latency in // the same address space as the rest of the graph. // @@ -789,6 +789,7 @@ type VectorItem struct { // translate via `1 - distance` for cosine. type VectorHit struct { NodeID string + ParentID string Distance float64 } @@ -847,6 +848,44 @@ type VectorSearcher interface { GetEmbeddings(ids []string) map[string][]float32 } +// VectorCorpusItem is one row in an atomically replaceable repository vector +// corpus. ParentID is empty for an ordinary symbol vector and identifies the +// owning symbol for a synthetic chunk vector. +type VectorCorpusItem struct { + NodeID string + ParentID string + Vec []float32 +} + +// VectorCorpusStats describes the complete committed durable corpus for one +// embedding dimension. ChunkCount counts rows with a non-empty ParentID. The +// Repository fields are populated by replacement and repo-scoped discovery so +// warm startup can distinguish "another repo has vectors" from "this repo's +// migration-cleared corpus still needs rebuilding". +type VectorCorpusStats struct { + VectorCount int + ChunkCount int + RepositoryVectorCount int + RepositoryChunkCount int + Dims int +} + +// AtomicVectorCorpusInstaller is an optional durable-store capability. A +// replacement validates the complete input before mutation, swaps exactly one +// repository's rows in one transaction, and returns complete post-commit stats +// for dims across every repository in the store. An empty item slice is an +// authoritative empty corpus and therefore removes stale rows for repoPrefix. +// +// VectorCorpusStats filters by dims when dims is positive. When dims is zero or +// negative, an empty store returns zero stats, a single-dimension store returns +// that resolved dimension, and a mixed-dimension store returns an error rather +// than guessing which corpus a warm-start delegate should publish. +type AtomicVectorCorpusInstaller interface { + ReplaceVectorCorpus(ctx context.Context, repoPrefix string, dims int, items []VectorCorpusItem) (VectorCorpusStats, error) + VectorCorpusStats(ctx context.Context, dims int) (VectorCorpusStats, error) + VectorCorpusStatsForRepo(ctx context.Context, repoPrefix string, dims int) (VectorCorpusStats, error) +} + // PageRankOpts tunes the PageRank computation. Zero values request // the backend default — only set fields you genuinely want to // override so backends can pick their own parallel-tuned defaults diff --git a/internal/graph/store_sqlite/add_batch_set.go b/internal/graph/store_sqlite/add_batch_set.go index 7110a4e35..37441b464 100644 --- a/internal/graph/store_sqlite/add_batch_set.go +++ b/internal/graph/store_sqlite/add_batch_set.go @@ -3,7 +3,6 @@ package store_sqlite import ( "context" "database/sql" - "log" "strings" "github.com/zzet/gortex/internal/graph" @@ -514,9 +513,13 @@ func (s *Store) addBatchSetOriented(nodes []*graph.Node, edges []*graph.Edge) (s // and AddBatch route here; the bulk JSONB chunks are emitted below). // See graph.StructuralEdgeTargetInvalid for the mapper-bug class this // stops at the door. - if kept, dropped := graph.FilterStructuralEdgeViolations(edges); dropped > 0 { + if kept, rejected := graph.FilterStructuralEdgeViolations(edges); len(rejected) > 0 { edges = kept - log.Printf("store_sqlite: dropped %d structurally invalid edges (kind cannot target a param/local node)", dropped) + inputRepos := structuralInputNodeRepos(nodes) + for _, edge := range rejected { + repo := s.structuralWriteRepo(edge, inputRepos) + s.recordStructuralEdge(graph.StructuralDropWrite, graph.StructuralPathSQLiteAddBatch, repo, edge) + } } // Lazy builtin-sentinel materialization, mirroring Graph.AddBatch: give // every ::builtin:: edge target a real KindBuiltin node so those edges diff --git a/internal/graph/store_sqlite/analysis_projection.go b/internal/graph/store_sqlite/analysis_projection.go index e43b7be79..efedca612 100644 --- a/internal/graph/store_sqlite/analysis_projection.go +++ b/internal/graph/store_sqlite/analysis_projection.go @@ -51,7 +51,7 @@ func (s *Store) EdgesLightSeq(kinds ...graph.EdgeKind) iter.Seq[*graph.Edge] { } defer rows.Close() for rows.Next() { - edge, scanErr := scanEdgeLight(rows) + edge, scanErr := s.scanEdgeLight(rows) if scanErr != nil { panicOnFatal(scanErr) return diff --git a/internal/graph/store_sqlite/dataflow_batch.go b/internal/graph/store_sqlite/dataflow_batch.go index 3ebb07267..1b50f0b52 100644 --- a/internal/graph/store_sqlite/dataflow_batch.go +++ b/internal/graph/store_sqlite/dataflow_batch.go @@ -9,14 +9,6 @@ import ( const maxEdgeKindScanBatch = 4096 -// ScanDataflowEdgesBatched retains the original dataflow-specific capability -// while delegating to the generic bounded kind scanner. -func (s *Store) ScanDataflowEdgesBatched(batchSize int, yield func([]*graph.Edge) bool) { - s.ScanEdgesByKindsBatched( - []graph.EdgeKind{graph.EdgeArgOf, graph.EdgeReturnsTo}, batchSize, yield, - ) -} - // ScanEdgesByKindsBatched walks full edge rows through bounded row-id pages. // The high-water mark is captured before the first page, so delete+insert // identity rewrites cannot make a just-rewritten row reappear in the same @@ -229,7 +221,7 @@ func (s *Store) queryDataflowLightActive(query string, args ...any) []*graph.Edg defer rows.Close() var out []*graph.Edge for rows.Next() { - edge, err := scanEdgeLight(rows) + edge, err := s.scanEdgeLight(rows) if err != nil { panicOnFatal(err) return out diff --git a/internal/graph/store_sqlite/dataflow_batch_test.go b/internal/graph/store_sqlite/dataflow_batch_test.go index 07d6a7486..0452f04f8 100644 --- a/internal/graph/store_sqlite/dataflow_batch_test.go +++ b/internal/graph/store_sqlite/dataflow_batch_test.go @@ -11,7 +11,7 @@ import ( "github.com/zzet/gortex/internal/graph" ) -func TestScanDataflowEdgesBatchedUsesFixedHighWaterAcrossReindex(t *testing.T) { +func TestScanEdgesByKindsBatchedUsesFixedHighWaterAcrossReindex(t *testing.T) { store, err := Open(filepath.Join(t.TempDir(), "graph.sqlite")) require.NoError(t, err) t.Cleanup(func() { require.NoError(t, store.Close()) }) @@ -30,7 +30,7 @@ func TestScanDataflowEdgesBatchedUsesFixedHighWaterAcrossReindex(t *testing.T) { seen := make(map[string]int, count) batches := 0 - store.ScanDataflowEdgesBatched(3, func(batch []*graph.Edge) bool { + store.ScanEdgesByKindsBatched([]graph.EdgeKind{graph.EdgeArgOf, graph.EdgeReturnsTo}, 3, func(batch []*graph.Edge) bool { batches++ reindexes := make([]graph.EdgeReindex, 0, len(batch)) for _, edge := range batch { diff --git a/internal/graph/store_sqlite/edge_identity_lookup.go b/internal/graph/store_sqlite/edge_identity_lookup.go index 6a526cc41..221fe942a 100644 --- a/internal/graph/store_sqlite/edge_identity_lookup.go +++ b/internal/graph/store_sqlite/edge_identity_lookup.go @@ -108,7 +108,7 @@ func (s *Store) findEdgesByIdentities(identities []graph.EdgeIdentity) (map[grap } for rows.Next() { - edge, scanErr := scanEdgeCursor(rows) + edge, scanErr := s.scanEdgeCursor(rows) if scanErr != nil { _ = rows.Close() panicOnFatal(scanErr) diff --git a/internal/graph/store_sqlite/lsp_projection.go b/internal/graph/store_sqlite/lsp_projection.go index 1d7dbad95..5b6925c59 100644 --- a/internal/graph/store_sqlite/lsp_projection.go +++ b/internal/graph/store_sqlite/lsp_projection.go @@ -146,7 +146,7 @@ ORDER BY e.from_id, e.to_id, e.kind, e.file_path, e.line`, languagesJSON, filesJ defer rows.Close() var out []*graph.Edge for rows.Next() { - edge, err := scanEdgeCursor(rows) + edge, err := s.scanEdgeCursor(rows) if err != nil { panicOnFatal(err) return out @@ -207,7 +207,7 @@ ORDER BY e.from_id, e.to_id, e.kind, e.file_path, e.line`, languagesJSON, filesJ defer rows.Close() var out []*graph.Edge for rows.Next() { - edge, err := scanEdgeCursor(rows) + edge, err := s.scanEdgeCursor(rows) if err != nil { panicOnFatal(err) return out @@ -295,7 +295,7 @@ ORDER BY e.from_id, e.to_id, e.kind, e.file_path, e.line`, idsJSON, kindsJSON) defer rows.Close() var out []*graph.Edge for rows.Next() { - edge, err := scanEdgeCursor(rows) + edge, err := s.scanEdgeCursor(rows) if err != nil { panicOnFatal(err) return out diff --git a/internal/graph/store_sqlite/migration_unprefixed_purge_test.go b/internal/graph/store_sqlite/migration_unprefixed_purge_test.go index 753ec34c6..e469ad8ac 100644 --- a/internal/graph/store_sqlite/migration_unprefixed_purge_test.go +++ b/internal/graph/store_sqlite/migration_unprefixed_purge_test.go @@ -71,8 +71,9 @@ func TestOpenV6PurgesUnprefixedSoloRepoRows(t *testing.T) { } } - // Embeddings are keyed by node_id alone (vectors has no repo_prefix - // column), so they are only reachable through node membership. + // Legacy embeddings have no trustworthy repository or chunk-parent + // ownership. The later v10 vector-only migration clears this derived + // table while preserving the graph topology exercised above. for _, id := range []string{"internal/foo.go::Bar", "repo/internal/foo.go::Bar"} { if _, err := db.Exec( `INSERT INTO vectors (node_id, dims, vec) VALUES (?, 1, ?)`, id, []byte{0}); err != nil { @@ -119,7 +120,7 @@ func TestOpenV6PurgesUnprefixedSoloRepoRows(t *testing.T) { []string{"repo/internal/foo.go::Bar -> dep::go.uber.org/zap::Logger"}) vectors := queryIDs(t, s.db, `SELECT node_id FROM vectors ORDER BY node_id`) - assertStringsEqual(t, "surviving vectors", vectors, []string{"repo/internal/foo.go::Bar"}) + assertStringsEqual(t, "vectors after v10 derived-cache reset", vectors, nil) mtimes := queryIDs(t, s.db, `SELECT repo_prefix || ':' || file_path FROM file_mtimes ORDER BY 1`) assertStringsEqual(t, "surviving file_mtimes", mtimes, []string{"repo:repo/internal/foo.go"}) diff --git a/internal/graph/store_sqlite/out_edges_light_test.go b/internal/graph/store_sqlite/out_edges_light_test.go index 107b3dbeb..84fa68420 100644 --- a/internal/graph/store_sqlite/out_edges_light_test.go +++ b/internal/graph/store_sqlite/out_edges_light_test.go @@ -9,11 +9,10 @@ import ( "github.com/zzet/gortex/internal/graph" ) -// TestGetOutEdgesLight_SkipsMetaKeepsEndpoints proves the light out-edge -// fetch returns the same endpoints/kind/line as GetOutEdges while leaving -// Meta nil — it must never pay the per-edge meta JSON decode. This is the -// fetch findCallTarget uses on the dataflow hot path. -func TestGetOutEdgesLight_SkipsMetaKeepsEndpoints(t *testing.T) { +// TestAllEdgesLight_SkipsMetaKeepsEndpoints proves the live light-edge scan +// returns the same endpoints/kind/line as GetOutEdges while leaving Meta nil — +// it must never pay the per-edge meta JSON decode. +func TestAllEdgesLight_SkipsMetaKeepsEndpoints(t *testing.T) { s := openTestStore(t) want := &graph.Edge{ @@ -31,7 +30,7 @@ func TestGetOutEdgesLight_SkipsMetaKeepsEndpoints(t *testing.T) { require.NotNil(t, full[0].Meta, "GetOutEdges must decode Meta") assert.Equal(t, "unresolved::Callee", full[0].Meta["callee_target"]) - light := s.GetOutEdgesLight("pkg/x.go::Caller") + light := s.AllEdgesLight() require.Len(t, light, 1) assert.Equal(t, full[0].From, light[0].From) assert.Equal(t, full[0].To, light[0].To) @@ -40,6 +39,4 @@ func TestGetOutEdgesLight_SkipsMetaKeepsEndpoints(t *testing.T) { assert.Equal(t, full[0].FilePath, light[0].FilePath) assert.Nil(t, light[0].Meta, "light fetch must not decode the meta blob") - // A node with no out-edges returns nothing on both paths. - assert.Empty(t, s.GetOutEdgesLight("pkg/x.go::Callee")) } diff --git a/internal/graph/store_sqlite/rawbytes_scan_test.go b/internal/graph/store_sqlite/rawbytes_scan_test.go index 87eb6099b..b3b2f5e4b 100644 --- a/internal/graph/store_sqlite/rawbytes_scan_test.go +++ b/internal/graph/store_sqlite/rawbytes_scan_test.go @@ -51,7 +51,7 @@ func TestCursorMetadataDecodedBeforeRowsNext(t *testing.T) { } var edges []*graph.Edge for edgeRows.Next() { - edge, scanErr := scanEdgeCursor(edgeRows) + edge, scanErr := store.scanEdgeCursor(edgeRows) if scanErr != nil { _ = edgeRows.Close() t.Fatal(scanErr) @@ -211,10 +211,10 @@ func benchmarkEdgeMetadataCursor(b *testing.B, store *Store, raw bool) { for rows.Next() { var edge *graph.Edge if raw { - edge, err = scanEdgeCursor(rows) + edge, err = store.scanEdgeCursor(rows) } else { var metaBlob []byte - edge, err = scanEdgeWithMeta(rows, &metaBlob) + edge, err = scanEdgeWithMeta(store, rows, &metaBlob) } if err != nil { _ = rows.Close() diff --git a/internal/graph/store_sqlite/read_heal_test.go b/internal/graph/store_sqlite/read_heal_test.go index 89dae1085..2e4af70d5 100644 --- a/internal/graph/store_sqlite/read_heal_test.go +++ b/internal/graph/store_sqlite/read_heal_test.go @@ -11,8 +11,8 @@ import ( // A store written BEFORE the write backstop existed can carry structurally // impossible edges. Read paths must heal them — never materialize the junk -// into Go objects (pure GC pressure) — and the drop counter must move, the -// engineer-facing signal that the on-disk store needs an audit/rebuild. +// into Go objects (pure GC pressure) — and the store-scoped integrity signal +// must move so operators know the on-disk store needs an audit/rebuild. func TestReadPathsHealStructurallyInvalidRows(t *testing.T) { s, err := Open(filepath.Join(t.TempDir(), "heal.sqlite")) require.NoError(t, err) @@ -31,11 +31,12 @@ func TestReadPathsHealStructurallyInvalidRows(t *testing.T) { VALUES ('a/t.go::T', 'a/f.go::F#param:ctx', 'implements', 'a/t.go', 1, 1.0, 'EXTRACTED', 'lsp_dispatch', '', 0, NULL)`) require.NoError(t, err) - before := StructuralReadDrops() + before := s.StructuralIntegritySnapshot(graph.StructuralIntegritySnapshotOptions{}).Totals.ReadSuppressed out := s.GetOutEdges("a/t.go::T") for _, e := range out { assert.NotEqual(t, graph.EdgeImplements, e.Kind, "junk implements row must be healed on read") } require.Len(t, out, 1, "the legitimate call edge must survive the heal") - assert.Greater(t, StructuralReadDrops(), before, "healing must move the engineer-facing counter") + after := s.StructuralIntegritySnapshot(graph.StructuralIntegritySnapshotOptions{}).Totals.ReadSuppressed + assert.Greater(t, after, before, "healing must move the store-scoped integrity signal") } diff --git a/internal/graph/store_sqlite/repo_projections.go b/internal/graph/store_sqlite/repo_projections.go index 67f585fb8..23b1744a3 100644 --- a/internal/graph/store_sqlite/repo_projections.go +++ b/internal/graph/store_sqlite/repo_projections.go @@ -326,7 +326,7 @@ func (s *Store) RepoEdgesByKinds(repoPrefixes []string, kinds []graph.EdgeKind) var out []graph.RepoEdgeRow for rows.Next() { var repoPrefix string - edge, err := scanEdgeCursor(repoPrefixedEdgeScanner{scanner: rows, repo: &repoPrefix}) + edge, err := s.scanEdgeCursor(repoPrefixedEdgeScanner{scanner: rows, repo: &repoPrefix}) if err != nil { panicOnFatal(err) return out diff --git a/internal/graph/store_sqlite/scan_error_surfacing_test.go b/internal/graph/store_sqlite/scan_error_surfacing_test.go index bc4ad17f8..fa3fe8eea 100644 --- a/internal/graph/store_sqlite/scan_error_surfacing_test.go +++ b/internal/graph/store_sqlite/scan_error_surfacing_test.go @@ -77,8 +77,8 @@ func prepareOnReader(t *testing.T, store *Store, q string) *sql.Stmt { // TestEdgeScansSurfaceRowIterationErrors pins the contract that a driver error // raised part-way through an edge cursor is reported rather than silently -// shortening the returned slice. Both edge scanners take a prepared statement, -// so the failure is injected through the statement's own SQL. +// shortening the returned slice. The scanner takes a prepared statement, so +// the failure is injected through the statement's own SQL. func TestEdgeScansSurfaceRowIterationErrors(t *testing.T) { store := newScanErrorStore(t) @@ -86,16 +86,9 @@ func TestEdgeScansSurfaceRowIterationErrors(t *testing.T) { if got := len(store.queryEdges(healthy)); got != 3 { t.Fatalf("healthy edge scan returned %d edges, want 3", got) } - healthyLight := prepareOnReader(t, store, `SELECT `+edgeColsLight+` FROM edges`) - if got := len(store.queryEdgesLight(healthyLight)); got != 3 { - t.Fatalf("healthy light edge scan returned %d edges, want 3", got) - } - poisoned := prepareOnReader(t, store, `SELECT `+lookupEdgeCols+` FROM edges`+poisonAfterFirstEdge) assertScanFailureSurfaces(t, "queryEdges", func() { store.queryEdges(poisoned) }) - poisonedLight := prepareOnReader(t, store, `SELECT `+edgeColsLight+` FROM edges`+poisonAfterFirstEdge) - assertScanFailureSurfaces(t, "queryEdgesLight", func() { store.queryEdgesLight(poisonedLight) }) } // TestInlineSQLScansSurfaceQueryAndRowErrors covers the raw-SQL siblings used diff --git a/internal/graph/store_sqlite/schema.go b/internal/graph/store_sqlite/schema.go index ecaa1d857..34b856178 100644 --- a/internal/graph/store_sqlite/schema.go +++ b/internal/graph/store_sqlite/schema.go @@ -421,6 +421,18 @@ CREATE TABLE IF NOT EXISTS analysis_blobs ( ) WITHOUT ROWID; ` +const vectorTableSQL = ` +CREATE TABLE IF NOT EXISTS vectors ( + node_id TEXT PRIMARY KEY, + repo_prefix TEXT NOT NULL DEFAULT '', + parent_id TEXT NOT NULL DEFAULT '', + dims INTEGER NOT NULL, + vec BLOB NOT NULL +) WITHOUT ROWID; +` + +const vectorRepoIndexSQL = `CREATE INDEX IF NOT EXISTS vectors_by_repo ON vectors(repo_prefix, node_id)` + // schemaSQL is the canonical DDL applied on Open. Statements are // idempotent (IF NOT EXISTS) so they run cleanly against a fresh DB // and against an existing one. @@ -689,11 +701,7 @@ CREATE INDEX IF NOT EXISTS ref_facts_by_file ON ref_facts(repo_prefix, file_path -- ref_facts scan — the PK leads with from_id, not to_id. CREATE INDEX IF NOT EXISTS ref_facts_by_target ON ref_facts(repo_prefix, to_id); -CREATE TABLE IF NOT EXISTS vectors ( - node_id TEXT PRIMARY KEY, - dims INTEGER NOT NULL, - vec BLOB NOT NULL -) WITHOUT ROWID; +` + vectorTableSQL + ` -- churn_enrichment is the per-node git-churn sidecar (change A: move -- enrichment OUT of nodes.meta so the node hot path stops encoding @@ -747,7 +755,7 @@ CREATE TABLE IF NOT EXISTS blame_enrichment ( CREATE INDEX IF NOT EXISTS blame_by_repo ON blame_enrichment(repo_prefix) WHERE repo_prefix <> ''; -- symbol_fts is the FTS5 full-text index over pre-tokenised symbol --- names. It replaces the multi-GB in-heap Bleve/BM25 index with an +-- names. It replaces the multi-GB in-heap BM25 index with an -- on-disk inverted index the SymbolSearcher / SymbolBundleSearcher -- query through. A standard (NOT contentless) FTS5 table; individual -- rows are deleted by their FTS5 docid via the symbol_fts_rowid sidecar diff --git a/internal/graph/store_sqlite/schema_version.go b/internal/graph/store_sqlite/schema_version.go index 4c5d043e5..9163a632c 100644 --- a/internal/graph/store_sqlite/schema_version.go +++ b/internal/graph/store_sqlite/schema_version.go @@ -32,7 +32,7 @@ import ( // index changes in a way an old on-disk DB would not already have, and append a // matching schemaMigrations entry describing how to bring an older store // forward (in place, or by rebuild). -const currentSchemaVersion = 9 +const currentSchemaVersion = 10 // schemaMigration is one forward step. Exactly one strategy applies: // - rebuild=true: the change introduces structure/data that can only come @@ -74,6 +74,10 @@ var schemaMigrations = []schemaMigration{ {version: 7, name: "purge unprefixed solo-repo rows", inPlace: purgeUnprefixedRepoRows}, {version: 8, name: "allow duplicate qualified names", inPlace: relaxNodeQualNameUniqueness}, {version: 9, name: "drop unused semantic pending index", inPlace: dropUnusedSemanticPendingIndex}, + // Vector ownership and chunk-parent identity cannot be reconstructed from + // the legacy (node_id, dims, vec) rows for every ID shape. Rebuild only the + // derived vector sidecar rather than discarding otherwise-valid topology. + {version: 10, name: "rebuild vector corpus ownership and parents", inPlace: rebuildVectorCorpusSchema}, } // dropUnusedSemanticPendingIndex removes an experimental index for a query @@ -84,6 +88,19 @@ func dropUnusedSemanticPendingIndex(tx *sql.Tx) error { return err } +// rebuildVectorCorpusSchema intentionally discards only the durable vector +// sidecar. Legacy rows do not carry repository ownership or chunk parents, so +// retaining them would make per-repository replacement and warm de-chunking +// unsound. The graph topology remains intact and the next embedding pass +// repopulates this derived cache. +func rebuildVectorCorpusSchema(tx *sql.Tx) error { + if _, err := tx.Exec(`DROP TABLE IF EXISTS vectors`); err != nil { + return err + } + _, err := tx.Exec(vectorTableSQL) + return err +} + // relaxNodeQualNameUniqueness removes the historical assumption that a // language-level qualified name is a global graph identity. Resource manifests, // forks, worktrees and overload-like constructs may legitimately repeat one. diff --git a/internal/graph/store_sqlite/scoped_projection.go b/internal/graph/store_sqlite/scoped_projection.go index 2ad3cbff8..b4d1fb510 100644 --- a/internal/graph/store_sqlite/scoped_projection.go +++ b/internal/graph/store_sqlite/scoped_projection.go @@ -190,7 +190,7 @@ func (s *Store) streamScopedEdges(query string, args []any, maxID int64, yield f page := make([]*graph.Edge, 0, scopedProjectionPage) for rows.Next() { var edgeID int64 - edge, scanErr := scanEdgeCursor(edgeIDScanner{scanner: rows, id: &edgeID}) + edge, scanErr := s.scanEdgeCursor(edgeIDScanner{scanner: rows, id: &edgeID}) if scanErr != nil { _ = rows.Close() panicOnFatal(scanErr) diff --git a/internal/graph/store_sqlite/sqlite_busy.go b/internal/graph/store_sqlite/sqlite_busy.go index bde0d0415..50c0bbdec 100644 --- a/internal/graph/store_sqlite/sqlite_busy.go +++ b/internal/graph/store_sqlite/sqlite_busy.go @@ -27,21 +27,6 @@ const ( sqliteBusyRetryMaxDelay = 250 * time.Millisecond ) -// SQLiteBusyRetryStats is a monotonic process-local view of transaction-level -// lock contention. Retries counts whole transaction replays; Exhausted counts -// operations that still failed after the bounded retry window. -type SQLiteBusyRetryStats struct { - Retries uint64 - Exhausted uint64 -} - -func (s *Store) BusyRetryStats() SQLiteBusyRetryStats { - return SQLiteBusyRetryStats{ - Retries: s.busyRetries.Load(), - Exhausted: s.busyRetryExhausted.Load(), - } -} - // isSQLiteBusyErr matches both primary and extended BUSY/LOCKED result codes. // Extended codes keep the primary result in the low byte. func isSQLiteBusyErr(err error) bool { @@ -107,13 +92,11 @@ func (s *Store) withSQLiteBusyRetry( remaining := time.Until(retryDeadline) if remaining <= 0 { - s.busyRetryExhausted.Add(1) log.Printf("store_sqlite: sqlite busy exhausted operation=%s retries=%d elapsed=%s error=%q", operation, retries, time.Since(started), lastBusy) return fmt.Errorf("%s: %w", operation, errors.Join(errSQLiteBusyRetryExhausted, lastBusy, context.DeadlineExceeded)) } retries++ - s.busyRetries.Add(1) wait := minDuration(delay, remaining) timer := time.NewTimer(wait) select { @@ -122,7 +105,6 @@ func (s *Store) withSQLiteBusyRetry( if !timer.Stop() { <-timer.C } - s.busyRetryExhausted.Add(1) log.Printf("store_sqlite: sqlite busy exhausted operation=%s retries=%d elapsed=%s error=%q", operation, retries, time.Since(started), lastBusy) return fmt.Errorf("%s: %w", operation, errors.Join(errSQLiteBusyRetryExhausted, lastBusy, parent.Err())) } diff --git a/internal/graph/store_sqlite/sqlite_busy_test.go b/internal/graph/store_sqlite/sqlite_busy_test.go index fb3446961..039b4ccdf 100644 --- a/internal/graph/store_sqlite/sqlite_busy_test.go +++ b/internal/graph/store_sqlite/sqlite_busy_test.go @@ -137,16 +137,11 @@ func TestReindexEdgesRetriesWholeTransactionAfterBusy(t *testing.T) { _, err := store.reindexEdgesSetOriented(batch) result <- err }() - require.Eventually(t, func() bool { - return store.BusyRetryStats().Retries > 0 - }, 2*time.Second, 2*time.Millisecond, "reindex should observe and retry SQLITE_BUSY") + requireBusyOperationBlocked(t, result, "reindex") require.NoError(t, lockTx.Rollback()) require.NoError(t, locker.Close()) require.NoError(t, <-result) - stats := store.BusyRetryStats() - assert.Greater(t, stats.Retries, uint64(0)) - assert.Zero(t, stats.Exhausted) requireReindexedTarget(t, store, "repo/target.go::Target") require.NoError(t, store.Close()) @@ -181,15 +176,12 @@ func TestReceiverRebindRetriesBusyBeginOnPinnedWriter(t *testing.T) { changed, rebindErr := store.RebindGoMethodReceivers("") done <- result{changed: changed, err: rebindErr} }() - require.Eventually(t, func() bool { - return store.BusyRetryStats().Retries > 0 - }, 2*time.Second, 2*time.Millisecond, "receiver rebind should retry a busy IMMEDIATE begin") + requireBusyOperationBlocked(t, done, "receiver rebind") require.NoError(t, lockTx.Rollback()) require.NoError(t, locker.Close()) got := <-done require.NoError(t, got.err) assert.Equal(t, 1, got.changed) - assert.Zero(t, store.BusyRetryStats().Exhausted) } func TestReindexEdgesPersistentBusySurfacesAndRollsBack(t *testing.T) { @@ -200,12 +192,14 @@ func TestReindexEdgesPersistentBusySurfacesAndRollsBack(t *testing.T) { store.busyRetryTimeout = 60 * time.Millisecond locker, lockTx := holdExternalWriter(t, path) + started := time.Now() _, err := store.reindexEdgesSetOriented(batch) + elapsed := time.Since(started) require.Error(t, err) assert.True(t, isSQLiteBusyErr(err), "the exhausted error must retain the SQLite result code") - stats := store.BusyRetryStats() - assert.Greater(t, stats.Retries, uint64(0)) - assert.Equal(t, uint64(1), stats.Exhausted) + assert.ErrorIs(t, err, errSQLiteBusyRetryExhausted) + assert.GreaterOrEqual(t, elapsed, 50*time.Millisecond) + assert.Less(t, elapsed, time.Second) requireReindexedTarget(t, store, "unresolved::Target") require.NoError(t, lockTx.Rollback()) @@ -245,7 +239,6 @@ func TestLongWALReaderDoesNotBlockReindexWriter(t *testing.T) { _, err := store.reindexEdgesSetOriented(batch) require.NoError(t, err) assert.Less(t, time.Since(started), time.Second) - assert.Zero(t, store.BusyRetryStats().Retries) for _, reader := range held { require.NoError(t, reader.rows.Close()) @@ -453,11 +446,9 @@ func TestCheckpointBusyResultIsRetriedAndNeverReportedAsSuccess(t *testing.T) { require.NoError(t, readTx.QueryRow(`SELECT COUNT(*) FROM nodes`).Scan(&count)) store.AddNode(&graph.Node{ID: "repo/b.go::B", Kind: graph.KindFunction, Name: "B"}) - beforePassive := store.BusyRetryStats() started := time.Now() store.checkpointWALPassive() assert.Less(t, time.Since(started), time.Second, "PASSIVE maintenance must not wait for the long reader") - assert.Equal(t, beforePassive, store.BusyRetryStats(), "periodic PASSIVE checkpoint must be one-shot") setWriterBusyTimeout(t, store, 0) store.busyRetryTimeout = 80 * time.Millisecond @@ -465,11 +456,12 @@ func TestCheckpointBusyResultIsRetriedAndNeverReportedAsSuccess(t *testing.T) { started = time.Now() err = store.checkpointWALWithContext(ctx) cancel() + elapsed := time.Since(started) require.Error(t, err) assert.ErrorIs(t, err, errSQLiteCheckpointIncomplete) assert.ErrorIs(t, err, errSQLiteBusyRetryExhausted) - assert.Less(t, time.Since(started), time.Second) - assert.Greater(t, store.BusyRetryStats().Retries, beforePassive.Retries) + assert.GreaterOrEqual(t, elapsed, 60*time.Millisecond) + assert.Less(t, elapsed, time.Second) require.NoError(t, readTx.Rollback()) require.NoError(t, readConn.Close()) @@ -551,9 +543,7 @@ func TestEvictFileRetriesBusyBeginAndCommitsAtomically(t *testing.T) { nodes, edges := store.EvictFile(evictFixtureFile) done <- [2]int{nodes, edges} }() - require.Eventually(t, func() bool { - return store.BusyRetryStats().Retries > 0 - }, 2*time.Second, 2*time.Millisecond, "file eviction should retry its IMMEDIATE begin") + requireBusyOperationBlocked(t, done, "file eviction") require.NoError(t, lockTx.Rollback()) require.NoError(t, locker.Close()) released = true @@ -566,7 +556,6 @@ func TestEvictFileRetriesBusyBeginAndCommitsAtomically(t *testing.T) { } require.Equal(t, [2]int{2, 3}, got) requireAtomicFileEvictionCommitted(t, store) - require.Greater(t, store.BusyRetryStats().Retries, uint64(0)) require.NoError(t, store.Close()) reopened, err := Open(path) @@ -737,6 +726,15 @@ func openBusyReindexFixture(t *testing.T, path string) (*Store, []graph.EdgeRein return store, []graph.EdgeReindex{{Edge: &resolved, OldTo: old.To}} } +func requireBusyOperationBlocked[T any](t *testing.T, result <-chan T, operation string) { + t.Helper() + select { + case value := <-result: + t.Fatalf("%s completed while the external writer lock was held: %v", operation, value) + case <-time.After(50 * time.Millisecond): + } +} + func setWriterBusyTimeout(t *testing.T, store *Store, milliseconds int) { t.Helper() conn, err := store.writerDB.Conn(context.Background()) diff --git a/internal/graph/store_sqlite/store.go b/internal/graph/store_sqlite/store.go index 342d97c6c..d89ccad47 100644 --- a/internal/graph/store_sqlite/store.go +++ b/internal/graph/store_sqlite/store.go @@ -51,9 +51,7 @@ type Store struct { // busyRetryTimeout is the whole-transaction contention budget. The zero // value selects defaultSQLiteBusyRetryTimeout; tests shorten it to exercise // persistent-lock exhaustion deterministically. - busyRetryTimeout time.Duration - busyRetries atomic.Uint64 - busyRetryExhausted atomic.Uint64 + busyRetryTimeout time.Duration // passiveCheckpointTimeout bounds one periodic PASSIVE checkpoint. The zero // value selects walPassiveCheckpointTimeout; tests shorten it to exercise @@ -70,6 +68,13 @@ type Store struct { // re-indexes don't re-upsert identical stubs on every batch. builtinSeen sync.Map + // Structural integrity is owned by this logical store. Shadows forward + // rejected attempts into the same recorder; warnings are rate-limited per + // Store so independent workspaces never suppress each other's diagnostics. + structuralIntegrity graph.StructuralIntegrityMeter + structuralWriteWarned atomic.Bool + structuralReadWarned atomic.Bool + // preparedSQL registers every statement prepared at Open so the plan // fence can EXPLAIN the entire prepared surface against a fixture and // reject big-table scans mechanically. @@ -205,7 +210,6 @@ type Store struct { stmtInsertEdge *sql.Stmt unresolvedInserts atomic.Uint64 stmtOutEdges *sql.Stmt - stmtOutEdgesLight *sql.Stmt stmtInEdges *sql.Stmt stmtRepoEdges *sql.Stmt stmtAllEdges *sql.Stmt @@ -482,6 +486,13 @@ func openWith(path string, current int, migrations []schemaMigration, allowRebui return nil, fmt.Errorf("sqlite stamp schema version: %w", err) } } + // The repository index references columns introduced by the v10 vector + // migration, so create it only after pending migrations have rebuilt the + // legacy table. On current and fresh stores this is an idempotent no-op. + if _, err := db.Exec(vectorRepoIndexSQL); err != nil { + _ = db.Close() + return nil, fmt.Errorf("sqlite vector repository index: %w", err) + } // A schema transition invalidates any generation produced against the old // graph shape. The v4 migration also drops the unreleased blob-only table. if stored != current { @@ -780,7 +791,7 @@ func (s *Store) Close() error { s.stmtAllRepoCountsNodes, s.stmtAllRepoCountsEdges, s.stmtAllRepoStateCounts, s.stmtStatsByKind, s.stmtStatsByLanguage, - s.stmtInsertEdge, s.stmtOutEdges, s.stmtOutEdgesLight, s.stmtInEdges, + s.stmtInsertEdge, s.stmtOutEdges, s.stmtInEdges, s.stmtRepoEdges, s.stmtAllEdges, s.stmtEdgeCount, s.stmtRemoveEdge, s.stmtUpdateEdgeOrigin, s.stmtUpdateEdgeAttrs, s.stmtSelectEdgeOrigin, s.stmtDeleteEdgeByKey, @@ -912,12 +923,6 @@ func (s *Store) prepare() error { // b-tree. prep(&s.stmtOutEdges, `SELECT `+edgeCols+` FROM edges WHERE from_id = ? ORDER BY line, id`) - // edgeColsLight is the package-level meta-less projection (store_light_edges.go), - // shared with AllEdgesLight so this prepared statement and the whole-graph scan - // can never drift apart. The ordering must match stmtOutEdges for the same - // reason: callers switch between the two purely to skip the Meta blob. - prep(&s.stmtOutEdgesLight, - `SELECT `+edgeColsLight+` FROM edges WHERE from_id = ? ORDER BY line, id`) prep(&s.stmtInEdges, `SELECT `+edgeCols+` FROM edges WHERE to_id = ? ORDER BY kind, id`) prep(&s.stmtRepoEdges, @@ -1056,12 +1061,12 @@ func scanNodeSummary(scanner interface { // scanEdgeCursor is cursor-only: metadata is decoded before Rows.Next, so the // driver-owned RawBytes never escapes into the returned edge. -func scanEdgeCursor(scanner rowScanner) (*graph.Edge, error) { +func (s *Store) scanEdgeCursor(scanner rowScanner) (*graph.Edge, error) { var metaBlob sql.RawBytes - return scanEdgeWithMeta(scanner, &metaBlob) + return scanEdgeWithMeta(s, scanner, &metaBlob) } -func scanEdgeWithMeta[B ~[]byte](scanner rowScanner, metaBlob *B) (*graph.Edge, error) { +func scanEdgeWithMeta[B ~[]byte](store *Store, scanner rowScanner, metaBlob *B) (*graph.Edge, error) { var ( e graph.Edge crossRepo int64 @@ -1088,43 +1093,24 @@ func scanEdgeWithMeta[B ~[]byte](scanner rowScanner, metaBlob *B) (*graph.Edge, // is left alone so any blob-carried value survives. restorePromotedEdgeMeta(&e, p) if graph.StructuralEdgeTargetInvalid(e.Kind, e.To) { - noteStructuralReadDrop() + store.noteStructuralReadDrop(graph.StructuralPathSQLiteFullRead, &e) return nil, nil } return &e, nil } -// structuralReadDrops counts structurally invalid rows healed on read from -// stores written before the write-funnel backstop existed. Every read path -// dropping such a row means the on-disk store carries pre-gate corruption: -// the first occurrence logs an engineer-facing signal (the feedback loop for -// "something impossible reached disk"), and the audit battery reads the -// counter. New stores must never increment it — the write backstop drops the -// shape before it lands. -var ( - structuralReadDrops atomic.Int64 - structuralReadDropsOnce sync.Once -) - -func noteStructuralReadDrop() { - structuralReadDrops.Add(1) - structuralReadDropsOnce.Do(func() { - log.Printf("store_sqlite: store contains structurally invalid edges (pre-backstop corruption); healing on read — rebuild or audit the store (see store_audit.sql A1)") - }) -} - -// StructuralReadDrops reports how many structurally invalid edge rows read -// paths have healed since process start. -func StructuralReadDrops() int64 { - return structuralReadDrops.Load() -} - // scanEdgeLight scans an edge WITHOUT decoding its meta blob -- for hot // read paths (dataflow call-target lookup) that read only endpoints, // kind, and line. Skipping the meta column avoids the JSON decode + map // allocation that dominates large edge scans on this backend; the // returned edge's Meta is nil. -func scanEdgeLight(scanner interface { +func (s *Store) scanEdgeLight(scanner interface { + Scan(...any) error +}) (*graph.Edge, error) { + return scanEdgeLightForStore(s, scanner) +} + +func scanEdgeLightForStore(store *Store, scanner interface { Scan(...any) error }) (*graph.Edge, error) { var ( @@ -1141,7 +1127,7 @@ func scanEdgeLight(scanner interface { } e.CrossRepo = crossRepo != 0 if graph.StructuralEdgeTargetInvalid(e.Kind, e.To) { - noteStructuralReadDrop() + store.noteStructuralReadDrop(graph.StructuralPathSQLiteLightRead, &e) return nil, nil } return &e, nil @@ -1205,6 +1191,10 @@ func (s *Store) AddEdge(e *graph.Edge) { if e == nil || graph.IsProxyID(e.From) || graph.IsProxyID(e.To) { return } + if graph.StructuralEdgeTargetInvalid(e.Kind, e.To) { + s.recordStructuralEdge(graph.StructuralDropWrite, graph.StructuralPathSQLiteAddEdge, s.structuralWriteRepo(e, nil), e) + return + } // Route through the set-oriented writer. During a coordinated cold load the // single writer connection is pinned; using a prepared statement through // database/sql here would wait for a second writer slot. AddBatch reuses the @@ -1908,13 +1898,6 @@ func (s *Store) EdgeExists(from, to string, kind graph.EdgeKind, filePath string return true } -// GetOutEdgesLight returns a node's out-edges without decoding the -// per-edge Meta blob -- for hot dataflow lookups that need only -// endpoints/kind/line. The returned edges have a nil Meta. -func (s *Store) GetOutEdgesLight(nodeID string) []*graph.Edge { - return s.queryEdgesLight(s.stmtOutEdgesLight, nodeID) -} - func (s *Store) GetInEdges(nodeID string) []*graph.Edge { return s.queryEdges(s.stmtInEdges, nodeID) } @@ -1985,7 +1968,7 @@ func (s *Store) queryEdges(stmt *sql.Stmt, args ...any) []*graph.Edge { defer rows.Close() var out []*graph.Edge for rows.Next() { - e, err := scanEdgeCursor(rows) + e, err := s.scanEdgeCursor(rows) if err != nil { panicOnFatal(err) return out @@ -2004,34 +1987,6 @@ func (s *Store) queryEdges(stmt *sql.Stmt, args ...any) []*graph.Edge { return out } -// queryEdgesLight mirrors queryEdges but scans each row without its -// meta blob (scanEdgeLight), leaving Meta nil. Only for callers that -// never read edge Meta. -func (s *Store) queryEdgesLight(stmt *sql.Stmt, args ...any) []*graph.Edge { - rows, err := stmt.Query(args...) - if err != nil { - panicOnFatal(err) - return nil - } - defer rows.Close() - var out []*graph.Edge - for rows.Next() { - e, err := scanEdgeLight(rows) - if err != nil { - panicOnFatal(err) - return out - } - if e == nil { - continue - } - out = append(out, e) - } - if err := rows.Err(); err != nil { - panicOnFatal(err) - } - return out -} - // -- counts and stats ----------------------------------------------------- func (s *Store) NodeCount() int { @@ -2545,7 +2500,7 @@ func (s *Store) queryEdgesSQL(q string, args ...any) []*graph.Edge { defer rows.Close() var out []*graph.Edge for rows.Next() { - e, err := scanEdgeCursor(rows) + e, err := s.scanEdgeCursor(rows) if err != nil { panicOnFatal(err) return out diff --git a/internal/graph/store_sqlite/store_fts.go b/internal/graph/store_sqlite/store_fts.go index dd472eba4..3853d744e 100644 --- a/internal/graph/store_sqlite/store_fts.go +++ b/internal/graph/store_sqlite/store_fts.go @@ -13,7 +13,7 @@ import ( // This file implements graph.SymbolSearcher + graph.SymbolBundleSearcher // on the SQLite backend using the FTS5 virtual table declared in // schema.go (symbol_fts). It is the on-disk replacement for the -// multi-GB in-heap Bleve/BM25 index: the FTS5 inverted index lives in +// multi-GB in-heap BM25 index: the FTS5 inverted index lives in // the same .sqlite file as the graph, and a tier-0 exact-name boost // short-circuits identifier queries so // search quality holds or improves while the heap shrinks. diff --git a/internal/graph/store_sqlite/store_light_edges.go b/internal/graph/store_sqlite/store_light_edges.go index 9bb3162b7..f76593ebe 100644 --- a/internal/graph/store_sqlite/store_light_edges.go +++ b/internal/graph/store_sqlite/store_light_edges.go @@ -7,10 +7,10 @@ var _ graph.LightEdgeScanner = (*Store)(nil) // edgeColsLight is the meta-less edge column projection: the promoted struct // columns WITHOUT the meta blob (and without resolve_terminal, which lives in -// Meta). It is exactly the ten columns scanEdgeLight scans, and is shared with -// the stmtOutEdgesLight prepared statement so the projection can never drift -// from the scanner. Adding meta back here would defeat the whole point — the -// per-row JSON decode this projection exists to skip. +// Meta). It is exactly the ten columns scanEdgeLight scans and is shared by +// every meta-less SQLite edge query so projections cannot drift from the +// scanner. Adding meta back here would defeat the whole point — the per-row +// JSON decode this projection exists to skip. const edgeColsLight = `from_id, to_id, kind, file_path, line, confidence, confidence_label, origin, tier, cross_repo` // AllEdgesLight implements graph.LightEdgeScanner: a kind-scoped edge scan that @@ -47,7 +47,7 @@ func (s *Store) queryEdgesLightSQL(q string, args ...any) []*graph.Edge { defer rows.Close() var out []*graph.Edge for rows.Next() { - e, err := scanEdgeLight(rows) + e, err := s.scanEdgeLight(rows) if err != nil { panicOnFatal(err) return out diff --git a/internal/graph/store_sqlite/store_lookups.go b/internal/graph/store_sqlite/store_lookups.go index 41142ca35..2980a5ebf 100644 --- a/internal/graph/store_sqlite/store_lookups.go +++ b/internal/graph/store_sqlite/store_lookups.go @@ -282,7 +282,7 @@ func (s *Store) GetOutEdgesByNodeIDsContext(ctx context.Context, ids []string, l return out, true, err } for rows.Next() { - e, scanErr := scanEdgeLight(rows) + e, scanErr := s.scanEdgeLight(rows) if scanErr != nil { _ = rows.Close() return out, true, scanErr @@ -336,7 +336,7 @@ func (s *Store) GetInEdgesByNodeIDsContext(ctx context.Context, ids []string, li return out, true, err } for rows.Next() { - e, scanErr := scanEdgeLight(rows) + e, scanErr := s.scanEdgeLight(rows) if scanErr != nil { _ = rows.Close() return out, true, scanErr @@ -556,7 +556,7 @@ func (s *Store) queryEdgeCandidatesSQL(query string, args ...any) ([]*graph.Edge } var out []*graph.Edge for rows.Next() { - edge, scanErr := scanEdgeCursor(rows) + edge, scanErr := s.scanEdgeCursor(rows) if scanErr != nil { _ = rows.Close() return nil, scanErr diff --git a/internal/graph/store_sqlite/store_purge.go b/internal/graph/store_sqlite/store_purge.go index 78f47b421..fda669884 100644 --- a/internal/graph/store_sqlite/store_purge.go +++ b/internal/graph/store_sqlite/store_purge.go @@ -30,9 +30,8 @@ import ( // column a plain `DELETE ... WHERE repo_prefix = ?` keys on. The two FTS5 // vtables (symbol_fts, content_fts) carry repo_prefix UNINDEXED, so their // delete is a full scan — acceptable for a purge (a rare, whole-repo op), -// unlike the per-edit hot path. `vectors` is deliberately absent: it has NO -// repo_prefix column (keyed by node_id alone), so PurgeRepo deletes its rows -// by node-id membership instead (see deleteByIDColumnsTx below). +// unlike the per-edit hot path. Vectors are repo-keyed too; deleting them by +// repo_prefix is essential because synthetic chunk IDs are not graph node IDs. var purgeSidecarTables = []string{ "file_mtimes", "repo_index_state", @@ -44,6 +43,7 @@ var purgeSidecarTables = []string{ "constant_values", "files", "ref_facts", + "vectors", "churn_enrichment", "coverage_enrichment", "release_enrichment", @@ -92,11 +92,10 @@ func (s *Store) PurgeRepo(prefix string) error { } defer tx.Rollback() //nolint:errcheck // rollback after Commit is a no-op - // Collect this repo's node IDs first: edges and vectors are keyed off - // them (edges by from_id/to_id, vectors by node_id — neither carries a - // repo_prefix column). Edge deletion semantics mirror scope eviction's - // (store.go): delete every edge touching one of these nodes, then the - // nodes themselves. + // Collect this repo's node IDs first for edge deletion. Vectors are removed + // below by repo_prefix so synthetic chunk rows are covered too. Edge deletion + // semantics mirror scope eviction's (store.go): delete every edge touching + // one of these nodes, then the nodes themselves. ids, err := repoNodeIDsTx(tx, prefix) if err != nil { return err @@ -104,10 +103,6 @@ func (s *Store) PurgeRepo(prefix string) error { if err := deleteByIDColumnsTx(tx, "edges", []string{"from_id", "to_id"}, ids); err != nil { return fmt.Errorf("store_sqlite: PurgeRepo edges: %w", err) } - if err := deleteByIDColumnsTx(tx, "vectors", []string{"node_id"}, ids); err != nil { - return fmt.Errorf("store_sqlite: PurgeRepo vectors: %w", err) - } - changed := len(ids) > 0 for _, table := range purgeSidecarTables { res, err := tx.Exec(`DELETE FROM `+table+` WHERE repo_prefix = ?`, prefix) @@ -138,15 +133,12 @@ func (s *Store) PurgeRepo(prefix string) error { } // orphanScanTables are the tables OrphanRepoPrefixes unions DISTINCT -// repo_prefix over. These six span the residue space: nodes (the primary -// keyed store), file_mtimes + repo_index_state (the warm-restart provenance -// that lingers when nodes are gone but sidecars survive — the exact shape a -// leaked untrack leaves), enrichment_state (per-provider provenance), files -// (per-file metadata), and semantic_binding_types (compiler-derived contract -// bindings). A prefix whose nodes are gone but whose -// sidecars remain is invisible to a nodes-only scan, which is why the -// sidecar tables are unioned in; scanning still more tables would only -// rediscover the same prefixes at higher cost. +// repo_prefix over. They span the primary graph, warm-restart provenance, +// enrichment/file metadata, compiler-derived bindings, clone corpus state, +// and durable vectors. Vector coverage matters because a failed legacy +// untrack can leave only synthetic chunk rows after graph nodes are gone. +// A prefix whose nodes are gone but whose sidecars remain is invisible to a +// nodes-only scan, which is why the sidecar tables are unioned in. var orphanScanTables = []string{ "nodes", "file_mtimes", @@ -155,6 +147,7 @@ var orphanScanTables = []string{ "files", "semantic_binding_types", "clone_corpus_state", + "vectors", } // OrphanRepoPrefixes returns every repo_prefix present in the store but @@ -310,11 +303,17 @@ func (s *Store) RekeyRepoPrefix(oldPrefix, newPrefix string) error { changed = true } } - // vectors is intentionally omitted: it has NO repo_prefix column (keyed - // by node_id alone), so it cannot be addressed here by prefix. Any '' - // embeddings are node_id-keyed against now-evicted unprefixed ids — - // dangling, and absent in the common case (embeddings are opt-in). They - // are left to a node-membership vector GC rather than guessed at here. + // Vectors are handled explicitly instead of joining rekeyDropTables because + // that shared list is also used by the historical v6→v7 migration, whose + // vector schema predates repo_prefix. At the current schema the old node and + // parent IDs cannot be relabeled safely, so drop the complete old corpus. + res, err := tx.Exec(`DELETE FROM vectors WHERE repo_prefix = ?`, oldPrefix) + if err != nil { + return fmt.Errorf("store_sqlite: RekeyRepoPrefix drop vectors: %w", err) + } + if n, rowsErr := res.RowsAffected(); rowsErr == nil && n > 0 { + changed = true + } if err := tx.Commit(); err != nil { return err diff --git a/internal/graph/store_sqlite/store_purge_test.go b/internal/graph/store_sqlite/store_purge_test.go index ccd6975c6..7a412e97a 100644 --- a/internal/graph/store_sqlite/store_purge_test.go +++ b/internal/graph/store_sqlite/store_purge_test.go @@ -28,6 +28,7 @@ func openPurgeStore(t *testing.T) *Store { func seedRepoRows(t *testing.T, db *sql.DB, prefix string) { t.Helper() nodeID := prefix + "::a.go::X" + chunkID := nodeID + "#chunk0" exec := func(q string, args ...any) { t.Helper() _, err := db.Exec(q, args...) @@ -38,7 +39,8 @@ func seedRepoRows(t *testing.T, db *sql.DB, prefix string) { // must delete this edge (its from_id is a repo node) but NEVER the '' // target node. exec(`INSERT INTO edges (from_id, to_id, kind) VALUES (?, 'external_call::dep:shared', 'calls')`, nodeID) - exec(`INSERT INTO vectors (node_id, dims, vec) VALUES (?, 1, X'00')`, nodeID) + exec(`INSERT INTO vectors (node_id, repo_prefix, parent_id, dims, vec) VALUES (?, ?, '', 1, X'00')`, nodeID, prefix) + exec(`INSERT INTO vectors (node_id, repo_prefix, parent_id, dims, vec) VALUES (?, ?, ?, 1, X'00')`, chunkID, prefix, nodeID) exec(`INSERT INTO file_mtimes (repo_prefix, file_path, mtime_ns) VALUES (?, 'a.go', 123)`, prefix) exec(`INSERT INTO repo_index_state (repo_prefix, indexed_sha) VALUES (?, 'sha')`, prefix) @@ -70,15 +72,6 @@ func countByPrefix(t *testing.T, db *sql.DB, table, prefix string) int { return n } -// countByNodeIDLike reports how many rows a node_id-keyed table (vectors) -// holds whose node_id starts with `::`. -func countByNodeIDLike(t *testing.T, db *sql.DB, table, prefix string) int { - t.Helper() - var n int - require.NoError(t, db.QueryRow(`SELECT COUNT(*) FROM `+table+` WHERE node_id LIKE ?`, prefix+"::%").Scan(&n)) - return n -} - // prefixKeyedTables is every repo_prefix-keyed table PurgeRepo/Rekey touch, // minus nodes (asserted separately) — used to loop assertions. var prefixKeyedTables = []string{ @@ -101,7 +94,7 @@ func TestPurgeRepo_ClearsEveryTable_LeavesOthersAndGlobals(t *testing.T) { // repoA: nodes, edges, vectors, and every sidecar cleared. assert.Equal(t, 0, countByPrefix(t, s.db, "nodes", "repoA"), "repoA nodes gone") - assert.Equal(t, 0, countByNodeIDLike(t, s.db, "vectors", "repoA"), "repoA vectors gone") + assert.Equal(t, 0, countByPrefix(t, s.db, "vectors", "repoA"), "repoA ordinary and chunk vectors gone") for _, tbl := range prefixKeyedTables { assert.Equal(t, 0, countByPrefix(t, s.db, tbl, "repoA"), "repoA %s cleared", tbl) } @@ -111,7 +104,7 @@ func TestPurgeRepo_ClearsEveryTable_LeavesOthersAndGlobals(t *testing.T) { // repoB untouched across the board. assert.Equal(t, 1, countByPrefix(t, s.db, "nodes", "repoB"), "repoB nodes intact") - assert.Equal(t, 1, countByNodeIDLike(t, s.db, "vectors", "repoB"), "repoB vectors intact") + assert.Equal(t, 2, countByPrefix(t, s.db, "vectors", "repoB"), "repoB ordinary and chunk vectors intact") for _, tbl := range prefixKeyedTables { assert.Equal(t, 1, countByPrefix(t, s.db, tbl, "repoB"), "repoB %s intact", tbl) } @@ -137,17 +130,23 @@ func TestOrphanRepoPrefixes_SidecarOnlyResidue(t *testing.T) { require.NoError(t, err) _, err = s.writerDB.Exec(`INSERT INTO repo_index_state (repo_prefix) VALUES ('gone')`) require.NoError(t, err) + // A vector-only legacy leak must also be discoverable even when every graph + // node and other sidecar for the repository is gone. + _, err = s.writerDB.Exec(`INSERT INTO vectors (node_id, repo_prefix, parent_id, dims, vec) VALUES ('vector-only::ghost#chunk0', 'vector-only', 'vector-only::ghost', 1, X'00')`) + require.NoError(t, err) seedRepoRows(t, s.writerDB, "live") // A '' row must never be reported as an orphan. _, err = s.writerDB.Exec(`INSERT INTO file_mtimes (repo_prefix, file_path, mtime_ns) VALUES ('', 'g.go', 1)`) require.NoError(t, err) orphans := s.OrphanRepoPrefixes([]string{"live"}) - assert.Equal(t, []string{"gone"}, orphans, "only the nodes-less residue prefix is an orphan") + assert.ElementsMatch(t, []string{"gone", "vector-only"}, orphans, + "both sidecar-only and vector-only residue must be reported") // Case-fold safety net: a case-only spelling drift of a tracked repo is // NOT an orphan. - assert.Empty(t, s.OrphanRepoPrefixes([]string{"LIVE", "GONE"}), "case-insensitive known set covers both prefixes") + assert.Empty(t, s.OrphanRepoPrefixes([]string{"LIVE", "GONE", "VECTOR-ONLY"}), + "case-insensitive known set covers every prefix") } func TestRekeyRepoPrefix_MovesProvenanceDropsNodeIDKeyed(t *testing.T) { @@ -175,6 +174,8 @@ func TestRekeyRepoPrefix_MovesProvenanceDropsNodeIDKeyed(t *testing.T) { assert.Equal(t, 0, countByPrefix(t, s.db, tbl, ""), "%s '' rows dropped", tbl) assert.Equal(t, 0, countByPrefix(t, s.db, tbl, "drools"), "%s NOT relabeled to new prefix", tbl) } + assert.Equal(t, 0, countByPrefix(t, s.db, "vectors", ""), "old-prefix vectors dropped") + assert.Equal(t, 0, countByPrefix(t, s.db, "vectors", "drools"), "vectors are not relabeled to new IDs") assert.Error(t, s.RekeyRepoPrefix("repoA", ""), "rekey INTO the empty prefix is refused") } diff --git a/internal/graph/store_sqlite/store_vector.go b/internal/graph/store_sqlite/store_vector.go index d8691db50..b2b49df86 100644 --- a/internal/graph/store_sqlite/store_vector.go +++ b/internal/graph/store_sqlite/store_vector.go @@ -39,10 +39,10 @@ var errInvalidDims = errors.New("store_sqlite: invalid vector dims") // index structure to build, since SimilarTo computes over the table // directly. -// vectorChunk bounds rows per multi-row INSERT in BulkUpsertEmbeddings. -// 3 host params per row, SQLite's default limit is 999 → 333 max; 300 -// leaves headroom. -const vectorChunk = 300 +// vectorChunk bounds rows per multi-row INSERT in BulkUpsertEmbeddings and +// ReplaceVectorCorpus. Five host parameters per row, SQLite's conservative +// default limit is 999; 180 leaves headroom. +const vectorChunk = 180 // encodeVec serialises a float32 slice to a little-endian BLOB // (4 bytes per element). @@ -73,8 +73,8 @@ func (s *Store) UpsertEmbedding(nodeID string, vec []float32) error { s.writeMu.Lock() defer s.writeMu.Unlock() _, err := s.execActiveWriteLocked(context.Background(), - `INSERT OR REPLACE INTO vectors (node_id, dims, vec) VALUES (?, ?, ?)`, - nodeID, len(vec), encodeVec(vec), + `INSERT OR REPLACE INTO vectors (node_id, repo_prefix, parent_id, dims, vec) VALUES (?, ?, '', ?, ?)`, + nodeID, graph.RepoPrefixOfID(nodeID), len(vec), encodeVec(vec), ) return err } @@ -103,15 +103,15 @@ func (s *Store) BulkUpsertEmbeddings(items []graph.VectorItem) error { } batch := items[start:end] - args := make([]any, 0, len(batch)*3) - stmt := make([]byte, 0, 64+len(batch)*16) - stmt = append(stmt, "INSERT OR REPLACE INTO vectors (node_id, dims, vec) VALUES "...) + args := make([]any, 0, len(batch)*4) + stmt := make([]byte, 0, 96+len(batch)*20) + stmt = append(stmt, "INSERT OR REPLACE INTO vectors (node_id, repo_prefix, parent_id, dims, vec) VALUES "...) for i, it := range batch { if i > 0 { stmt = append(stmt, ',') } - stmt = append(stmt, "(?, ?, ?)"...) - args = append(args, it.NodeID, len(it.Vec), encodeVec(it.Vec)) + stmt = append(stmt, "(?, ?, '', ?, ?)"...) + args = append(args, it.NodeID, graph.RepoPrefixOfID(it.NodeID), len(it.Vec), encodeVec(it.Vec)) } if _, err := tx.Exec(string(stmt), args...); err != nil { return err @@ -209,7 +209,7 @@ func (s *Store) SimilarTo(vec []float32, limit int) ([]graph.VectorHit, error) { return nil, nil } - rows, err := s.db.Query(`SELECT node_id, vec FROM vectors`) + rows, err := s.db.Query(`SELECT node_id, parent_id, vec FROM vectors WHERE dims = ?`, len(vec)) if err != nil { return nil, err } @@ -220,9 +220,9 @@ func (s *Store) SimilarTo(vec []float32, limit int) ([]graph.VectorHit, error) { // `limit` and yields an exact top-k. h := &hitHeap{} for rows.Next() { - var id string + var id, parentID string var blob sql.RawBytes - if err := rows.Scan(&id, &blob); err != nil { + if err := rows.Scan(&id, &parentID, &blob); err != nil { return nil, err } cand := decodeVec(blob) @@ -236,9 +236,9 @@ func (s *Store) SimilarTo(vec []float32, limit int) ([]graph.VectorHit, error) { dist := cosineDistance(vec, cand, qNorm, cNorm) if h.Len() < limit { - heap.Push(h, graph.VectorHit{NodeID: id, Distance: dist}) + heap.Push(h, graph.VectorHit{NodeID: id, ParentID: parentID, Distance: dist}) } else if dist < (*h)[0].Distance { - (*h)[0] = graph.VectorHit{NodeID: id, Distance: dist} + (*h)[0] = graph.VectorHit{NodeID: id, ParentID: parentID, Distance: dist} heap.Fix(h, 0) } } diff --git a/internal/graph/store_sqlite/store_vector_corpus.go b/internal/graph/store_sqlite/store_vector_corpus.go new file mode 100644 index 000000000..1b52418b5 --- /dev/null +++ b/internal/graph/store_sqlite/store_vector_corpus.go @@ -0,0 +1,395 @@ +package store_sqlite + +import ( + "context" + "database/sql" + "errors" + "fmt" + "math" + "sort" + "strconv" + "strings" + + "github.com/zzet/gortex/internal/graph" +) + +var _ graph.AtomicVectorCorpusInstaller = (*Store)(nil) + +var ( + errVectorCorpusInvalidDims = errors.New("store_sqlite: vector corpus dims must be positive") + errVectorCorpusAmbiguousDims = errors.New("store_sqlite: durable vector corpus has multiple dimensions") +) + +// ReplaceVectorCorpus atomically replaces every durable vector owned by one +// repository. All input validation and cross-repository ownership checks occur +// before the first DELETE. The returned stats describe the complete committed +// corpus at dims, not just the replaced repository. +func (s *Store) ReplaceVectorCorpus( + ctx context.Context, + repoPrefix string, + dims int, + items []graph.VectorCorpusItem, +) (graph.VectorCorpusStats, error) { + if err := validateVectorCorpus(ctx, dims, items); err != nil { + return graph.VectorCorpusStats{}, err + } + if err := s.writeMu.LockContext(ctx); err != nil { + return graph.VectorCorpusStats{}, err + } + defer s.writeMu.Unlock() + + tx, err := s.beginWriteContext(ctx) + if err != nil { + return graph.VectorCorpusStats{}, err + } + defer tx.Rollback() //nolint:errcheck // rollback after Commit is a no-op + + if err := validateVectorCorpusOwnershipTx(ctx, tx, repoPrefix, items); err != nil { + return graph.VectorCorpusStats{}, err + } + if _, err := tx.ExecContext(ctx, `DELETE FROM vectors WHERE repo_prefix = ?`, repoPrefix); err != nil { + return graph.VectorCorpusStats{}, err + } + if err := insertVectorCorpusTx(ctx, tx, repoPrefix, dims, items); err != nil { + return graph.VectorCorpusStats{}, err + } + stats, err := vectorCorpusStatsForRepoDims(ctx, tx, repoPrefix, dims) + if err != nil { + return graph.VectorCorpusStats{}, err + } + if err := ctx.Err(); err != nil { + return graph.VectorCorpusStats{}, err + } + if err := tx.Commit(); err != nil { + return graph.VectorCorpusStats{}, err + } + return stats, nil +} + +// VectorCorpusStats reports the durable corpus without materialising vectors. +// A non-positive dims value is discovery mode for warm restart: it succeeds +// only when the store is empty or has exactly one positive dimension. +func (s *Store) VectorCorpusStats(ctx context.Context, dims int) (graph.VectorCorpusStats, error) { + if err := ctx.Err(); err != nil { + return graph.VectorCorpusStats{}, err + } + if dims > 0 { + return vectorCorpusStatsForDims(ctx, s.db, dims) + } + + rows, err := s.db.QueryContext(ctx, ` +SELECT dims, + COUNT(*), + COALESCE(SUM(CASE WHEN parent_id <> '' THEN 1 ELSE 0 END), 0) +FROM vectors +GROUP BY dims +ORDER BY dims +LIMIT 2`) + if err != nil { + return graph.VectorCorpusStats{}, err + } + defer rows.Close() + + if !rows.Next() { + if err := rows.Err(); err != nil { + return graph.VectorCorpusStats{}, err + } + return graph.VectorCorpusStats{}, nil + } + var stats graph.VectorCorpusStats + if err := rows.Scan(&stats.Dims, &stats.VectorCount, &stats.ChunkCount); err != nil { + return graph.VectorCorpusStats{}, err + } + if stats.Dims <= 0 { + return graph.VectorCorpusStats{}, fmt.Errorf("store_sqlite: durable vector corpus has invalid dimension %d", stats.Dims) + } + if rows.Next() { + return graph.VectorCorpusStats{}, errVectorCorpusAmbiguousDims + } + if err := rows.Err(); err != nil { + return graph.VectorCorpusStats{}, err + } + return stats, nil +} + +// VectorCorpusStatsForRepo returns global publication counts together with the +// selected repository's contribution. Discovery mode resolves the sole stored +// dimension in one aggregate query, avoiding a cross-query generation race. +func (s *Store) VectorCorpusStatsForRepo( + ctx context.Context, + repoPrefix string, + dims int, +) (graph.VectorCorpusStats, error) { + if err := ctx.Err(); err != nil { + return graph.VectorCorpusStats{}, err + } + if dims > 0 { + return vectorCorpusStatsForRepoDims(ctx, s.db, repoPrefix, dims) + } + + var stats graph.VectorCorpusStats + var distinctDims int + err := s.db.QueryRowContext(ctx, ` +SELECT COALESCE(MIN(dims), 0), + COUNT(DISTINCT dims), + COUNT(*), + COALESCE(SUM(CASE WHEN parent_id <> '' THEN 1 ELSE 0 END), 0), + COALESCE(SUM(CASE WHEN repo_prefix = ? THEN 1 ELSE 0 END), 0), + COALESCE(SUM(CASE WHEN repo_prefix = ? AND parent_id <> '' THEN 1 ELSE 0 END), 0) +FROM vectors`, repoPrefix, repoPrefix).Scan( + &stats.Dims, + &distinctDims, + &stats.VectorCount, + &stats.ChunkCount, + &stats.RepositoryVectorCount, + &stats.RepositoryChunkCount, + ) + if err != nil { + return graph.VectorCorpusStats{}, err + } + if distinctDims > 1 { + return graph.VectorCorpusStats{}, errVectorCorpusAmbiguousDims + } + if stats.VectorCount > 0 && stats.Dims <= 0 { + return graph.VectorCorpusStats{}, fmt.Errorf("store_sqlite: durable vector corpus has invalid dimension %d", stats.Dims) + } + return stats, nil +} + +func validateVectorCorpus(ctx context.Context, dims int, items []graph.VectorCorpusItem) error { + if dims <= 0 { + return errVectorCorpusInvalidDims + } + seen := make(map[string]struct{}, len(items)) + for i, item := range items { + if i&255 == 0 { + if err := ctx.Err(); err != nil { + return err + } + } + if strings.TrimSpace(item.NodeID) == "" { + return fmt.Errorf("store_sqlite: vector corpus item %d has empty node ID", i) + } + if _, ok := seen[item.NodeID]; ok { + return fmt.Errorf("store_sqlite: vector corpus contains duplicate node ID %q", item.NodeID) + } + seen[item.NodeID] = struct{}{} + if len(item.Vec) != dims { + return fmt.Errorf("store_sqlite: vector corpus item %q has %d dimensions, want %d", item.NodeID, len(item.Vec), dims) + } + for _, value := range item.Vec { + f := float64(value) + if math.IsNaN(f) || math.IsInf(f, 0) { + return fmt.Errorf("store_sqlite: vector corpus item %q contains a non-finite value", item.NodeID) + } + } + } + return ctx.Err() +} + +func validateVectorCorpusOwnershipTx( + ctx context.Context, + tx *sql.Tx, + repoPrefix string, + items []graph.VectorCorpusItem, +) error { + if len(items) == 0 { + return nil + } + + // Every vector must be anchored in the durable graph. Ordinary vectors use + // their own node ID; synthetic chunks use their durable parent symbol. + ownershipTargets := make(map[string]struct{}, len(items)) + for _, item := range items { + targetID := item.NodeID + if item.ParentID != "" { + if err := validateCanonicalVectorChunkID(item.NodeID, item.ParentID); err != nil { + return err + } + targetID = item.ParentID + } + ownershipTargets[targetID] = struct{}{} + } + targetIDs := make([]string, 0, len(ownershipTargets)) + for id := range ownershipTargets { + targetIDs = append(targetIDs, id) + } + sort.Strings(targetIDs) + + for start := 0; start < len(targetIDs); start += vectorChunk { + end := start + vectorChunk + if end > len(targetIDs) { + end = len(targetIDs) + } + batch := targetIDs[start:end] + stmt := make([]byte, 0, 72+len(batch)*2) + stmt = append(stmt, "SELECT id, repo_prefix FROM nodes WHERE id IN ("...) + args := make([]any, 0, len(batch)) + for i, id := range batch { + if i > 0 { + stmt = append(stmt, ',') + } + stmt = append(stmt, '?') + args = append(args, id) + } + stmt = append(stmt, ')') + + rows, err := tx.QueryContext(ctx, string(stmt), args...) + if err != nil { + return err + } + owners := make(map[string]string, len(batch)) + for rows.Next() { + var id, owner string + if err := rows.Scan(&id, &owner); err != nil { + _ = rows.Close() + return err + } + owners[id] = owner + } + if err := rows.Err(); err != nil { + _ = rows.Close() + return err + } + if err := rows.Close(); err != nil { + return err + } + for _, id := range batch { + owner, ok := owners[id] + if !ok { + return fmt.Errorf("store_sqlite: vector ownership node %q does not exist", id) + } + if owner != repoPrefix { + return fmt.Errorf("store_sqlite: vector ownership node %q belongs to repository %q, not %q", id, owner, repoPrefix) + } + } + } + + // Keep the vector-table ownership check as a second line of defence. It + // prevents a malformed historical row from being reassigned through the + // global node_id primary key even when the durable node anchor is valid. + for start := 0; start < len(items); start += vectorChunk { + end := start + vectorChunk + if end > len(items) { + end = len(items) + } + batch := items[start:end] + stmt := make([]byte, 0, 96+len(batch)*2) + stmt = append(stmt, "SELECT node_id, repo_prefix FROM vectors WHERE node_id IN ("...) + args := make([]any, 0, len(batch)+1) + for i, item := range batch { + if i > 0 { + stmt = append(stmt, ',') + } + stmt = append(stmt, '?') + args = append(args, item.NodeID) + } + stmt = append(stmt, ") AND repo_prefix <> ? LIMIT 1"...) + args = append(args, repoPrefix) + + var nodeID, owner string + err := tx.QueryRowContext(ctx, string(stmt), args...).Scan(&nodeID, &owner) + switch { + case err == nil: + return fmt.Errorf("store_sqlite: vector %q is owned by repository %q, not %q", nodeID, owner, repoPrefix) + case errors.Is(err, sql.ErrNoRows): + continue + default: + return err + } + } + return nil +} + +func validateCanonicalVectorChunkID(nodeID, parentID string) error { + const marker = "#chunk" + suffix, ok := strings.CutPrefix(nodeID, parentID+marker) + if !ok || suffix == "" { + return fmt.Errorf("store_sqlite: vector chunk %q is not derived from parent %q", nodeID, parentID) + } + ordinal, err := strconv.Atoi(suffix) + if err != nil || ordinal < 0 || strconv.Itoa(ordinal) != suffix { + return fmt.Errorf("store_sqlite: vector chunk %q has a non-canonical ordinal", nodeID) + } + return nil +} + +func insertVectorCorpusTx( + ctx context.Context, + tx *sql.Tx, + repoPrefix string, + dims int, + items []graph.VectorCorpusItem, +) error { + for start := 0; start < len(items); start += vectorChunk { + if err := ctx.Err(); err != nil { + return err + } + end := start + vectorChunk + if end > len(items) { + end = len(items) + } + batch := items[start:end] + + stmt := make([]byte, 0, 96+len(batch)*20) + stmt = append(stmt, "INSERT INTO vectors (node_id, repo_prefix, parent_id, dims, vec) VALUES "...) + args := make([]any, 0, len(batch)*5) + for i, item := range batch { + if i > 0 { + stmt = append(stmt, ',') + } + stmt = append(stmt, "(?, ?, ?, ?, ?)"...) + args = append(args, item.NodeID, repoPrefix, item.ParentID, dims, encodeVec(item.Vec)) + } + if _, err := tx.ExecContext(ctx, string(stmt), args...); err != nil { + return err + } + } + return nil +} + +type vectorCorpusStatsQuerier interface { + QueryRowContext(context.Context, string, ...any) *sql.Row +} + +func vectorCorpusStatsForDims( + ctx context.Context, + q vectorCorpusStatsQuerier, + dims int, +) (graph.VectorCorpusStats, error) { + stats := graph.VectorCorpusStats{Dims: dims} + err := q.QueryRowContext(ctx, ` +SELECT COUNT(*), + COALESCE(SUM(CASE WHEN parent_id <> '' THEN 1 ELSE 0 END), 0) +FROM vectors +WHERE dims = ?`, dims).Scan(&stats.VectorCount, &stats.ChunkCount) + if err != nil { + return graph.VectorCorpusStats{}, err + } + return stats, nil +} + +func vectorCorpusStatsForRepoDims( + ctx context.Context, + q vectorCorpusStatsQuerier, + repoPrefix string, + dims int, +) (graph.VectorCorpusStats, error) { + stats := graph.VectorCorpusStats{Dims: dims} + err := q.QueryRowContext(ctx, ` +SELECT COUNT(*), + COALESCE(SUM(CASE WHEN parent_id <> '' THEN 1 ELSE 0 END), 0), + COALESCE(SUM(CASE WHEN repo_prefix = ? THEN 1 ELSE 0 END), 0), + COALESCE(SUM(CASE WHEN repo_prefix = ? AND parent_id <> '' THEN 1 ELSE 0 END), 0) +FROM vectors +WHERE dims = ?`, repoPrefix, repoPrefix, dims).Scan( + &stats.VectorCount, + &stats.ChunkCount, + &stats.RepositoryVectorCount, + &stats.RepositoryChunkCount, + ) + if err != nil { + return graph.VectorCorpusStats{}, err + } + return stats, nil +} diff --git a/internal/graph/store_sqlite/store_vector_corpus_test.go b/internal/graph/store_sqlite/store_vector_corpus_test.go new file mode 100644 index 000000000..a8ee31dda --- /dev/null +++ b/internal/graph/store_sqlite/store_vector_corpus_test.go @@ -0,0 +1,518 @@ +package store_sqlite + +import ( + "context" + "database/sql" + "errors" + "fmt" + "path/filepath" + "sort" + "strings" + "sync" + "testing" + + "github.com/zzet/gortex/internal/graph" +) + +func TestReplaceVectorCorpusIsAtomicPerRepositoryAndReportsStats(t *testing.T) { + store := openVectorCorpusTestStore(t, filepath.Join(t.TempDir(), "corpus.sqlite")) + ctx := context.Background() + addVectorCorpusNodes(t, store, "A", "A/one", "A/two", "A/new") + addVectorCorpusNodes(t, store, "B", "B/keep") + + stats, err := store.ReplaceVectorCorpus(ctx, "A", 2, []graph.VectorCorpusItem{ + {NodeID: "A/one", Vec: []float32{1, 0}}, + {NodeID: "A/two#chunk0", ParentID: "A/two", Vec: []float32{0, 1}}, + }) + if err != nil { + t.Fatal(err) + } + assertVectorCorpusStats(t, stats, graph.VectorCorpusStats{ + VectorCount: 2, ChunkCount: 1, + RepositoryVectorCount: 2, RepositoryChunkCount: 1, + Dims: 2, + }) + + stats, err = store.ReplaceVectorCorpus(ctx, "B", 2, []graph.VectorCorpusItem{ + {NodeID: "B/keep", Vec: []float32{1, 1}}, + }) + if err != nil { + t.Fatal(err) + } + assertVectorCorpusStats(t, stats, graph.VectorCorpusStats{ + VectorCount: 3, ChunkCount: 1, + RepositoryVectorCount: 1, + Dims: 2, + }) + + stats, err = store.ReplaceVectorCorpus(ctx, "A", 2, []graph.VectorCorpusItem{ + {NodeID: "A/new", Vec: []float32{1, -1}}, + }) + if err != nil { + t.Fatal(err) + } + assertVectorCorpusStats(t, stats, graph.VectorCorpusStats{ + VectorCount: 2, + RepositoryVectorCount: 1, + Dims: 2, + }) + assertVectorCorpusRows(t, store, []vectorCorpusTestRow{ + {nodeID: "A/new", repoPrefix: "A", dims: 2}, + {nodeID: "B/keep", repoPrefix: "B", dims: 2}, + }) + + stats, err = store.ReplaceVectorCorpus(ctx, "A", 2, nil) + if err != nil { + t.Fatal(err) + } + assertVectorCorpusStats(t, stats, graph.VectorCorpusStats{VectorCount: 1, Dims: 2}) + assertVectorCorpusRows(t, store, []vectorCorpusTestRow{ + {nodeID: "B/keep", repoPrefix: "B", dims: 2}, + }) +} + +func TestVectorCorpusStatsForRepoReportsGlobalAndScopedCounts(t *testing.T) { + store := openVectorCorpusTestStore(t, filepath.Join(t.TempDir(), "repo-stats.sqlite")) + ctx := context.Background() + addVectorCorpusNodes(t, store, "A", "A/one", "A/two") + addVectorCorpusNodes(t, store, "B", "B/keep") + + if _, err := store.ReplaceVectorCorpus(ctx, "A", 2, []graph.VectorCorpusItem{ + {NodeID: "A/one", Vec: []float32{1, 0}}, + {NodeID: "A/two#chunk0", ParentID: "A/two", Vec: []float32{0, 1}}, + }); err != nil { + t.Fatal(err) + } + if _, err := store.ReplaceVectorCorpus(ctx, "B", 2, []graph.VectorCorpusItem{ + {NodeID: "B/keep", Vec: []float32{1, 1}}, + }); err != nil { + t.Fatal(err) + } + + stats, err := store.VectorCorpusStatsForRepo(ctx, "A", 2) + if err != nil { + t.Fatal(err) + } + assertVectorCorpusStats(t, stats, graph.VectorCorpusStats{ + VectorCount: 3, ChunkCount: 1, + RepositoryVectorCount: 2, RepositoryChunkCount: 1, + Dims: 2, + }) + + stats, err = store.VectorCorpusStatsForRepo(ctx, "B", 0) + if err != nil { + t.Fatal(err) + } + assertVectorCorpusStats(t, stats, graph.VectorCorpusStats{ + VectorCount: 3, ChunkCount: 1, + RepositoryVectorCount: 1, + Dims: 2, + }) + + stats, err = store.VectorCorpusStatsForRepo(ctx, "missing", 0) + if err != nil { + t.Fatal(err) + } + assertVectorCorpusStats(t, stats, graph.VectorCorpusStats{ + VectorCount: 3, ChunkCount: 1, Dims: 2, + }) +} + +func TestVectorCorpusChunkParentsSurviveReopen(t *testing.T) { + path := filepath.Join(t.TempDir(), "chunks.sqlite") + store, err := Open(path) + if err != nil { + t.Fatal(err) + } + addVectorCorpusNodes(t, store, "repo", "repo/symbol") + _, err = store.ReplaceVectorCorpus(context.Background(), "repo", 2, []graph.VectorCorpusItem{ + {NodeID: "repo/symbol#chunk0", ParentID: "repo/symbol", Vec: []float32{1, 0}}, + }) + if err != nil { + t.Fatal(err) + } + if err := store.Close(); err != nil { + t.Fatal(err) + } + + store, err = Open(path) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { _ = store.Close() }) + + hits, err := store.SimilarTo([]float32{1, 0}, 10) + if err != nil { + t.Fatal(err) + } + if len(hits) != 1 || hits[0].NodeID != "repo/symbol#chunk0" || hits[0].ParentID != "repo/symbol" { + t.Fatalf("durable chunk identity lost after reopen: %#v", hits) + } + stats, err := store.VectorCorpusStats(context.Background(), 0) + if err != nil { + t.Fatal(err) + } + assertVectorCorpusStats(t, stats, graph.VectorCorpusStats{VectorCount: 1, ChunkCount: 1, Dims: 2}) +} + +func TestReplaceVectorCorpusFailuresPreserveCommittedRows(t *testing.T) { + store := openVectorCorpusTestStore(t, filepath.Join(t.TempDir(), "rollback.sqlite")) + ctx := context.Background() + addVectorCorpusNodes(t, store, "A", "A/old", "A/new", "A/fail") + addVectorCorpusNodes(t, store, "B", "B/keep", "B/fresh", "B/foreign-parent") + if _, err := store.ReplaceVectorCorpus(ctx, "A", 2, []graph.VectorCorpusItem{ + {NodeID: "A/old", Vec: []float32{1, 0}}, + }); err != nil { + t.Fatal(err) + } + if _, err := store.ReplaceVectorCorpus(ctx, "B", 2, []graph.VectorCorpusItem{ + {NodeID: "B/keep", Vec: []float32{0, 1}}, + }); err != nil { + t.Fatal(err) + } + + assertPreserved := func(label string) { + t.Helper() + assertVectorCorpusRows(t, store, []vectorCorpusTestRow{ + {nodeID: "A/old", repoPrefix: "A", dims: 2}, + {nodeID: "B/keep", repoPrefix: "B", dims: 2}, + }) + } + + if _, err := store.ReplaceVectorCorpus(ctx, "A", 2, []graph.VectorCorpusItem{ + {NodeID: "A/bad", Vec: []float32{1}}, + }); err == nil { + t.Fatal("dimension mismatch unexpectedly succeeded") + } + assertPreserved("invalid dimensions") + + if _, err := store.ReplaceVectorCorpus(ctx, "A", 2, []graph.VectorCorpusItem{ + {NodeID: "B/fresh", Vec: []float32{1, 1}}, + }); err == nil { + t.Fatal("fresh cross-repository node ownership conflict unexpectedly succeeded") + } + assertPreserved("fresh ownership conflict") + + if _, err := store.ReplaceVectorCorpus(ctx, "A", 2, []graph.VectorCorpusItem{ + {NodeID: "B/foreign-parent#chunk0", ParentID: "B/foreign-parent", Vec: []float32{1, 1}}, + }); err == nil { + t.Fatal("foreign chunk parent unexpectedly succeeded") + } + assertPreserved("foreign chunk parent") + + if _, err := store.ReplaceVectorCorpus(ctx, "A", 2, []graph.VectorCorpusItem{ + {NodeID: "A/missing#chunk0", ParentID: "A/missing", Vec: []float32{1, 1}}, + }); err == nil { + t.Fatal("missing chunk parent unexpectedly succeeded") + } + assertPreserved("missing chunk parent") + + if _, err := store.ReplaceVectorCorpus(ctx, "A", 2, []graph.VectorCorpusItem{ + {NodeID: "A/old#chunk01", ParentID: "A/old", Vec: []float32{1, 1}}, + }); err == nil { + t.Fatal("malformed chunk ID unexpectedly succeeded") + } + assertPreserved("malformed chunk ID") + + cancelled, cancel := context.WithCancel(ctx) + cancel() + if _, err := store.ReplaceVectorCorpus(cancelled, "A", 2, []graph.VectorCorpusItem{ + {NodeID: "A/cancelled", Vec: []float32{1, 1}}, + }); !errors.Is(err, context.Canceled) { + t.Fatalf("cancelled replacement error = %v, want context.Canceled", err) + } + assertPreserved("cancellation") + + if _, err := store.writerDB.Exec(` +CREATE TRIGGER fail_vector_corpus_insert +BEFORE INSERT ON vectors +WHEN NEW.node_id = 'A/fail' +BEGIN + SELECT RAISE(ABORT, 'injected vector failure'); +END`); err != nil { + t.Fatal(err) + } + if _, err := store.ReplaceVectorCorpus(ctx, "A", 2, []graph.VectorCorpusItem{ + {NodeID: "A/new", Vec: []float32{1, 1}}, + {NodeID: "A/fail", Vec: []float32{-1, 1}}, + }); err == nil { + t.Fatal("injected insertion failure unexpectedly succeeded") + } + assertPreserved("transaction failure") +} + +func TestVectorCorpusStatsDiscoveryRejectsMixedDimensions(t *testing.T) { + store := openVectorCorpusTestStore(t, filepath.Join(t.TempDir(), "stats.sqlite")) + ctx := context.Background() + addVectorCorpusNodes(t, store, "A", "A/three") + addVectorCorpusNodes(t, store, "B", "B/two") + + stats, err := store.VectorCorpusStats(ctx, 0) + if err != nil { + t.Fatal(err) + } + assertVectorCorpusStats(t, stats, graph.VectorCorpusStats{}) + + if _, err := store.ReplaceVectorCorpus(ctx, "A", 3, []graph.VectorCorpusItem{ + {NodeID: "A/three", Vec: []float32{1, 0, 0}}, + }); err != nil { + t.Fatal(err) + } + stats, err = store.VectorCorpusStats(ctx, -1) + if err != nil { + t.Fatal(err) + } + assertVectorCorpusStats(t, stats, graph.VectorCorpusStats{VectorCount: 1, Dims: 3}) + + if _, err := store.ReplaceVectorCorpus(ctx, "B", 2, []graph.VectorCorpusItem{ + {NodeID: "B/two", Vec: []float32{0, 1}}, + }); err != nil { + t.Fatal(err) + } + if _, err := store.VectorCorpusStats(ctx, 0); !errors.Is(err, errVectorCorpusAmbiguousDims) { + t.Fatalf("mixed-dimension discovery error = %v, want %v", err, errVectorCorpusAmbiguousDims) + } + stats, err = store.VectorCorpusStats(ctx, 3) + if err != nil { + t.Fatal(err) + } + assertVectorCorpusStats(t, stats, graph.VectorCorpusStats{VectorCount: 1, Dims: 3}) +} + +func TestSimilarToObservesCompleteCorpusDuringReplacement(t *testing.T) { + store := openVectorCorpusTestStore(t, filepath.Join(t.TempDir(), "visibility.sqlite")) + ctx := context.Background() + oldItems := vectorCorpusItems("repo/old-", 64) + newItems := vectorCorpusItems("repo/new-", 64) + nodeIDs := make([]string, 0, len(oldItems)+len(newItems)) + for _, item := range oldItems { + nodeIDs = append(nodeIDs, item.NodeID) + } + for _, item := range newItems { + nodeIDs = append(nodeIDs, item.NodeID) + } + addVectorCorpusNodes(t, store, "repo", nodeIDs...) + if _, err := store.ReplaceVectorCorpus(ctx, "repo", 2, oldItems); err != nil { + t.Fatal(err) + } + + stop := make(chan struct{}) + observed := make(chan struct{}) + errCh := make(chan error, 1) + var once sync.Once + go func() { + for { + select { + case <-stop: + errCh <- nil + return + default: + } + hits, err := store.SimilarTo([]float32{1, 1}, 128) + if err != nil { + errCh <- err + return + } + if err := classifyCompleteVectorCorpus(hits, len(oldItems)); err != nil { + errCh <- err + return + } + once.Do(func() { close(observed) }) + } + }() + <-observed + + for i := 0; i < 24; i++ { + items := newItems + if i%2 == 1 { + items = oldItems + } + if _, err := store.ReplaceVectorCorpus(ctx, "repo", 2, items); err != nil { + close(stop) + <-errCh + t.Fatal(err) + } + } + close(stop) + if err := <-errCh; err != nil { + t.Fatal(err) + } +} + +func TestOpenV9RebuildsOnlyVectorCorpus(t *testing.T) { + path := filepath.Join(t.TempDir(), "v9.sqlite") + store, err := Open(path) + if err != nil { + t.Fatal(err) + } + store.AddNode(&graph.Node{ + ID: "repo/file.go::Kept", + Kind: graph.KindFunction, + Name: "Kept", + FilePath: "repo/file.go", + RepoPrefix: "repo", + }) + if err := store.Close(); err != nil { + t.Fatal(err) + } + + withRawDB(t, path, func(db *sql.DB) { + if _, err := db.Exec(`DROP INDEX IF EXISTS vectors_by_repo`); err != nil { + t.Fatal(err) + } + if _, err := db.Exec(`DROP TABLE vectors`); err != nil { + t.Fatal(err) + } + if _, err := db.Exec(`CREATE TABLE vectors ( + node_id TEXT PRIMARY KEY, + dims INTEGER NOT NULL, + vec BLOB NOT NULL + ) WITHOUT ROWID`); err != nil { + t.Fatal(err) + } + if _, err := db.Exec(`INSERT INTO vectors(node_id, dims, vec) VALUES (?, ?, ?)`, "repo/legacy", 2, encodeVec([]float32{1, 0})); err != nil { + t.Fatal(err) + } + if _, err := db.Exec(`PRAGMA user_version = 9`); err != nil { + t.Fatal(err) + } + }) + + store, err = Open(path) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { _ = store.Close() }) + if store.NeedsRebuild() { + t.Fatal("v9 vector-sidecar migration unexpectedly wiped the graph store") + } + if got := store.GetNode("repo/file.go::Kept"); got == nil { + t.Fatal("v9 vector migration discarded graph topology") + } + stats, err := store.VectorCorpusStats(context.Background(), 0) + if err != nil { + t.Fatal(err) + } + assertVectorCorpusStats(t, stats, graph.VectorCorpusStats{}) + + columns := make(map[string]bool) + rows, err := store.db.Query(`PRAGMA table_info(vectors)`) + if err != nil { + t.Fatal(err) + } + defer rows.Close() + for rows.Next() { + var cid, notNull, pk int + var name, typ string + var defaultValue any + if err := rows.Scan(&cid, &name, &typ, ¬Null, &defaultValue, &pk); err != nil { + t.Fatal(err) + } + columns[name] = true + } + if err := rows.Err(); err != nil { + t.Fatal(err) + } + if !columns["repo_prefix"] || !columns["parent_id"] { + t.Fatalf("migrated vector columns = %v", columns) + } +} + +type vectorCorpusTestRow struct { + nodeID string + repoPrefix string + parentID string + dims int +} + +func openVectorCorpusTestStore(t *testing.T, path string) *Store { + t.Helper() + store, err := Open(path) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { _ = store.Close() }) + return store +} + +func addVectorCorpusNodes(t *testing.T, store *Store, repoPrefix string, ids ...string) { + t.Helper() + nodes := make([]*graph.Node, 0, len(ids)) + for _, id := range ids { + nodes = append(nodes, &graph.Node{ + ID: id, + Kind: graph.KindFunction, + Name: id, + FilePath: id, + RepoPrefix: repoPrefix, + }) + } + store.AddBatch(nodes, nil) +} + +func assertVectorCorpusStats(t *testing.T, got, want graph.VectorCorpusStats) { + t.Helper() + if got != want { + t.Fatalf("vector corpus stats = %+v, want %+v", got, want) + } +} + +func assertVectorCorpusRows(t *testing.T, store *Store, want []vectorCorpusTestRow) { + t.Helper() + rows, err := store.db.Query(`SELECT node_id, repo_prefix, parent_id, dims FROM vectors ORDER BY node_id`) + if err != nil { + t.Fatal(err) + } + defer rows.Close() + var got []vectorCorpusTestRow + for rows.Next() { + var row vectorCorpusTestRow + if err := rows.Scan(&row.nodeID, &row.repoPrefix, &row.parentID, &row.dims); err != nil { + t.Fatal(err) + } + got = append(got, row) + } + if err := rows.Err(); err != nil { + t.Fatal(err) + } + if fmt.Sprint(got) != fmt.Sprint(want) { + t.Fatalf("vector rows = %#v, want %#v", got, want) + } +} + +func vectorCorpusItems(prefix string, count int) []graph.VectorCorpusItem { + items := make([]graph.VectorCorpusItem, count) + for i := range items { + items[i] = graph.VectorCorpusItem{ + NodeID: fmt.Sprintf("%s%03d", prefix, i), + Vec: []float32{1, 1}, + } + } + return items +} + +func classifyCompleteVectorCorpus(hits []graph.VectorHit, want int) error { + if len(hits) != want { + return fmt.Errorf("observed partial vector corpus: got %d hits, want %d", len(hits), want) + } + ids := make([]string, len(hits)) + for i, hit := range hits { + ids[i] = hit.NodeID + } + sort.Strings(ids) + old := strings.HasPrefix(ids[0], "repo/old-") + newCorpus := strings.HasPrefix(ids[0], "repo/new-") + if !old && !newCorpus { + return fmt.Errorf("unexpected vector corpus ID %q", ids[0]) + } + prefix := "repo/old-" + if newCorpus { + prefix = "repo/new-" + } + for _, id := range ids { + if !strings.HasPrefix(id, prefix) { + return fmt.Errorf("observed mixed vector corpus: %q and prefix %q", id, prefix) + } + } + return nil +} diff --git a/internal/graph/store_sqlite/structural_integrity.go b/internal/graph/store_sqlite/structural_integrity.go new file mode 100644 index 000000000..1ee8a6802 --- /dev/null +++ b/internal/graph/store_sqlite/structural_integrity.go @@ -0,0 +1,225 @@ +package store_sqlite + +import ( + "context" + "fmt" + "log" + "sort" + "strings" + + "github.com/zzet/gortex/internal/graph" +) + +const ( + structuralExactDefaultSamples = 20 + structuralExactMaxSamples = 100 + structuralAuditRepoExpr = `COALESCE(NULLIF(n.repo_prefix, ''), 'unknown')` + structuralAuditReasonExpr = `CASE WHEN instr(e.to_id, '#param:') > 0 THEN 'parameter_target' ELSE 'local_target' END` + structuralAuditOriginExpr = `CASE WHEN trim(COALESCE(e.origin, '')) = '' THEN 'unknown' ELSE lower(substr(trim(e.origin), 1, 48)) END` + structuralAuditFromSQL = ` FROM edges e LEFT JOIN nodes n ON n.id = e.from_id WHERE e.kind IN (?, ?, ?, ?, ?) AND (instr(e.to_id, '#param:') > 0 OR instr(e.to_id, '#local:') > 0)` +) + +var structuralAuditKinds = []any{ + string(graph.EdgeImplements), + string(graph.EdgeExtends), + string(graph.EdgeOverrides), + string(graph.EdgeInstantiates), + string(graph.EdgeMemberOf), +} + +var ( + _ graph.StructuralIntegrityEventRecorder = (*Store)(nil) + _ graph.StructuralIntegritySnapshotter = (*Store)(nil) + _ graph.StructuralIntegrityAuditor = (*Store)(nil) +) + +func (s *Store) RecordStructuralIntegrityEvent(event graph.StructuralIntegrityEvent) { + if s == nil { + return + } + s.structuralIntegrity.Record(event) + switch event.Direction { + case graph.StructuralDropWrite: + if s.structuralWriteWarned.CompareAndSwap(false, true) { + log.Printf("store_sqlite: structurally invalid edge rejected; run analyze(kind=edge_audit) for attributed diagnostics") + } + case graph.StructuralDropRead: + if s.structuralReadWarned.CompareAndSwap(false, true) { + log.Printf("store_sqlite: store contains structurally invalid legacy edges; suppressing them on read — rebuild or run analyze(kind=edge_audit)") + } + } +} + +func (s *Store) StructuralIntegritySnapshot(opts graph.StructuralIntegritySnapshotOptions) graph.StructuralIntegritySnapshot { + if s == nil { + return graph.StructuralIntegritySnapshot{} + } + return s.structuralIntegrity.Snapshot(opts) +} + +func (s *Store) recordStructuralEdge(direction graph.StructuralDropDirection, path graph.StructuralDropPath, repo string, edge *graph.Edge) { + if event, ok := graph.StructuralIntegrityEventForEdge(direction, path, repo, edge); ok { + s.RecordStructuralIntegrityEvent(event) + } +} + +func structuralInputNodeRepos(nodes []*graph.Node) map[string]string { + var repos map[string]string + for _, node := range nodes { + if node == nil || node.ID == "" || node.RepoPrefix == "" { + continue + } + if repos == nil { + repos = make(map[string]string) + } + repos[node.ID] = node.RepoPrefix + } + return repos +} + +func (s *Store) structuralWriteRepo(edge *graph.Edge, inputRepos map[string]string) string { + if edge == nil { + return "" + } + if repo := inputRepos[edge.From]; repo != "" { + return repo + } + if source := s.GetNode(edge.From); source != nil { + return source.RepoPrefix + } + return "" +} + +func (s *Store) noteStructuralReadDrop(path graph.StructuralDropPath, edge *graph.Edge) { + // Scanner rows do not carry authoritative source ownership. Do not guess a + // repository from the edge ID while the cursor is open; exact audit joins + // the source node explicitly after the routine scan has completed. + s.recordStructuralEdge(graph.StructuralDropRead, path, "", edge) +} + +func normalizedStructuralAuditRepos(repos []string) []string { + set := make(map[string]struct{}, len(repos)) + for _, repo := range repos { + repo = strings.TrimSpace(repo) + if repo == "" { + repo = "unknown" + } + set[repo] = struct{}{} + } + out := make([]string, 0, len(set)) + for repo := range set { + out = append(out, repo) + } + sort.Strings(out) + return out +} + +func structuralAuditScopeSQL(repos []string, active bool) (string, []any) { + repos = normalizedStructuralAuditRepos(repos) + if len(repos) == 0 { + if active { + return " AND 1 = 0", nil + } + return "", nil + } + placeholders := make([]string, len(repos)) + args := make([]any, len(repos)) + for i, repo := range repos { + placeholders[i] = "?" + args[i] = repo + } + return " AND " + structuralAuditRepoExpr + " IN (" + strings.Join(placeholders, ", ") + ")", args +} + +func structuralAuditArgs(scopeArgs []any) []any { + args := make([]any, 0, len(structuralAuditKinds)+len(scopeArgs)+1) + args = append(args, structuralAuditKinds...) + args = append(args, scopeArgs...) + return args +} + +// AuditStructuralIntegrity performs the explicit persisted-row audit. Routine +// status never calls this capability. +func (s *Store) AuditStructuralIntegrity(ctx context.Context, opts graph.StructuralIntegrityAuditOptions) (graph.StructuralIntegrityExactAudit, error) { + out := graph.StructuralIntegrityExactAudit{Status: graph.StructuralAuditSupported} + if err := ctx.Err(); err != nil { + return out, err + } + scopeSQL, scopeArgs := structuralAuditScopeSQL(opts.RepoPrefixes, opts.RepoScopeActive) + groupSQL := `SELECT ` + structuralAuditRepoExpr + `, e.kind, ` + structuralAuditReasonExpr + `, ` + structuralAuditOriginExpr + `, COUNT(*)` + + structuralAuditFromSQL + scopeSQL + ` GROUP BY 1, 2, 3, 4 ORDER BY 1, 2, 3, 4` + rows, err := s.db.QueryContext(ctx, groupSQL, structuralAuditArgs(scopeArgs)...) + if err != nil { + return out, fmt.Errorf("store_sqlite: structural integrity grouping: %w", err) + } + for rows.Next() { + var ( + group graph.StructuralIntegrityExactGroup + kind string + reason string + ) + if err := rows.Scan(&group.Repo, &kind, &reason, &group.Origin, &group.Count); err != nil { + _ = rows.Close() + return out, fmt.Errorf("store_sqlite: structural integrity group scan: %w", err) + } + group.Kind = graph.EdgeKind(kind) + group.Reason = graph.StructuralDropReason(reason) + out.TotalRows += group.Count + out.Groups = append(out.Groups, group) + } + if err := rows.Err(); err != nil { + _ = rows.Close() + return out, fmt.Errorf("store_sqlite: structural integrity group rows: %w", err) + } + if err := rows.Close(); err != nil { + return out, fmt.Errorf("store_sqlite: structural integrity group close: %w", err) + } + if err := ctx.Err(); err != nil { + return out, err + } + + limit := opts.SampleLimit + if limit <= 0 { + limit = structuralExactDefaultSamples + } + if limit > structuralExactMaxSamples { + limit = structuralExactMaxSamples + } + sampleSQL := `SELECT ` + structuralAuditRepoExpr + `, e.kind, ` + structuralAuditReasonExpr + `, ` + structuralAuditOriginExpr + `, e.from_id, e.to_id, e.file_path, e.line` + + structuralAuditFromSQL + scopeSQL + ` ORDER BY 1, 2, 3, 4, e.from_id, e.to_id, e.file_path, e.line, e.id LIMIT ?` + args := structuralAuditArgs(scopeArgs) + args = append(args, limit+1) + rows, err = s.db.QueryContext(ctx, sampleSQL, args...) + if err != nil { + return out, fmt.Errorf("store_sqlite: structural integrity samples: %w", err) + } + for rows.Next() { + var ( + sample graph.StructuralIntegrityExactSample + kind string + reason string + ) + if err := rows.Scan(&sample.Repo, &kind, &reason, &sample.Origin, &sample.From, &sample.To, &sample.FilePath, &sample.Line); err != nil { + _ = rows.Close() + return out, fmt.Errorf("store_sqlite: structural integrity sample scan: %w", err) + } + sample.Kind = graph.EdgeKind(kind) + sample.Reason = graph.StructuralDropReason(reason) + out.Samples = append(out.Samples, sample) + } + if err := rows.Err(); err != nil { + _ = rows.Close() + return out, fmt.Errorf("store_sqlite: structural integrity sample rows: %w", err) + } + if err := rows.Close(); err != nil { + return out, fmt.Errorf("store_sqlite: structural integrity sample close: %w", err) + } + if len(out.Samples) > limit { + out.Samples = out.Samples[:limit] + out.Truncated = true + } + if err := ctx.Err(); err != nil { + return out, err + } + return out, nil +} diff --git a/internal/graph/store_sqlite/structural_integrity_test.go b/internal/graph/store_sqlite/structural_integrity_test.go new file mode 100644 index 000000000..9726123de --- /dev/null +++ b/internal/graph/store_sqlite/structural_integrity_test.go @@ -0,0 +1,172 @@ +package store_sqlite + +import ( + "context" + "errors" + "path/filepath" + "testing" + + "github.com/zzet/gortex/internal/graph" +) + +func openStructuralIntegrityTestStore(t *testing.T, name string) *Store { + t.Helper() + store, err := Open(filepath.Join(t.TempDir(), name+".sqlite")) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { + if err := store.Close(); err != nil { + t.Errorf("close store: %v", err) + } + }) + return store +} + +func rawInsertStructuralViolation(t *testing.T, store *Store, from, to string, kind graph.EdgeKind, origin, file string, line int) { + t.Helper() + _, err := store.writerDB.Exec(`INSERT INTO edges + (from_id, to_id, kind, file_path, line, confidence, confidence_label, origin, tier, cross_repo, meta) + VALUES (?, ?, ?, ?, ?, 1.0, 'EXTRACTED', ?, '', 0, NULL)`, from, to, string(kind), file, line, origin) + if err != nil { + t.Fatal(err) + } +} + +func TestSQLiteStructuralWriteRejectionsExactOnceAndAttributed(t *testing.T) { + store := openStructuralIntegrityTestStore(t, "writes") + store.AddNode(&graph.Node{ID: "a/source", Kind: graph.KindFunction, RepoPrefix: "repo-a", FilePath: "a.go"}) + store.AddEdge(&graph.Edge{From: "a/source", To: "a/target#param:x", Kind: graph.EdgeImplements, Origin: "LSP_DISPATCH"}) + store.AddEdge(nil) + store.AddBatch( + []*graph.Node{{ID: "b/source", Kind: graph.KindFunction, RepoPrefix: "repo-b", FilePath: "b.go"}}, + []*graph.Edge{ + {From: "b/source", To: "b/target#local:y", Kind: graph.EdgeOverrides, Origin: "AST"}, + {From: "b/source", To: "b/target#param:ok", Kind: graph.EdgeCalls, Origin: "AST"}, + {From: "b/source", To: "b/target#local:ok", Kind: graph.EdgeReferences, Origin: "AST"}, + }, + ) + + snapshot := store.StructuralIntegritySnapshot(graph.StructuralIntegritySnapshotOptions{IncludeAttribution: true, IncludeSamples: true}) + if snapshot.Totals.WriteRejected != 2 || snapshot.Totals.ReadSuppressed != 0 { + t.Fatalf("unexpected write totals: %+v", snapshot.Totals) + } + if len(snapshot.Attribution) != 2 { + t.Fatalf("want one AddEdge and one AddBatch dimension: %+v", snapshot.Attribution) + } + if snapshot.Attribution[0].Repo != "repo-a" || snapshot.Attribution[0].Path != graph.StructuralPathSQLiteAddEdge || snapshot.Attribution[0].Count != 1 { + t.Fatalf("AddEdge attribution mismatch: %+v", snapshot.Attribution[0]) + } + if snapshot.Attribution[1].Repo != "repo-b" || snapshot.Attribution[1].Path != graph.StructuralPathSQLiteAddBatch || snapshot.Attribution[1].Count != 1 { + t.Fatalf("AddBatch attribution mismatch: %+v", snapshot.Attribution[1]) + } + if !store.structuralWriteWarned.Load() { + t.Fatal("first write rejection must trigger this store's rate-limited warning") + } + edges := store.AllEdges() + if len(edges) != 2 || edges[0].Kind != graph.EdgeCalls || edges[1].Kind != graph.EdgeReferences { + t.Fatalf("valid call/reference edges to params/locals must survive: %+v", edges) + } + + other := openStructuralIntegrityTestStore(t, "isolated") + if got := other.StructuralIntegritySnapshot(graph.StructuralIntegritySnapshotOptions{}).Totals; !got.Empty() { + t.Fatalf("store telemetry leaked across instances: %+v", got) + } + if other.structuralWriteWarned.Load() || other.structuralReadWarned.Load() { + t.Fatal("warning rate limits must be store-scoped") + } +} + +func TestSQLiteStructuralReadSuppressionAndExactAudit(t *testing.T) { + store := openStructuralIntegrityTestStore(t, "read-audit") + store.AddNode(&graph.Node{ID: "a/source", Kind: graph.KindFunction, RepoPrefix: "repo-a", FilePath: "a.go"}) + rawInsertStructuralViolation(t, store, "a/source", "a/target#param:x", graph.EdgeImplements, " LSP_DISPATCH ", "a.go", 11) + + if got := store.AllEdges(); len(got) != 0 { + t.Fatalf("full read materialized structurally invalid row: %+v", got) + } + if got := store.AllEdges(); len(got) != 0 { + t.Fatalf("repeated full read materialized structurally invalid row: %+v", got) + } + if got := store.AllEdgesLight(); len(got) != 0 { + t.Fatalf("light read materialized structurally invalid row: %+v", got) + } + + snapshot := store.StructuralIntegritySnapshot(graph.StructuralIntegritySnapshotOptions{IncludeAttribution: true, IncludeSamples: true}) + if snapshot.Totals.ReadSuppressed != 3 || snapshot.Totals.WriteRejected != 0 { + t.Fatalf("read observations must count every suppressed observation: %+v", snapshot.Totals) + } + if len(snapshot.Attribution) != 2 || snapshot.Attribution[0].Count != 2 || snapshot.Attribution[1].Count != 1 { + t.Fatalf("full/light read attribution mismatch: %+v", snapshot.Attribution) + } + for _, dimension := range snapshot.Attribution { + if dimension.Repo != "unknown" { + t.Fatalf("routine scanner must not guess repository ownership: %+v", dimension) + } + } + if !store.structuralReadWarned.Load() { + t.Fatal("first read suppression must trigger this store's rate-limited warning") + } + + audit, err := store.AuditStructuralIntegrity(context.Background(), graph.StructuralIntegrityAuditOptions{SampleLimit: 10}) + if err != nil { + t.Fatal(err) + } + if audit.Status != graph.StructuralAuditSupported || audit.TotalRows != 1 || len(audit.Groups) != 1 || len(audit.Samples) != 1 { + t.Fatalf("unexpected exact audit: %+v", audit) + } + group := audit.Groups[0] + if group.Repo != "repo-a" || group.Kind != graph.EdgeImplements || group.Reason != graph.StructuralReasonParameterTarget || group.Origin != "lsp_dispatch" || group.Count != 1 { + t.Fatalf("exact persisted attribution mismatch: %+v", group) + } +} + +func TestSQLiteStructuralExactAuditScopeCapAndCancellation(t *testing.T) { + store := openStructuralIntegrityTestStore(t, "scope") + store.AddBatch([]*graph.Node{ + {ID: "a/source", Kind: graph.KindFunction, RepoPrefix: "repo-a"}, + {ID: "b/source", Kind: graph.KindFunction, RepoPrefix: "repo-b"}, + }, nil) + rawInsertStructuralViolation(t, store, "a/source", "a/z#param:x", graph.EdgeImplements, "BETA", "z.go", 3) + rawInsertStructuralViolation(t, store, "a/source", "a/a#local:y", graph.EdgeExtends, "ALPHA", "a.go", 1) + rawInsertStructuralViolation(t, store, "a/source", "a/m#param:z", graph.EdgeOverrides, "GAMMA", "m.go", 2) + rawInsertStructuralViolation(t, store, "b/source", "b/a#param:q", graph.EdgeMemberOf, "OTHER", "b.go", 1) + + audit, err := store.AuditStructuralIntegrity(context.Background(), graph.StructuralIntegrityAuditOptions{ + RepoPrefixes: []string{"repo-a"}, RepoScopeActive: true, SampleLimit: 2, + }) + if err != nil { + t.Fatal(err) + } + if audit.TotalRows != 3 || len(audit.Samples) != 2 || !audit.Truncated { + t.Fatalf("scope/cap mismatch: %+v", audit) + } + for _, group := range audit.Groups { + if group.Repo != "repo-a" { + t.Fatalf("group leaked out-of-scope repository: %+v", group) + } + } + for _, sample := range audit.Samples { + if sample.Repo != "repo-a" { + t.Fatalf("sample leaked out-of-scope repository: %+v", sample) + } + } + if audit.Samples[0].Origin > audit.Samples[1].Origin { + t.Fatalf("samples are not deterministically ordered: %+v", audit.Samples) + } + + empty, err := store.AuditStructuralIntegrity(context.Background(), graph.StructuralIntegrityAuditOptions{RepoScopeActive: true}) + if err != nil { + t.Fatal(err) + } + if empty.TotalRows != 0 || len(empty.Groups) != 0 || len(empty.Samples) != 0 { + t.Fatalf("active empty scope must match nothing: %+v", empty) + } + + ctx, cancel := context.WithCancel(context.Background()) + cancel() + _, err = store.AuditStructuralIntegrity(ctx, graph.StructuralIntegrityAuditOptions{}) + if !errors.Is(err, context.Canceled) { + t.Fatalf("audit must honor cancellation, got %v", err) + } +} diff --git a/internal/graph/storetest/backend_resolver.go b/internal/graph/storetest/backend_resolver.go deleted file mode 100644 index 2400de99c..000000000 --- a/internal/graph/storetest/backend_resolver.go +++ /dev/null @@ -1,272 +0,0 @@ -package storetest - -import ( - "testing" - - "github.com/zzet/gortex/internal/graph" -) - -// RunBackendResolverConformance exercises every method of the -// graph.BackendResolver interface against a Factory that produces a -// store implementing both graph.Store and graph.BackendResolver. The -// shape mirrors RunConformance (the main Store contract): a known -// fixture graph, run the rule, assert the post-state matches the -// expected resolution. -// -// Backends that haven't implemented a rule yet ship the Phase 1 stub -// that returns (0, nil); those subtests pass trivially because the -// fixture also asserts zero-progress doesn't break correctness. -func RunBackendResolverConformance(t *testing.T, factory Factory) { - t.Helper() - t.Run("BackendResolver_SameFile", func(t *testing.T) { testBRSameFile(t, factory) }) - t.Run("BackendResolver_SamePackage", func(t *testing.T) { testBRSamePackage(t, factory) }) - t.Run("BackendResolver_ImportAware", func(t *testing.T) { testBRImportAware(t, factory) }) - t.Run("BackendResolver_RelativeImports", func(t *testing.T) { testBRRelativeImports(t, factory) }) - t.Run("BackendResolver_CrossRepo", func(t *testing.T) { testBRCrossRepo(t, factory) }) - t.Run("BackendResolver_UniqueNames", func(t *testing.T) { testBRUniqueNames(t, factory) }) - t.Run("BackendResolver_ExternalCallStubs", func(t *testing.T) { testBRExternalCallStubs(t, factory) }) - t.Run("BackendResolver_AllBulk", func(t *testing.T) { testBRAllBulk(t, factory) }) -} - -func asBackendResolver(t *testing.T, s graph.Store) graph.BackendResolver { - t.Helper() - br, ok := s.(graph.BackendResolver) - if !ok { - t.Skip("store does not implement graph.BackendResolver") - } - return br -} - -func testBRSameFile(t *testing.T, factory Factory) { - t.Helper() - s := factory(t) - br := asBackendResolver(t, s) - // caller and target in same file — unambiguous match - s.AddNode(mkNode("a.go::Foo", "Foo", "a.go", graph.KindFunction)) - s.AddNode(mkNode("a.go::Bar", "Bar", "a.go", graph.KindFunction)) - s.AddEdge(&graph.Edge{ - From: "a.go::Foo", To: "unresolved::Bar", Kind: graph.EdgeCalls, - FilePath: "a.go", Line: 1, Origin: "", - }) - n, err := br.ResolveSameFile() - if err != nil { - t.Fatalf("ResolveSameFile: %v", err) - } - if n == 0 { - // stub backend — skip the post-state assertions - return - } - if n != 1 { - t.Fatalf("ResolveSameFile resolved %d, want 1", n) - } - // edge should now point at a.go::Bar with origin ast_resolved - got := s.GetOutEdges("a.go::Foo") - if len(got) != 1 || got[0].To != "a.go::Bar" || got[0].Origin != graph.OriginASTResolved { - t.Fatalf("ResolveSameFile post-state: edges=%+v", got) - } -} - -func testBRSamePackage(t *testing.T, factory Factory) { - t.Helper() - s := factory(t) - br := asBackendResolver(t, s) - // caller in pkg/a.go, target in pkg/b.go — same directory - s.AddNode(mkRepoNode("pkg/a.go::Caller", "Caller", "pkg/a.go", "r1", graph.KindFunction)) - s.AddNode(mkRepoNode("pkg/b.go::Target", "Target", "pkg/b.go", "r1", graph.KindFunction)) - s.AddEdge(&graph.Edge{ - From: "pkg/a.go::Caller", To: "unresolved::Target", Kind: graph.EdgeCalls, - FilePath: "pkg/a.go", Line: 1, Origin: "", - }) - n, err := br.ResolveSamePackage() - if err != nil { - t.Fatalf("ResolveSamePackage: %v", err) - } - if n == 0 { - return - } - if n != 1 { - t.Fatalf("ResolveSamePackage resolved %d, want 1", n) - } - got := s.GetOutEdges("pkg/a.go::Caller") - if len(got) != 1 || got[0].To != "pkg/b.go::Target" || got[0].Origin != graph.OriginASTResolved { - t.Fatalf("ResolveSamePackage post-state: edges=%+v", got) - } -} - -func testBRImportAware(t *testing.T, factory Factory) { - t.Helper() - s := factory(t) - br := asBackendResolver(t, s) - // caller.go imports lib.go which exports Target - s.AddNode(mkNode("caller.go", "caller.go", "caller.go", graph.KindFile)) - s.AddNode(mkNode("lib.go", "lib.go", "lib.go", graph.KindFile)) - s.AddNode(mkNode("caller.go::Caller", "Caller", "caller.go", graph.KindFunction)) - s.AddNode(mkNode("lib.go::Target", "Target", "lib.go", graph.KindFunction)) - // the imports edge - s.AddEdge(&graph.Edge{ - From: "caller.go", To: "lib.go", Kind: graph.EdgeImports, - FilePath: "caller.go", Line: 1, Origin: graph.OriginASTResolved, - }) - // the unresolved call - s.AddEdge(&graph.Edge{ - From: "caller.go::Caller", To: "unresolved::Target", Kind: graph.EdgeCalls, - FilePath: "caller.go", Line: 5, Origin: "", - }) - n, err := br.ResolveImportAware() - if err != nil { - t.Fatalf("ResolveImportAware: %v", err) - } - if n == 0 { - return - } - if n != 1 { - t.Fatalf("ResolveImportAware resolved %d, want 1", n) - } - got := s.GetOutEdges("caller.go::Caller") - var found bool - for _, e := range got { - if e.To == "lib.go::Target" { - found = true - } - } - if !found { - t.Fatalf("ResolveImportAware post-state: edges=%+v, want one to lib.go::Target", got) - } -} - -func testBRRelativeImports(t *testing.T, factory Factory) { - t.Helper() - s := factory(t) - br := asBackendResolver(t, s) - // python relative-import stub - s.AddNode(mkNode("app/util.py", "app/util.py", "app/util.py", graph.KindFile)) - s.AddNode(mkNode("app/main.py", "app/main.py", "app/main.py", graph.KindFile)) - s.AddEdge(&graph.Edge{ - From: "app/main.py", To: "unresolved::pyrel::app/util", Kind: graph.EdgeImports, - FilePath: "app/main.py", Line: 1, Origin: "", - }) - n, err := br.ResolveRelativeImports("python") - if err != nil { - t.Fatalf("ResolveRelativeImports: %v", err) - } - if n == 0 { - return - } - if n != 1 { - t.Fatalf("ResolveRelativeImports resolved %d, want 1", n) - } - got := s.GetOutEdges("app/main.py") - var found bool - for _, e := range got { - if e.To == "app/util.py" { - found = true - } - } - if !found { - t.Fatalf("ResolveRelativeImports post-state: edges=%+v, want one to app/util.py", got) - } -} - -func testBRCrossRepo(t *testing.T, factory Factory) { - t.Helper() - s := factory(t) - br := asBackendResolver(t, s) - s.AddNode(mkRepoNode("r1/a.go::Caller", "Caller", "r1/a.go", "r1", graph.KindFunction)) - s.AddNode(mkRepoNode("r2/x.go::Target", "Target", "r2/x.go", "r2", graph.KindFunction)) - s.AddEdge(&graph.Edge{ - From: "r1/a.go::Caller", To: "unresolved::Target", Kind: graph.EdgeCalls, - FilePath: "r1/a.go", Line: 1, Origin: "", - }) - n, err := br.ResolveCrossRepo() - if err != nil { - t.Fatalf("ResolveCrossRepo: %v", err) - } - if n == 0 { - return - } - if n != 1 { - t.Fatalf("ResolveCrossRepo resolved %d, want 1", n) - } - got := s.GetOutEdges("r1/a.go::Caller") - if len(got) != 1 || got[0].To != "r2/x.go::Target" || !got[0].CrossRepo { - t.Fatalf("ResolveCrossRepo post-state: edges=%+v", got) - } -} - -func testBRUniqueNames(t *testing.T, factory Factory) { - t.Helper() - s := factory(t) - br := asBackendResolver(t, s) - // One unique-name candidate in the graph. - s.AddNode(mkNode("a.go::Foo", "Foo", "a.go", graph.KindFunction)) - s.AddNode(mkNode("b.go::Target", "Target", "b.go", graph.KindFunction)) - s.AddEdge(&graph.Edge{ - From: "a.go::Foo", To: "unresolved::Target", Kind: graph.EdgeCalls, - FilePath: "a.go", Line: 1, Origin: "", - }) - n, err := br.ResolveUniqueNames() - if err != nil { - t.Fatalf("ResolveUniqueNames: %v", err) - } - if n == 0 { - return - } - if n != 1 { - t.Fatalf("ResolveUniqueNames resolved %d, want 1", n) - } - got := s.GetOutEdges("a.go::Foo") - if len(got) != 1 || got[0].To != "b.go::Target" { - t.Fatalf("ResolveUniqueNames post-state: edges=%+v", got) - } -} - -func testBRExternalCallStubs(t *testing.T, factory Factory) { - t.Helper() - s := factory(t) - br := asBackendResolver(t, s) - s.AddNode(mkNode("a.go::Caller", "Caller", "a.go", graph.KindFunction)) - // edge to external::npm/foo::bar with no stub node - s.AddEdge(&graph.Edge{ - From: "a.go::Caller", To: "external::npm/foo::bar", Kind: graph.EdgeCalls, - FilePath: "a.go", Line: 1, Origin: "", - }) - n, err := br.ResolveExternalCallStubs() - if err != nil { - t.Fatalf("ResolveExternalCallStubs: %v", err) - } - if n == 0 { - return - } - if n < 1 { - t.Fatalf("ResolveExternalCallStubs resolved %d, want >= 1", n) - } - // stub node must now exist - if s.GetNode("external::npm/foo::bar") == nil { - t.Fatalf("external stub node not created") - } -} - -func testBRAllBulk(t *testing.T, factory Factory) { - t.Helper() - s := factory(t) - br := asBackendResolver(t, s) - // Mix of resolvable + stub cases. - s.AddNode(mkNode("a.go::Foo", "Foo", "a.go", graph.KindFunction)) - s.AddNode(mkNode("a.go::Bar", "Bar", "a.go", graph.KindFunction)) - s.AddNode(mkNode("b.go::Unique", "Unique", "b.go", graph.KindFunction)) - // same-file - s.AddEdge(&graph.Edge{ - From: "a.go::Foo", To: "unresolved::Bar", Kind: graph.EdgeCalls, - FilePath: "a.go", Line: 1, Origin: "", - }) - // unique-name - s.AddEdge(&graph.Edge{ - From: "a.go::Foo", To: "unresolved::Unique", Kind: graph.EdgeCalls, - FilePath: "a.go", Line: 2, Origin: "", - }) - n, err := br.ResolveAllBulk() - if err != nil { - t.Fatalf("ResolveAllBulk: %v", err) - } - _ = n // 0 on stub backends, >0 on real -} diff --git a/internal/graph/structural_edge_gate_test.go b/internal/graph/structural_edge_gate_test.go index f6403e12f..995ad6094 100644 --- a/internal/graph/structural_edge_gate_test.go +++ b/internal/graph/structural_edge_gate_test.go @@ -38,7 +38,8 @@ func TestFilterStructuralEdgeViolationsCopiesOnlyOnViolation(t *testing.T) { {From: "a::T3", To: "a::I", Kind: EdgeExtends}, } kept, dropped = FilterStructuralEdgeViolations(mixed) - assert.Equal(t, 1, dropped) + require.Len(t, dropped, 1) + assert.Same(t, mixed[1], dropped[0]) require.Len(t, kept, 2) assert.Equal(t, "a::I", kept[0].To) assert.Equal(t, "a::I", kept[1].To) diff --git a/internal/graph/structural_integrity.go b/internal/graph/structural_integrity.go new file mode 100644 index 000000000..f1d3b4be6 --- /dev/null +++ b/internal/graph/structural_integrity.go @@ -0,0 +1,504 @@ +package graph + +import ( + "context" + "sort" + "strings" + "sync" + "sync/atomic" +) + +const ( + structuralIntegrityMaxRepos = 256 + structuralIntegrityMaxDimensions = 256 + structuralIntegrityMaxSamples = 64 + structuralIntegrityOriginLimit = 48 + structuralIntegrityDetailLimit = 512 +) + +type StructuralDropDirection string + +const ( + StructuralDropWrite StructuralDropDirection = "write_rejection" + StructuralDropRead StructuralDropDirection = "read_suppression" +) + +type StructuralDropPath string + +const ( + StructuralPathUnknown StructuralDropPath = "unknown" + StructuralPathGraphAddEdge StructuralDropPath = "graph_add_edge" + StructuralPathGraphAddBatch StructuralDropPath = "graph_add_batch" + StructuralPathShadowCold StructuralDropPath = "shadow_cold_batch" + StructuralPathShadowStreaming StructuralDropPath = "shadow_streaming_batch" + StructuralPathSQLiteAddEdge StructuralDropPath = "sqlite_add_edge" + StructuralPathSQLiteAddBatch StructuralDropPath = "sqlite_add_batch" + StructuralPathSQLiteFullRead StructuralDropPath = "sqlite_full" + StructuralPathSQLiteLightRead StructuralDropPath = "sqlite_light" +) + +type StructuralDropReason string + +const ( + StructuralReasonParameterTarget StructuralDropReason = "parameter_target" + StructuralReasonLocalTarget StructuralDropReason = "local_target" +) + +type StructuralIntegrityEvent struct { + Direction StructuralDropDirection + Path StructuralDropPath + Repo string + Kind EdgeKind + Reason StructuralDropReason + Origin string + From string + To string + FilePath string + Line int +} + +type StructuralIntegrityTotals struct { + WriteRejected uint64 `json:"write_rejected,omitempty"` + ReadSuppressed uint64 `json:"read_suppressed,omitempty"` +} + +func (t StructuralIntegrityTotals) Empty() bool { + return t.WriteRejected == 0 && t.ReadSuppressed == 0 +} + +type StructuralIntegrityAttribution struct { + Direction StructuralDropDirection `json:"direction"` + Path StructuralDropPath `json:"path"` + Repo string `json:"repo"` + Kind EdgeKind `json:"kind"` + Reason StructuralDropReason `json:"reason"` + Origin string `json:"origin"` + Count uint64 `json:"count"` +} + +type StructuralIntegritySample struct { + Direction StructuralDropDirection `json:"direction"` + Path StructuralDropPath `json:"path"` + Repo string `json:"repo"` + Kind EdgeKind `json:"kind"` + Reason StructuralDropReason `json:"reason"` + Origin string `json:"origin"` + From string `json:"from,omitempty"` + To string `json:"to,omitempty"` + FilePath string `json:"file_path,omitempty"` + Line int `json:"line,omitempty"` +} + +type StructuralIntegritySnapshotOptions struct { + RepoPrefixes []string + RepoScopeActive bool + IncludeAttribution bool + IncludeSamples bool +} + +type StructuralIntegritySnapshot struct { + Totals StructuralIntegrityTotals `json:"totals"` + Attribution []StructuralIntegrityAttribution `json:"attribution,omitempty"` + Samples []StructuralIntegritySample `json:"samples,omitempty"` + AttributionOverflow uint64 `json:"attribution_overflow,omitempty"` + SampleOverflow uint64 `json:"sample_overflow,omitempty"` + RepoTotalsOverflow uint64 `json:"repo_totals_overflow,omitempty"` + TotalsLowerBound bool `json:"totals_lower_bound,omitempty"` +} + +type StructuralIntegrityAuditOptions struct { + RepoPrefixes []string + RepoScopeActive bool + SampleLimit int +} + +type StructuralIntegrityExactGroup struct { + Repo string `json:"repo"` + Kind EdgeKind `json:"kind"` + Reason StructuralDropReason `json:"reason"` + Origin string `json:"origin"` + Count uint64 `json:"count"` +} + +type StructuralIntegrityExactSample struct { + Repo string `json:"repo"` + Kind EdgeKind `json:"kind"` + Reason StructuralDropReason `json:"reason"` + Origin string `json:"origin"` + From string `json:"from"` + To string `json:"to"` + FilePath string `json:"file_path,omitempty"` + Line int `json:"line,omitempty"` +} + +type StructuralIntegrityExactAudit struct { + Status string `json:"status"` + TotalRows uint64 `json:"total_rows"` + Groups []StructuralIntegrityExactGroup `json:"groups,omitempty"` + Samples []StructuralIntegrityExactSample `json:"samples,omitempty"` + Truncated bool `json:"truncated,omitempty"` +} + +const ( + StructuralAuditSupported = "supported" + StructuralAuditUnsupported = "unsupported" +) + +type StructuralIntegrityEventRecorder interface { + RecordStructuralIntegrityEvent(StructuralIntegrityEvent) +} + +type StructuralIntegritySnapshotter interface { + StructuralIntegritySnapshot(StructuralIntegritySnapshotOptions) StructuralIntegritySnapshot +} + +type StructuralIntegrityAuditor interface { + AuditStructuralIntegrity(context.Context, StructuralIntegrityAuditOptions) (StructuralIntegrityExactAudit, error) +} + +type structuralIntegrityKey struct { + direction StructuralDropDirection + path StructuralDropPath + repo string + kind EdgeKind + reason StructuralDropReason + origin string +} + +type structuralIntegritySampleKey struct { + structuralIntegrityKey + from string + to string + filePath string + line int +} + +type StructuralIntegrityMeter struct { + writeRejected atomic.Uint64 + readSuppressed atomic.Uint64 + + mu sync.Mutex + repoTotals map[string]StructuralIntegrityTotals + attribution map[structuralIntegrityKey]uint64 + samples map[structuralIntegritySampleKey]StructuralIntegritySample + repoTotalsOverflow uint64 + attributionOverflow uint64 + sampleOverflow uint64 +} + +func normalizeStructuralRepo(repo string) string { + repo = strings.TrimSpace(repo) + if repo == "" { + return "unknown" + } + return boundedStructuralString(repo, 128) +} + +func normalizeStructuralOrigin(origin string) string { + origin = strings.ToLower(strings.TrimSpace(origin)) + if origin == "" { + return "unknown" + } + return boundedStructuralString(origin, structuralIntegrityOriginLimit) +} + +func boundedStructuralString(value string, limit int) string { + if limit <= 0 { + return "" + } + runes := []rune(value) + if len(runes) > limit { + runes = runes[:limit] + } + return string(runes) +} + +func normalizeStructuralPath(path StructuralDropPath) StructuralDropPath { + switch path { + case StructuralPathGraphAddEdge, StructuralPathGraphAddBatch, + StructuralPathShadowCold, StructuralPathShadowStreaming, + StructuralPathSQLiteAddEdge, StructuralPathSQLiteAddBatch, + StructuralPathSQLiteFullRead, StructuralPathSQLiteLightRead: + return path + default: + return StructuralPathUnknown + } +} + +func validStructuralKind(kind EdgeKind) bool { + switch kind { + case EdgeImplements, EdgeExtends, EdgeOverrides, EdgeInstantiates, EdgeMemberOf: + return true + default: + return false + } +} + +func structuralReasonForTarget(target string) (StructuralDropReason, bool) { + switch { + case strings.Contains(target, "#param:"): + return StructuralReasonParameterTarget, true + case strings.Contains(target, "#local:"): + return StructuralReasonLocalTarget, true + default: + return "", false + } +} + +func StructuralIntegrityEventForEdge(direction StructuralDropDirection, path StructuralDropPath, repo string, edge *Edge) (StructuralIntegrityEvent, bool) { + if edge == nil || !validStructuralKind(edge.Kind) { + return StructuralIntegrityEvent{}, false + } + reason, ok := structuralReasonForTarget(edge.To) + if !ok { + return StructuralIntegrityEvent{}, false + } + return StructuralIntegrityEvent{ + Direction: direction, + Path: normalizeStructuralPath(path), + Repo: normalizeStructuralRepo(repo), + Kind: edge.Kind, + Reason: reason, + Origin: edge.Origin, + From: edge.From, + To: edge.To, + FilePath: edge.FilePath, + Line: edge.Line, + }, true +} + +func (m *StructuralIntegrityMeter) Record(event StructuralIntegrityEvent) { + if m == nil || !validStructuralKind(event.Kind) { + return + } + if event.Direction != StructuralDropWrite && event.Direction != StructuralDropRead { + return + } + if event.Reason != StructuralReasonParameterTarget && event.Reason != StructuralReasonLocalTarget { + return + } + event.Path = normalizeStructuralPath(event.Path) + event.Repo = normalizeStructuralRepo(event.Repo) + event.Origin = normalizeStructuralOrigin(event.Origin) + event.From = boundedStructuralString(event.From, structuralIntegrityDetailLimit) + event.To = boundedStructuralString(event.To, structuralIntegrityDetailLimit) + event.FilePath = boundedStructuralString(event.FilePath, structuralIntegrityDetailLimit) + + key := structuralIntegrityKey{ + direction: event.Direction, + path: event.Path, + repo: event.Repo, + kind: event.Kind, + reason: event.Reason, + origin: event.Origin, + } + sampleKey := structuralIntegritySampleKey{ + structuralIntegrityKey: key, + from: event.From, to: event.To, filePath: event.FilePath, line: event.Line, + } + + m.mu.Lock() + defer m.mu.Unlock() + if event.Direction == StructuralDropWrite { + m.writeRejected.Add(1) + } else { + m.readSuppressed.Add(1) + } + if m.repoTotals == nil { + m.repoTotals = make(map[string]StructuralIntegrityTotals) + } + if totals, exists := m.repoTotals[event.Repo]; exists || len(m.repoTotals) < structuralIntegrityMaxRepos { + if event.Direction == StructuralDropWrite { + totals.WriteRejected++ + } else { + totals.ReadSuppressed++ + } + m.repoTotals[event.Repo] = totals + } else { + m.repoTotalsOverflow++ + } + if m.attribution == nil { + m.attribution = make(map[structuralIntegrityKey]uint64) + } + if _, exists := m.attribution[key]; exists || len(m.attribution) < structuralIntegrityMaxDimensions { + m.attribution[key]++ + } else { + m.attributionOverflow++ + } + if m.samples == nil { + m.samples = make(map[structuralIntegritySampleKey]StructuralIntegritySample) + } + if _, exists := m.samples[sampleKey]; exists { + return + } + if len(m.samples) >= structuralIntegrityMaxSamples { + m.sampleOverflow++ + return + } + m.samples[sampleKey] = StructuralIntegritySample(event) +} + +func (m *StructuralIntegrityMeter) Snapshot(opts StructuralIntegritySnapshotOptions) StructuralIntegritySnapshot { + if m == nil { + return StructuralIntegritySnapshot{} + } + allowed := make(map[string]struct{}, len(opts.RepoPrefixes)) + for _, repo := range opts.RepoPrefixes { + allowed[normalizeStructuralRepo(repo)] = struct{}{} + } + scoped := opts.RepoScopeActive || len(opts.RepoPrefixes) > 0 + matches := func(repo string) bool { + if !scoped { + return true + } + _, ok := allowed[repo] + return ok + } + + out := StructuralIntegritySnapshot{} + m.mu.Lock() + out.AttributionOverflow = m.attributionOverflow + out.SampleOverflow = m.sampleOverflow + out.RepoTotalsOverflow = m.repoTotalsOverflow + if opts.IncludeAttribution { + for key, count := range m.attribution { + if !matches(key.repo) { + continue + } + out.Attribution = append(out.Attribution, StructuralIntegrityAttribution{ + Direction: key.direction, + Path: key.path, + Repo: key.repo, + Kind: key.kind, + Reason: key.reason, + Origin: key.origin, + Count: count, + }) + } + } + if !scoped { + out.Totals.WriteRejected = m.writeRejected.Load() + out.Totals.ReadSuppressed = m.readSuppressed.Load() + } else { + for repo := range allowed { + totals := m.repoTotals[repo] + out.Totals.WriteRejected += totals.WriteRejected + out.Totals.ReadSuppressed += totals.ReadSuppressed + } + out.TotalsLowerBound = m.repoTotalsOverflow > 0 + } + if opts.IncludeSamples { + for _, sample := range m.samples { + if matches(sample.Repo) { + out.Samples = append(out.Samples, sample) + } + } + } + m.mu.Unlock() + + sort.Slice(out.Attribution, func(i, j int) bool { + a, b := out.Attribution[i], out.Attribution[j] + if a.Repo != b.Repo { + return a.Repo < b.Repo + } + if a.Direction != b.Direction { + return a.Direction < b.Direction + } + if a.Path != b.Path { + return a.Path < b.Path + } + if a.Kind != b.Kind { + return a.Kind < b.Kind + } + if a.Reason != b.Reason { + return a.Reason < b.Reason + } + return a.Origin < b.Origin + }) + sort.Slice(out.Samples, func(i, j int) bool { + a, b := out.Samples[i], out.Samples[j] + if a.Repo != b.Repo { + return a.Repo < b.Repo + } + if a.Direction != b.Direction { + return a.Direction < b.Direction + } + if a.Path != b.Path { + return a.Path < b.Path + } + if a.Kind != b.Kind { + return a.Kind < b.Kind + } + if a.Reason != b.Reason { + return a.Reason < b.Reason + } + if a.Origin != b.Origin { + return a.Origin < b.Origin + } + if a.From != b.From { + return a.From < b.From + } + if a.To != b.To { + return a.To < b.To + } + if a.FilePath != b.FilePath { + return a.FilePath < b.FilePath + } + return a.Line < b.Line + }) + return out +} + +func (g *Graph) RecordStructuralIntegrityEvent(event StructuralIntegrityEvent) { + g.structuralIntegrity.Record(event) +} + +func (g *Graph) StructuralIntegritySnapshot(opts StructuralIntegritySnapshotOptions) StructuralIntegritySnapshot { + return g.structuralIntegrity.Snapshot(opts) +} + +func (g *Graph) recordStructuralRejections(path StructuralDropPath, edges []*Edge, nodes []*Node) { + if len(edges) == 0 { + return + } + if g.structuralIntegrityPath != "" { + path = g.structuralIntegrityPath + } + var sink StructuralIntegrityEventRecorder = g + if g.structuralIntegritySink != nil { + sink = g.structuralIntegritySink + } + configuredRepo := g.structuralIntegrityRepo + inputRepos := make(map[string]string) + if configuredRepo == "" || configuredRepo == "unknown" { + for _, node := range nodes { + if node != nil && node.ID != "" && node.RepoPrefix != "" { + inputRepos[node.ID] = node.RepoPrefix + } + } + } + for _, edge := range edges { + repo := configuredRepo + if repo == "" || repo == "unknown" { + repo = inputRepos[edge.From] + if repo == "" { + if source := g.GetNode(edge.From); source != nil { + repo = source.RepoPrefix + } + } + } + if event, ok := StructuralIntegrityEventForEdge(StructuralDropWrite, path, repo, edge); ok { + sink.RecordStructuralIntegrityEvent(event) + } + } +} + +func NewStructuralIntegrityShadow(owner Store, repo string, path StructuralDropPath) *Graph { + g := New() + if sink, ok := owner.(StructuralIntegrityEventRecorder); ok { + g.structuralIntegritySink = sink + } + g.structuralIntegrityRepo = normalizeStructuralRepo(repo) + g.structuralIntegrityPath = normalizeStructuralPath(path) + return g +} diff --git a/internal/graph/structural_integrity_test.go b/internal/graph/structural_integrity_test.go new file mode 100644 index 000000000..20a9e63a6 --- /dev/null +++ b/internal/graph/structural_integrity_test.go @@ -0,0 +1,181 @@ +package graph + +import ( + "fmt" + "strings" + "sync" + "testing" +) + +func invalidStructuralEdge(from, to string, kind EdgeKind, origin string) *Edge { + return &Edge{From: from, To: to, Kind: kind, Origin: origin, FilePath: "repo/file.go", Line: 7} +} + +func TestFilterStructuralEdgeViolationsPurePartition(t *testing.T) { + validCall := invalidStructuralEdge("src", "dst#param:x", EdgeCalls, "call") + invalidParam := invalidStructuralEdge("src", "dst#param:x", EdgeImplements, "lsp") + invalidLocal := invalidStructuralEdge("src", "dst#local:y", EdgeOverrides, "ast") + validStructural := invalidStructuralEdge("src", "dst", EdgeExtends, "ast") + input := []*Edge{nil, validCall, invalidParam, validStructural, invalidLocal} + + kept, rejected := FilterStructuralEdgeViolations(input) + if len(input) != 5 || input[2] != invalidParam || input[4] != invalidLocal { + t.Fatalf("filter mutated input: %#v", input) + } + if len(kept) != 3 || kept[0] != nil || kept[1] != validCall || kept[2] != validStructural { + t.Fatalf("unexpected kept partition: %#v", kept) + } + if len(rejected) != 2 || rejected[0] != invalidParam || rejected[1] != invalidLocal { + t.Fatalf("unexpected rejected partition: %#v", rejected) + } +} + +func TestGraphStructuralIntegrityExactOnceAndRepoAttribution(t *testing.T) { + g := New() + g.AddNode(&Node{ID: "repo/a.go::Source", Kind: KindFunction, RepoPrefix: "repo-a"}) + g.AddEdge(invalidStructuralEdge("repo/a.go::Source", "repo/a.go::Target#param:x", EdgeImplements, "LSP_DISPATCH")) + g.AddBatch( + []*Node{{ID: "repo/b.go::Source", Kind: KindFunction, RepoPrefix: "repo-b"}}, + []*Edge{ + invalidStructuralEdge("repo/b.go::Source", "repo/b.go::Target#local:y", EdgeOverrides, "AST"), + invalidStructuralEdge("repo/b.go::Source", "repo/b.go::Target#param:ok", EdgeCalls, "AST"), + }, + ) + + snapshot := g.StructuralIntegritySnapshot(StructuralIntegritySnapshotOptions{IncludeAttribution: true, IncludeSamples: true}) + if snapshot.Totals.WriteRejected != 2 || snapshot.Totals.ReadSuppressed != 0 { + t.Fatalf("unexpected totals: %+v", snapshot.Totals) + } + if len(snapshot.Attribution) != 2 { + t.Fatalf("want two attributed dimensions, got %+v", snapshot.Attribution) + } + if snapshot.Attribution[0].Repo != "repo-a" || snapshot.Attribution[0].Origin != "lsp_dispatch" { + t.Fatalf("existing source attribution mismatch: %+v", snapshot.Attribution[0]) + } + if snapshot.Attribution[1].Repo != "repo-b" || snapshot.Attribution[1].Origin != "ast" { + t.Fatalf("batch input attribution mismatch: %+v", snapshot.Attribution[1]) + } + if got := len(g.AllEdges()); got != 1 || g.AllEdges()[0].Kind != EdgeCalls { + t.Fatalf("valid call-to-param edge must survive; edges=%+v", g.AllEdges()) + } + + repoA := g.StructuralIntegritySnapshot(StructuralIntegritySnapshotOptions{ + RepoPrefixes: []string{"repo-a"}, RepoScopeActive: true, + IncludeAttribution: true, IncludeSamples: true, + }) + if repoA.Totals.WriteRejected != 1 || len(repoA.Attribution) != 1 || len(repoA.Samples) != 1 { + t.Fatalf("scoped snapshot leaked or undercounted: %+v", repoA) + } + if repoA.Attribution[0].Repo != "repo-a" || repoA.Samples[0].Repo != "repo-a" { + t.Fatalf("scoped snapshot leaked another repository: %+v", repoA) + } + emptyScope := g.StructuralIntegritySnapshot(StructuralIntegritySnapshotOptions{RepoScopeActive: true, IncludeSamples: true}) + if !emptyScope.Totals.Empty() || len(emptyScope.Samples) != 0 { + t.Fatalf("active empty scope must match nothing: %+v", emptyScope) + } +} + +func TestStructuralIntegrityShadowForwardsAttemptOnce(t *testing.T) { + owner := New() + shadow := NewStructuralIntegrityShadow(owner, "repo-shadow", StructuralPathShadowCold) + shadow.AddBatch(nil, []*Edge{ + invalidStructuralEdge("src", "dst#param:x", EdgeImplements, "LSP"), + invalidStructuralEdge("src", "dst#param:x", EdgeCalls, "LSP"), + }) + + ownerSnapshot := owner.StructuralIntegritySnapshot(StructuralIntegritySnapshotOptions{IncludeAttribution: true}) + if ownerSnapshot.Totals.WriteRejected != 1 || len(ownerSnapshot.Attribution) != 1 { + t.Fatalf("shadow rejection was not forwarded exactly once: %+v", ownerSnapshot) + } + got := ownerSnapshot.Attribution[0] + if got.Repo != "repo-shadow" || got.Path != StructuralPathShadowCold || got.Count != 1 { + t.Fatalf("shadow attribution mismatch: %+v", got) + } + if shadow.StructuralIntegritySnapshot(StructuralIntegritySnapshotOptions{}).Totals.WriteRejected != 0 { + t.Fatal("forwarding shadow must not retain a duplicate local event") + } + if gotEdges := shadow.AllEdges(); len(gotEdges) != 1 || gotEdges[0].Kind != EdgeCalls { + t.Fatalf("valid edge missing from shadow: %+v", gotEdges) + } +} + +func TestStructuralIntegrityMeterBoundsDimensionsAndSignalsScopedLowerBound(t *testing.T) { + var meter StructuralIntegrityMeter + longOrigin := " " + strings.Repeat("Ab", 40) + " " + for i := 0; i < structuralIntegrityMaxRepos+1; i++ { + origin := fmt.Sprintf("origin-%03d", i) + if i == 0 { + origin = longOrigin + } + meter.Record(StructuralIntegrityEvent{ + Direction: StructuralDropWrite, + Path: StructuralPathGraphAddEdge, + Repo: fmt.Sprintf("repo-%03d", i), + Kind: EdgeImplements, + Reason: StructuralReasonParameterTarget, + Origin: origin, + From: fmt.Sprintf("from-%03d", i), + To: fmt.Sprintf("to-%03d#param:x", i), + Line: i, + }) + } + + all := meter.Snapshot(StructuralIntegritySnapshotOptions{IncludeAttribution: true, IncludeSamples: true}) + if all.Totals.WriteRejected != structuralIntegrityMaxRepos+1 { + t.Fatalf("global atomics must remain exact: %+v", all.Totals) + } + if len(all.Attribution) != structuralIntegrityMaxDimensions || all.AttributionOverflow != 1 { + t.Fatalf("dimension bound mismatch: len=%d overflow=%d", len(all.Attribution), all.AttributionOverflow) + } + if len(all.Samples) != structuralIntegrityMaxSamples || all.SampleOverflow != structuralIntegrityMaxRepos+1-structuralIntegrityMaxSamples { + t.Fatalf("sample bound mismatch: len=%d overflow=%d", len(all.Samples), all.SampleOverflow) + } + if all.RepoTotalsOverflow != 1 { + t.Fatalf("repo total overflow not surfaced: %+v", all) + } + wantOrigin := strings.ToLower(strings.TrimSpace(longOrigin))[:structuralIntegrityOriginLimit] + if all.Attribution[0].Origin != wantOrigin { + t.Fatalf("origin must be normalized and bounded: got %q want %q", all.Attribution[0].Origin, wantOrigin) + } + + overflowedRepo := meter.Snapshot(StructuralIntegritySnapshotOptions{ + RepoPrefixes: []string{fmt.Sprintf("repo-%03d", structuralIntegrityMaxRepos)}, + RepoScopeActive: true, + }) + if overflowedRepo.Totals.WriteRejected != 0 || !overflowedRepo.TotalsLowerBound || overflowedRepo.RepoTotalsOverflow != 1 { + t.Fatalf("scoped undercount must be explicit: %+v", overflowedRepo) + } +} + +func TestStructuralIntegrityMeterConcurrentRecordAndSnapshot(t *testing.T) { + var meter StructuralIntegrityMeter + const goroutines = 16 + const perGoroutine = 200 + var wg sync.WaitGroup + wg.Add(goroutines) + for i := 0; i < goroutines; i++ { + go func(worker int) { + defer wg.Done() + for j := 0; j < perGoroutine; j++ { + meter.Record(StructuralIntegrityEvent{ + Direction: StructuralDropRead, + Path: StructuralPathSQLiteFullRead, + Repo: fmt.Sprintf("repo-%d", worker%4), + Kind: EdgeExtends, + Reason: StructuralReasonLocalTarget, + Origin: "legacy", + From: fmt.Sprintf("from-%d", worker), + To: fmt.Sprintf("to-%d#local:x", j), + }) + if j%17 == 0 { + _ = meter.Snapshot(StructuralIntegritySnapshotOptions{IncludeAttribution: true, IncludeSamples: true}) + } + } + }(i) + } + wg.Wait() + got := meter.Snapshot(StructuralIntegritySnapshotOptions{}) + if got.Totals.ReadSuppressed != goroutines*perGoroutine { + t.Fatalf("lost concurrent events: got %d want %d", got.Totals.ReadSuppressed, goroutines*perGoroutine) + } +} diff --git a/internal/graph/stub.go b/internal/graph/stub.go index 5969415a8..f7d542856 100644 --- a/internal/graph/stub.go +++ b/internal/graph/stub.go @@ -2,7 +2,6 @@ package graph import ( "strings" - "sync/atomic" ) // Stub-node identifier conventions. @@ -107,10 +106,8 @@ var stubKinds = []string{ // IsStdlibStub etc are convenience predicates that don't make // the caller compare StubKind's return against a literal. -func IsStdlibStub(id string) bool { return StubKind(id) == StubKindStdlib } -func IsBuiltinStub(id string) bool { return StubKind(id) == StubKindBuiltin } -func IsExternalCallStub(id string) bool { return StubKind(id) == StubKindExternalCall } -func IsModuleStub(id string) bool { return StubKind(id) == StubKindModule } +func IsStdlibStub(id string) bool { return StubKind(id) == StubKindStdlib } +func IsBuiltinStub(id string) bool { return StubKind(id) == StubKindBuiltin } // StubRest returns the kind-specific tail of a stub id (the // portion after "::::" or "::"). Returns "" if @@ -173,42 +170,24 @@ func StructuralEdgeTargetInvalid(kind EdgeKind, toID string) bool { return strings.Contains(toID, "#param:") || strings.Contains(toID, "#local:") } -// structuralWriteDrops counts edges the write funnels refused — the -// feedback-loop counter: every increment means an upstream pass produced a -// structurally impossible edge and a gate (not luck) stopped it. Exposed via -// StructuralEdgeDropCount for the audit battery and tests. -var structuralWriteDrops atomic.Int64 - -// StructuralEdgeDropCount reports how many structurally invalid edges write -// funnels have refused since process start. -func StructuralEdgeDropCount() int64 { - return structuralWriteDrops.Load() -} - -// FilterStructuralEdgeViolations drops edges StructuralEdgeTargetInvalid -// rejects, copying the slice only when a violation exists — the clean path -// (every real batch) allocates nothing. Returns the kept slice and the -// number dropped, so write funnels can surface a one-line count instead of -// silently eating mapper bugs. -func FilterStructuralEdgeViolations(edges []*Edge) ([]*Edge, int) { - dropped := 0 - kept := edges - for i, e := range edges { - if e != nil && StructuralEdgeTargetInvalid(e.Kind, e.To) { - if dropped == 0 { +// FilterStructuralEdgeViolations is a pure partition. It copies only after a +// violation appears, returning both kept and rejected edges so the first write +// boundary can attribute every rejected attempt exactly once. +func FilterStructuralEdgeViolations(edges []*Edge) (kept, rejected []*Edge) { + kept = edges + for i, edge := range edges { + if edge != nil && StructuralEdgeTargetInvalid(edge.Kind, edge.To) { + if len(rejected) == 0 { kept = append(make([]*Edge, 0, len(edges)), edges[:i]...) } - dropped++ + rejected = append(rejected, edge) continue } - if dropped > 0 { - kept = append(kept, e) + if len(rejected) > 0 { + kept = append(kept, edge) } } - if dropped > 0 { - structuralWriteDrops.Add(int64(dropped)) - } - return kept, dropped + return kept, rejected } func IsUnresolvedTarget(id string) bool { diff --git a/internal/hooks/pretooluse.go b/internal/hooks/pretooluse.go index 9c04dd982..249c142db 100644 --- a/internal/hooks/pretooluse.go +++ b/internal/hooks/pretooluse.go @@ -5,7 +5,6 @@ import ( "encoding/json" "errors" "fmt" - "io" "os" "path/filepath" "strings" @@ -65,18 +64,6 @@ type enrichResult struct { reason string } -// RunPreToolUse reads a PreToolUse hook payload from stdin and handles it -// in the legacy deny posture. Kept as a public entry point for -// backward compatibility; new callers should use Run which dispatches -// based on hook_event_name and respects the configured Mode. -func RunPreToolUse(gortexPort int) { - data, err := io.ReadAll(os.Stdin) - if err != nil { - return - } - runPreToolUse(data, gortexPort, ModeDeny) -} - // gortexMCPToolPrefix is the namespace Claude Code gives Gortex's own // MCP tools (server name "gortex"). A tool call whose name starts with // this prefix is a graph query — the in-process hook sees it like any @@ -84,8 +71,8 @@ func RunPreToolUse(gortexPort int) { // the adaptive-nudge streak reset work without an external signal. const gortexMCPToolPrefix = "mcp__gortex__" -// runPreToolUse is the bytes-accepting helper used by both RunPreToolUse and -// the generic Run dispatcher. In ModeEnrich the deny branch is downgraded +// runPreToolUse is the bytes-accepting helper the generic Run dispatcher +// and the Codex bridge share. In ModeEnrich the deny branch is downgraded // to an additionalContext message — the agent is informed about the graph // alternative but the original call still runs and PostToolUse can layer // graph context on the actual output. diff --git a/internal/indexer/affected_by.go b/internal/indexer/affected_by.go index bfbea2d2d..346c1fd3b 100644 --- a/internal/indexer/affected_by.go +++ b/internal/indexer/affected_by.go @@ -546,21 +546,9 @@ func (idx *Indexer) reresolveAffectedBy(changedPath string, snap *affectedBySnap zap.Int("affected", len(files)), zap.Int("cap", maxFiles), zap.Int("dropped", len(files)-maxFiles)) - idx.affectedByDropped.Add(int64(len(files) - maxFiles)) files = files[:maxFiles] } - idx.affectedByPasses.Add(1) - idx.affectedByFilesResolved.Add(int64(len(files))) - idx.resolver.ResolveFilesAndIncoming(files) resolver.SynthesizeExternalCallsForFiles(idx.graph, idx.externalCallSynthesisEnabled(), files) idx.persistRefFactsForFiles(files) } - -// AffectedByCounts reports the affected-by pass activity for this -// indexer: passes run, referencing files re-resolved, and files dropped -// by the fan-out cap. Diagnostic/test hook — the body-only-edit gate is -// observable as an unchanged pass count. -func (idx *Indexer) AffectedByCounts() (passes, files, dropped int64) { - return idx.affectedByPasses.Load(), idx.affectedByFilesResolved.Load(), idx.affectedByDropped.Load() -} diff --git a/internal/indexer/affected_by_crosslang_test.go b/internal/indexer/affected_by_crosslang_test.go index a1d69e606..46366c0b7 100644 --- a/internal/indexer/affected_by_crosslang_test.go +++ b/internal/indexer/affected_by_crosslang_test.go @@ -41,10 +41,6 @@ func TestAffectedBy_TypeScriptParamChange_ReresolvesCaller(t *testing.T) { _, err = idx.IncrementalReindexPaths(dir, []string{aPath}) require.NoError(t, err) - passes, files, _ := idx.AffectedByCounts() - assert.Equal(t, int64(1), passes, - "a TypeScript parameter change must trigger the pass even though Meta[signature] is name-only") - assert.Equal(t, int64(1), files) assert.Equal(t, "a.ts::F", callTargetFrom(t, g, callerID), "the caller must be re-resolved to the fresh F") } @@ -64,14 +60,22 @@ func TestAffectedBy_TypeScriptBodyOnly_NoFanout(t *testing.T) { _, err := idx.Index(dir) require.NoError(t, err) idx.ResolveAll() + g := idx.Graph() + snap := idx.snapshotAffectedBy("a.ts") + require.NotNil(t, snap) bumpMtime(t, aPath, "export function F(x: number): number {\n return x + 1 + 2 + 3\n}\n") _, err = idx.IncrementalReindexPaths(dir, []string{aPath}) require.NoError(t, err) - passes, _, _ := idx.AffectedByCounts() - assert.Equal(t, int64(0), passes, - "a TypeScript body-only edit must not fan out") + var fresh []*graph.Node + for _, node := range g.AllNodes() { + if node != nil && node.FilePath == "a.ts" { + fresh = append(fresh, node) + } + } + assert.Empty(t, affectedByDelta(g, snap, fresh), + "a TypeScript body-only edit must not produce an affected-by frontier") } // TestAffectedByDelta_LineSuffixedID_NoSpuriousDelta proves the line- @@ -156,11 +160,6 @@ func TestAffectedBy_MinifiedSkip_PreservesFactsNoFanout(t *testing.T) { _, err = idx.IncrementalReindexPaths(dir, []string{aPath}) require.NoError(t, err) - passes, files, _ := idx.AffectedByCounts() - assert.Equal(t, int64(0), passes, - "a minified-skip (zero symbols) must not fan out as if every symbol was removed") - assert.Equal(t, int64(0), files) - after, err := store.LoadRefFactsByTargets("", []string{fID}) require.NoError(t, err) assert.Contains(t, after, "b.js", diff --git a/internal/indexer/affected_by_e2e_test.go b/internal/indexer/affected_by_e2e_test.go index 17f18df44..f92810cbe 100644 --- a/internal/indexer/affected_by_e2e_test.go +++ b/internal/indexer/affected_by_e2e_test.go @@ -7,6 +7,7 @@ import ( "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" "go.uber.org/zap" + "go.uber.org/zap/zaptest/observer" "github.com/zzet/gortex/internal/config" "github.com/zzet/gortex/internal/graph" @@ -33,8 +34,7 @@ func newSQLiteIndexer(t *testing.T) (*Indexer, *store_sqlite.Store) { // TestAffectedBy_SignatureChange_ReresolvesCaller is the headline case // on the in-memory backend (whose reverse lookup is the pre-evict // in-edge snapshot): b.go calls F defined in a.go; changing F's -// SIGNATURE re-resolves b.go — its call edge lands on the fresh F node -// and exactly one bounded affected-by pass ran over exactly one file. +// SIGNATURE re-resolves b.go — its call edge lands on the fresh F node. func TestAffectedBy_SignatureChange_ReresolvesCaller(t *testing.T) { dir := t.TempDir() aPath := filepath.Join(dir, "a.go") @@ -59,11 +59,6 @@ func TestAffectedBy_SignatureChange_ReresolvesCaller(t *testing.T) { newFID := fnNodeID(t, g, "a.go", "F") assert.Equal(t, newFID, callTargetFrom(t, g, callerID), "after F's signature changed, Caller's edge must be re-resolved to the fresh F") - - passes, files, dropped := idx.AffectedByCounts() - assert.Equal(t, int64(1), passes, "a signature change must trigger exactly one affected-by pass") - assert.Equal(t, int64(1), files, "the pass must re-resolve exactly the one referencing file") - assert.Equal(t, int64(0), dropped) } // TestAffectedBy_BodyOnlyEdit_NoFanout proves the gate: an edit that @@ -83,14 +78,21 @@ func TestAffectedBy_BodyOnlyEdit_NoFanout(t *testing.T) { require.NoError(t, err) callerID := fnNodeID(t, g, "b.go", "Caller") + snap := idx.snapshotAffectedBy("a.go") + require.NotNil(t, snap) bumpMtime(t, aPath, "package p\n\nfunc F(x int) int { return x + 1 }\n") res, err := idx.IncrementalReindexPaths(dir, nil) require.NoError(t, err) require.Equal(t, 1, res.StaleFileCount) - passes, files, _ := idx.AffectedByCounts() - assert.Equal(t, int64(0), passes, "a body-only edit must not trigger an affected-by pass") - assert.Equal(t, int64(0), files) + var fresh []*graph.Node + for _, node := range g.AllNodes() { + if node != nil && node.FilePath == "a.go" { + fresh = append(fresh, node) + } + } + assert.Empty(t, affectedByDelta(g, snap, fresh), + "a body-only edit must not produce an affected-by frontier") // The caller's edge still survives the definition re-index via the // existing restub + reverse-resolve pair. @@ -116,10 +118,8 @@ func TestAffectedBy_PerSaveIndexFile_ReresolvesCaller(t *testing.T) { writeFile(t, aPath, "package p\n\nfunc F(n int) int { return n }\n") require.NoError(t, idx.IndexFile(aPath)) - assert.Equal(t, fnNodeID(t, g, "a.go", "F"), callTargetFrom(t, g, callerID)) - passes, files, _ := idx.AffectedByCounts() - assert.Equal(t, int64(1), passes, "the per-save IndexFile route must run the pass") - assert.Equal(t, int64(1), files) + assert.Equal(t, fnNodeID(t, g, "a.go", "F"), callTargetFrom(t, g, callerID), + "the per-save IndexFile route must re-resolve the caller") } // TestAffectedBy_RemovedSymbol_SQLite removes a called symbol from its @@ -164,8 +164,6 @@ func TestAffectedBy_RemovedSymbol_SQLite(t *testing.T) { assert.NotEqual(t, fID, f.ToID, "the stale fact pointing at the removed symbol must be re-persisted away") } - passes, _, _ := idx.AffectedByCounts() - assert.Equal(t, int64(1), passes) } // TestAffectedBy_SidecarDiscovery_SQLite proves the persisted reverse @@ -199,17 +197,13 @@ func TestAffectedBy_SidecarDiscovery_SQLite(t *testing.T) { _, err = idx.IncrementalReindexPaths(dir, []string{aPath}) require.NoError(t, err) - passes, files, _ := idx.AffectedByCounts() - assert.Equal(t, int64(1), passes, - "the pass must run even with no live in-edges — discovery comes from the sidecar") - assert.Equal(t, int64(1), files) assert.Equal(t, fnNodeID(t, g, "a.go", "F"), callTargetFrom(t, g, callerID), "the sidecar-discovered caller must be re-resolved to the fresh F") } // TestAffectedBy_CapBoundsFanout configures a fan-out cap of 1 with -// three referencing files: the pass must re-resolve exactly one file -// and account for the two it dropped — the cap is loud, not silent. +// three referencing files. The live truncation warning must report the +// exact bounded set and the two files it dropped. func TestAffectedBy_CapBoundsFanout(t *testing.T) { dir := t.TempDir() aPath := filepath.Join(dir, "a.go") @@ -224,7 +218,8 @@ func TestAffectedBy_CapBoundsFanout(t *testing.T) { cfg := config.Default().Index cfg.Workers = 1 cfg.AffectedByReresolveMax = 1 - idx := New(g, reg, cfg, zap.NewNop()) + core, observed := observer.New(zap.DebugLevel) + idx := New(g, reg, cfg, zap.New(core)) _, err := idx.Index(dir) require.NoError(t, err) @@ -232,10 +227,12 @@ func TestAffectedBy_CapBoundsFanout(t *testing.T) { _, err = idx.IncrementalReindexPaths(dir, []string{aPath}) require.NoError(t, err) - passes, files, dropped := idx.AffectedByCounts() - assert.Equal(t, int64(1), passes) - assert.Equal(t, int64(1), files, "the cap must bound the re-resolve set") - assert.Equal(t, int64(2), dropped, "the truncated files must be accounted, not silently lost") + entries := observed.FilterMessage("affected-by: re-resolve set truncated").All() + require.Len(t, entries, 1) + fields := entries[0].ContextMap() + assert.EqualValues(t, 3, fields["affected"]) + assert.EqualValues(t, 1, fields["cap"]) + assert.EqualValues(t, 2, fields["dropped"]) } // TestAffectedBy_DeferredBatchPath_NoFanout proves the batch guard: a @@ -246,10 +243,17 @@ func TestAffectedBy_DeferredBatchPath_NoFanout(t *testing.T) { dir := t.TempDir() aPath := filepath.Join(dir, "a.go") writeFile(t, aPath, "package p\n\nfunc F(x int) int { return x }\n") - writeFile(t, filepath.Join(dir, "b.go"), "package p\n\nfunc Caller() int { return F(1) }\n") + writeFile(t, filepath.Join(dir, "b.go"), "package p\n\nfunc CallerB() int { return F(1) }\n") + writeFile(t, filepath.Join(dir, "c.go"), "package p\n\nfunc CallerC() int { return F(2) }\n") g := graph.New() - idx := newTestIndexer(g) + reg := parser.NewRegistry() + reg.Register(languages.NewGoExtractor()) + cfg := config.Default().Index + cfg.Workers = 1 + cfg.AffectedByReresolveMax = 1 + core, observed := observer.New(zap.DebugLevel) + idx := New(g, reg, cfg, zap.New(core)) _, err := idx.Index(dir) require.NoError(t, err) @@ -257,7 +261,6 @@ func TestAffectedBy_DeferredBatchPath_NoFanout(t *testing.T) { bumpMtime(t, aPath, "package p\n\nfunc F(x int, y int) int { return x + y }\n") require.NoError(t, idx.IndexFile(aPath)) - passes, _, _ := idx.AffectedByCounts() - assert.Equal(t, int64(0), passes, + assert.Empty(t, observed.FilterMessage("affected-by: re-resolve set truncated").All(), "deferred-batch indexing must not fan out per file — the batch caller resolves once at the end") } diff --git a/internal/indexer/batch_census_test.go b/internal/indexer/batch_census_test.go index d276466ff..f2033c46c 100644 --- a/internal/indexer/batch_census_test.go +++ b/internal/indexer/batch_census_test.go @@ -1,7 +1,6 @@ package indexer import ( - "context" "testing" "github.com/stretchr/testify/assert" @@ -24,14 +23,6 @@ func TestBatchScopeAndCensusBelongOnlyToEndBatch(t *testing.T) { mi.ArmBatchScope(map[string]struct{}{"repo-b": {}}) mi.ArmBatchCensusEligible() - // An unrelated public run is explicitly unscoped and cannot steal the - // one-shot state that belongs to EndBatch. - mi.RunGlobalGraphPasses(context.Background()) - mi.mu.RLock() - assert.Equal(t, map[string]struct{}{"repo-a": {}, "repo-b": {}}, mi.batchChangedPrefixes) - assert.True(t, mi.batchCensusEligible) - mi.mu.RUnlock() - mi.EndBatch() mi.mu.RLock() assert.Nil(t, mi.batchChangedPrefixes) diff --git a/internal/indexer/batch_transition_coordinator_test.go b/internal/indexer/batch_transition_coordinator_test.go index 09bb4380d..8d9f48eca 100644 --- a/internal/indexer/batch_transition_coordinator_test.go +++ b/internal/indexer/batch_transition_coordinator_test.go @@ -322,7 +322,7 @@ func (s *blockingBatchPurgeStore) PurgeRepo(string) error { return nil } -func TestUntrackTeardownBlocksDirectGlobalPass(t *testing.T) { +func TestUntrackTeardownBlocksReachLookup(t *testing.T) { releasePurge := make(chan struct{}) store := &blockingBatchPurgeStore{ Store: graph.New(), @@ -354,20 +354,6 @@ func TestUntrackTeardownBlocksDirectGlobalPass(t *testing.T) { t.Fatal("UntrackRepo did not reach the repository purge") } - globalSubmitted := make(chan struct{}) - globalDone := make(chan struct{}) - go func() { - close(globalSubmitted) - mi.RunGlobalGraphPasses(context.Background()) - close(globalDone) - }() - <-globalSubmitted - select { - case <-globalDone: - t.Fatal("direct global pass crossed an in-flight untrack teardown") - case <-time.After(75 * time.Millisecond): - } - lookupStarted := make(chan struct{}) lookupDone := make(chan struct{}) go func() { @@ -390,11 +376,6 @@ func TestUntrackTeardownBlocksDirectGlobalPass(t *testing.T) { t.Fatal("UntrackRepo did not finish after purge release") } select { - case <-globalDone: - case <-time.After(batchTransitionTestTimeout): - t.Fatal("direct global pass did not resume after untrack teardown") - } - select { case <-lookupDone: case <-time.After(batchTransitionTestTimeout): t.Fatal("reach lookup did not resume after untrack publication") diff --git a/internal/indexer/chunk_index_test.go b/internal/indexer/chunk_index_test.go index 7d222779b..246ca092c 100644 --- a/internal/indexer/chunk_index_test.go +++ b/internal/indexer/chunk_index_test.go @@ -1,6 +1,7 @@ package indexer import ( + "context" "os" "path/filepath" "strings" @@ -18,6 +19,29 @@ import ( "github.com/zzet/gortex/internal/search" ) +// stubEmbedder is a deterministic minimal embedder that lets the +// indexer wire up a HybridBackend in tests without pulling in the +// static-vector asset or a real ONNX runtime. It emits a 4-dim +// vector whose first element is the text length — enough for the +// vector backend to accept the adds and for Search to return +// something non-empty. +type stubEmbedder struct{} + +func (stubEmbedder) Embed(_ context.Context, text string) ([]float32, error) { + return []float32{float32(len(text)), 0, 0, 0}, nil +} + +func (stubEmbedder) EmbedBatch(_ context.Context, texts []string) ([][]float32, error) { + out := make([][]float32, len(texts)) + for i, t := range texts { + out[i] = []float32{float32(len(t)), 0, 0, 0} + } + return out, nil +} + +func (stubEmbedder) Dimensions() int { return 4 } +func (stubEmbedder) Close() error { return nil } + // indexedVectorBackend indexes dir with the given chunk options and // returns the vector backend buildSearchIndex produced. func indexedVectorBackend(t *testing.T, dir string, opts embedding.ChunkOptions) *search.VectorBackend { diff --git a/internal/indexer/clone_batch_hotpath_test.go b/internal/indexer/clone_batch_hotpath_test.go index 46d5a69e7..760d5c38a 100644 --- a/internal/indexer/clone_batch_hotpath_test.go +++ b/internal/indexer/clone_batch_hotpath_test.go @@ -1,6 +1,7 @@ package indexer import ( + "context" "fmt" "testing" @@ -73,7 +74,7 @@ func TestDetectClonesBatchesEndpointReadsAndEdgeWrites(t *testing.T) { base.AddBatch(nodes, nil) store := &cloneIOCountingStore{Store: base} - stats := detectClonesAndEmitEdges(store, "", 0) + stats, _ := detectClonesAndEmitEdgesWithBaselineCtx(context.Background(), store, "", 0) if stats.Items != 3 || stats.Pairs != 3 || stats.Edges != 6 { t.Fatalf("clone stats = items:%d pairs:%d edges:%d, want 3/3/6", stats.Items, stats.Pairs, stats.Edges) } diff --git a/internal/indexer/clone_incremental.go b/internal/indexer/clone_incremental.go index eb0dca082..c5f711c7c 100644 --- a/internal/indexer/clone_incremental.go +++ b/internal/indexer/clone_incremental.go @@ -296,12 +296,6 @@ func (ci *incrementalCloneIndex) Pending() bool { return ci.pending } -// CloneIndexPending reports whether clone-derived edges are awaiting an -// explicit global/clone-consuming rebuild. -func (idx *Indexer) CloneIndexPending() bool { - return idx != nil && idx.cloneIndex != nil && idx.cloneIndex.Pending() -} - // EvictFuncs removes a set of function/method nodes from the index: it // decrements their shingles out of the CMS, drops them from the LSH index // and the in-memory cache, and deletes their rows from the persisted diff --git a/internal/indexer/clone_sqlite_persistence_test.go b/internal/indexer/clone_sqlite_persistence_test.go index f2614d790..e14117ab4 100644 --- a/internal/indexer/clone_sqlite_persistence_test.go +++ b/internal/indexer/clone_sqlite_persistence_test.go @@ -121,7 +121,10 @@ func TestSQLiteCloneCorpusSurvivesReloadAndWarmReplay(t *testing.T) { counted.writeRows = 0 beforeEdges := reopened.EdgeCount() replayed := newTestIndexer(counted) - replayed.RunGlobalGraphPasses(context.Background()) + _, cloneBaseline := detectClonesAndEmitEdgesWithBaselineCtx( + context.Background(), counted, "", replayed.cloneThreshold(), + ) + replayed.cloneIndex.AdoptBaselineOrRebuild(counted, "", cloneBaseline) require.Equal(t, 1, counted.pageCalls, "warm global pass must scan the finalized corpus exactly once") require.Zero(t, counted.writeCalls, "warm finalized corpus must not rewrite signatures") require.Zero(t, counted.writeRows) diff --git a/internal/indexer/clones.go b/internal/indexer/clones.go index 913935806..f7bcf6272 100644 --- a/internal/indexer/clones.go +++ b/internal/indexer/clones.go @@ -659,8 +659,8 @@ func finaliseCloneSignaturesFromNodes(g graph.Store, repoPrefix string) ([]clone return items, len(bodies) } -// CloneDetectionStats summarises one detectClonesAndEmitEdges run for -// the caller's logger. Exposed so the orchestrator can surface what the +// CloneDetectionStats summarises one clone-detection run for the caller's +// logger. Exposed so the orchestrator can surface what the // per-bucket cap dropped — a high skippedBucketItems means the // workspace has a lot of templated boilerplate that LSH would have // over-fanned-out on. @@ -674,44 +674,6 @@ type CloneDetectionStats struct { DiffusedEdges int // EdgeSemanticallyRelated emitted (= 2·DiffusedPairs) } -// detectClonesAndEmitEdges is the graph-wide half of clone detection. -// It collects every function/method node carrying a clone_sig, runs -// the MinHash + LSH pass over their signatures, and materialises a -// symmetric pair of EdgeSimilarTo edges for each detected clone pair. -// -// threshold is the Jaccard similarity cutoff; pass 0 to use the -// clones package default. Returns clone stats including the per-bucket -// cap telemetry — the orchestrator logs that so a high skip count is -// visible during warmup. -// -// The pass is a full recompute and is idempotent: graph.AddEdge dedupes -// by edgeKey so re-emitting an unchanged pair is a no-op, and stale -// edges cannot survive — when either endpoint's file is reindexed, -// EvictFile removes that node's edges in both directions before this -// pass re-runs. -// -// repoPrefix scopes the pass to one repository's nodes: every whole-graph -// walk it drives (finalise, item gather, diffusion) is filtered to -// n.RepoPrefix == repoPrefix so no cross-repo candidate pair is ever -// formed. A standalone single-repo Indexer passes "" and its nodes carry -// RepoPrefix == "", so the equality matches all nodes and the single-repo -// result is unchanged. -func detectClonesAndEmitEdges(g graph.Store, repoPrefix string, threshold float64) CloneDetectionStats { - return detectClonesAndEmitEdgesCtx(context.Background(), g, repoPrefix, threshold) -} - -// detectClonesAndEmitEdgesCtx is the context-aware sibling of -// detectClonesAndEmitEdges. It emits sub-stage progress markers via -// the reporter attached to ctx (see progress.WithReporter): clone -// detection is the longest single stage on monorepo-scale graphs and -// without intra-stage reporters an operator sees just one -// "clone detection pass" marker followed by minutes of silence — no -// way to tell finalise-signatures from LSH from edge-emission. -func detectClonesAndEmitEdgesCtx(ctx context.Context, g graph.Store, repoPrefix string, threshold float64) CloneDetectionStats { - stats, _ := detectClonesAndEmitEdgesWithBaselineCtx(ctx, g, repoPrefix, threshold) - return stats -} - // detectClonesAndEmitEdgesWithBaselineCtx performs the context-aware clone // pass and also returns the finalized compact-corpus baseline. The baseline is // complete only when the paged projection was read successfully; callers hand diff --git a/internal/indexer/clones_indexer_test.go b/internal/indexer/clones_indexer_test.go index ff9831969..dc11a02e9 100644 --- a/internal/indexer/clones_indexer_test.go +++ b/internal/indexer/clones_indexer_test.go @@ -1,6 +1,7 @@ package indexer import ( + "context" "path/filepath" "testing" @@ -169,12 +170,12 @@ func TestDetectClonesAndEmitEdges(t *testing.T) { FilePath: "c.go", StartLine: 1, Language: "go", }) - stats := detectClonesAndEmitEdges(g, "", 0) + stats, _ := detectClonesAndEmitEdgesWithBaselineCtx(context.Background(), g, "", 0) assert.Equal(t, 1, stats.Pairs) assert.Equal(t, 2, stats.Edges) // Idempotent: a second run dedupes via graph.AddEdge. - detectClonesAndEmitEdges(g, "", 0) + detectClonesAndEmitEdgesWithBaselineCtx(context.Background(), g, "", 0) assert.Len(t, similarToEdges(g), 2, "second pass must not duplicate edges") } diff --git a/internal/indexer/clones_multirepo_test.go b/internal/indexer/clones_multirepo_test.go index e35cc0ac7..7fa94e57c 100644 --- a/internal/indexer/clones_multirepo_test.go +++ b/internal/indexer/clones_multirepo_test.go @@ -260,8 +260,8 @@ func TestClones_PerRepo_NoCrossRepoEdges(t *testing.T) { require.NoError(t, err) // Per-repo batch clone pass (the new MultiIndexer loop). - csA := detectClonesAndEmitEdgesCtx(ctx, gBatch, "repoA", 0) - csB := detectClonesAndEmitEdgesCtx(ctx, gBatch, "repoB", 0) + csA, _ := detectClonesAndEmitEdgesWithBaselineCtx(ctx, gBatch, "repoA", 0) + csB, _ := detectClonesAndEmitEdgesWithBaselineCtx(ctx, gBatch, "repoB", 0) require.Positive(t, csA.Items, "repoA must have clone-eligible bodies") require.Positive(t, csB.Items, "repoB must have clone-eligible bodies") diff --git a/internal/indexer/content_linking_test.go b/internal/indexer/content_linking_test.go index ac81a874a..b1b948e58 100644 --- a/internal/indexer/content_linking_test.go +++ b/internal/indexer/content_linking_test.go @@ -1,7 +1,6 @@ package indexer import ( - "context" "sort" "testing" @@ -11,6 +10,7 @@ import ( "github.com/zzet/gortex/internal/config" "github.com/zzet/gortex/internal/graph" "github.com/zzet/gortex/internal/parser" + "github.com/zzet/gortex/internal/resolver" ) func TestContentLinkEdgeBudget(t *testing.T) { @@ -73,7 +73,8 @@ func TestContentLinkingAndGlobalPassAvoidAllNodesWithCrossRepoParity(t *testing. require.Equal(t, want, contentLinkTargets(counting, graph.EdgeMotivates)) global := &allNodesCountingGraph{Graph: contentLinkFixture()} - newIndexer(global).RunGlobalGraphPasses(context.Background()) + newIndexer(global).linkContentToCode() + resolver.DetectCrossRepoEdges(global) require.Zero(t, global.allNodesCalls, "the automatic global cold path must never request a node snapshot") require.Equal(t, want, contentLinkTargets(global, graph.EdgeMotivates)) require.Equal(t, []string{"repoB/pkg/b.go::SharedSymbol"}, diff --git a/internal/indexer/contract_import_resolve.go b/internal/indexer/contract_import_resolve.go index 095102d88..7d00c7fd2 100644 --- a/internal/indexer/contract_import_resolve.go +++ b/internal/indexer/contract_import_resolve.go @@ -1167,13 +1167,9 @@ func (mi *MultiIndexer) reExportsFor(src, srcPath string) []reExportEdge { return out } -// followReExportChain returns the set of concrete file paths reachable -// from startFile by following re-exports of `name` — startFile itself -// plus every module a transparent `export *` / `export { name }` -// (TypeScript) or `pub use` (Rust) chain forwards through, up to -// maxReExportDepth. A symbol's real definition is in one of the -// returned files, so a caller matching an import target against graph -// nodes resolves through the barrel / re-exporting module. +// reExportChainResult records concrete files and symbol names reached while +// following TypeScript or Rust re-exports, plus whether ambiguity made the +// chain unsafe to use for resolution. type reExportChainResult struct { files map[string]bool names map[string]map[string]bool @@ -1190,10 +1186,6 @@ func addReExportChainTarget(result *reExportChainResult, file, name string) { } } -func (mi *MultiIndexer) followReExportChain(startFile, name string, srcCache map[string][]byte) map[string]bool { - return mi.followReExportChainDetailed(startFile, name, srcCache).files -} - func (mi *MultiIndexer) followReExportChainChecked(startFile, name string, srcCache map[string][]byte) (map[string]bool, bool) { result := mi.followReExportChainDetailed(startFile, name, srcCache) return result.files, result.unsafe diff --git a/internal/indexer/contract_import_resolve_test.go b/internal/indexer/contract_import_resolve_test.go index 51e3b6c8d..d19b0d437 100644 --- a/internal/indexer/contract_import_resolve_test.go +++ b/internal/indexer/contract_import_resolve_test.go @@ -68,7 +68,7 @@ func TestFollowReExportChain_DefaultAsSvelte(t *testing.T) { "src/lib/index.ts": []byte(`export { default as Button } from './Button.svelte';`), "src/lib/Button.svelte": []byte(``), } - reachable := mi.followReExportChain("src/lib/index.ts", "Button", srcCache) + reachable := mi.followReExportChainDetailed("src/lib/index.ts", "Button", srcCache).files require.True(t, reachable["src/lib/Button.svelte"], "the `default as` re-export must reach the .svelte component file") } @@ -103,7 +103,7 @@ func TestFollowReExportChain_BarrelToTerminal(t *testing.T) { "src/components/index.ts": []byte(`export * from './Widget';`), "src/components/Widget.ts": []byte(`export interface Widget { id: string }`), } - reachable := mi.followReExportChain("src/index.ts", "Widget", srcCache) + reachable := mi.followReExportChainDetailed("src/index.ts", "Widget", srcCache).files require.True(t, reachable["src/components/Widget.ts"], "re-export chain must reach the terminal module") } @@ -114,7 +114,7 @@ func TestFollowReExportChain_RenamedExport(t *testing.T) { "pkg/index.ts": []byte(`export { Internal as Public } from './impl';`), "pkg/impl.ts": []byte(`export class Internal {}`), } - reachable := mi.followReExportChain("pkg/index.ts", "Public", srcCache) + reachable := mi.followReExportChainDetailed("pkg/index.ts", "Public", srcCache).files require.True(t, reachable["pkg/impl.ts"], "must follow the `as` rename to the source module") } @@ -124,7 +124,7 @@ func TestFollowReExportChain_CircularTerminates(t *testing.T) { "a.ts": []byte(`export * from './b';`), "b.ts": []byte(`export * from './a';`), } - reachable := mi.followReExportChain("a.ts", "X", srcCache) + reachable := mi.followReExportChainDetailed("a.ts", "X", srcCache).files require.True(t, reachable["a.ts"]) require.True(t, reachable["b.ts"]) } @@ -304,7 +304,7 @@ func TestFollowReExportChain_WildcardMultiHop(t *testing.T) { "pkg/terminal.ts": []byte(`export class Engine {}`), "pkg/unrelated.ts": []byte(`export class Engine {}`), } - reachable := mi.followReExportChain("pkg/index.ts", "Engine", srcCache) + reachable := mi.followReExportChainDetailed("pkg/index.ts", "Engine", srcCache).files require.True(t, reachable["pkg/terminal.ts"], "multi-hop `export *` chain must reach the terminal module") require.False(t, reachable["pkg/unrelated.ts"], @@ -321,7 +321,7 @@ func TestFollowReExportChain_WildcardThroughNamed(t *testing.T) { "src/theme.ts": []byte(`export * from './tokens';`), "src/tokens.ts": []byte(`export interface Theme { name: string }`), } - reachable := mi.followReExportChain("src/index.ts", "Theme", srcCache) + reachable := mi.followReExportChainDetailed("src/index.ts", "Theme", srcCache).files require.True(t, reachable["src/tokens.ts"], "named → wildcard re-export chain must reach the definition module") } @@ -365,7 +365,7 @@ func TestFollowReExportChain_RustPubUseMultiHop(t *testing.T) { "crate/src/domain.rs": []byte(`pub struct User { id: u64 }`), "crate/src/unrelated.rs": []byte(`pub struct User;`), } - reachable := mi.followReExportChain("crate/src/lib.rs", "User", srcCache) + reachable := mi.followReExportChainDetailed("crate/src/lib.rs", "User", srcCache).files require.True(t, reachable["crate/src/domain.rs"], "multi-hop `pub use` chain must reach the defining module") require.False(t, reachable["crate/src/unrelated.rs"], @@ -381,7 +381,7 @@ func TestFollowReExportChain_RustGlobAndModDir(t *testing.T) { "src/shapes/mod.rs": []byte(`pub use self::circle::Circle;`), "src/shapes/circle.rs": []byte(`pub struct Circle { r: f64 }`), } - reachable := mi.followReExportChain("src/lib.rs", "Circle", srcCache) + reachable := mi.followReExportChainDetailed("src/lib.rs", "Circle", srcCache).files require.True(t, reachable["src/shapes/circle.rs"], "glob re-export into a mod.rs directory module must reach the leaf") } @@ -398,7 +398,7 @@ func TestFollowReExportChain_RustPrivateUseNotForwarded(t *testing.T) { "src/api.rs": []byte(`use crate::secret::Token;`), "src/secret.rs": []byte(`pub struct Token;`), } - reachable := mi.followReExportChain("src/lib.rs", "Token", srcCache) + reachable := mi.followReExportChainDetailed("src/lib.rs", "Token", srcCache).files require.False(t, reachable["src/secret.rs"], "a privately-`use`d symbol must not resolve through the re-export chain") } diff --git a/internal/indexer/crash_isolation.go b/internal/indexer/crash_isolation.go index 5806c988f..cf80625cf 100644 --- a/internal/indexer/crash_isolation.go +++ b/internal/indexer/crash_isolation.go @@ -120,6 +120,12 @@ func fallbackChunkerEnv(specs []config.FallbackChunkerSpec) string { // on. Returns (nil, nil) if no worker can spawn, so the caller falls // back to in-process extraction. func (idx *Indexer) sharedParsePool() (*crashpool.Pool, *crashpool.Quarantine) { + releaseLifecycle, err := idx.extractionLifecycle.admit() + if err != nil { + return nil, nil + } + defer releaseLifecycle() + idx.parsePoolMu.Lock() defer idx.parsePoolMu.Unlock() if idx.parsePool != nil { @@ -146,9 +152,9 @@ func (idx *Indexer) sharedParsePool() (*crashpool.Pool, *crashpool.Quarantine) { return idx.parsePool, idx.parseQuar } -// CloseParsePool tears down the long-lived crash-isolation pool if one +// closeParsePool tears down the long-lived crash-isolation pool if one // was created. It is safe to call when none exists, and idempotent. -func (idx *Indexer) CloseParsePool() { +func (idx *Indexer) closeParsePool() { idx.parsePoolMu.Lock() defer idx.parsePoolMu.Unlock() if idx.parsePool != nil { @@ -231,18 +237,6 @@ func (idx *Indexer) extractFileWithRawLease( ) } -func (idx *Indexer) extractFileCtx( - ctx context.Context, - nativeAdmission *nativeParseExtractionAdmission, - pool *crashpool.Pool, q *crashpool.Quarantine, - path, relPath, lang string, ext parser.Extractor, src []byte, -) (result *parser.ExtractionResult, skipped bool, err error) { - return idx.extractFileCtxWithRawLease( - ctx, nativeAdmission, nil, - pool, q, path, relPath, lang, ext, src, - ) -} - func (idx *Indexer) extractFileCtxWithRawLease( ctx context.Context, nativeAdmission *nativeParseExtractionAdmission, @@ -320,7 +314,15 @@ func (idx *Indexer) extractFileCtxWithRawLease( true, fmt.Errorf("skipped quarantined file %s", relPath) } - res := submitWithRetry(func() crashpool.Result { return pool.Submit(relPath, lang, src) }) + releaseLifecycle, lifecycleErr := idx.extractionLifecycle.admit() + if lifecycleErr != nil { + return nil, false, lifecycleErr + } + defer releaseLifecycle() + helperNames := idx.extractionOptionsValue().TemporalEnvHelpers() + res := submitWithRetry(func() crashpool.Result { + return pool.SubmitWithOptions(relPath, lang, src, helperNames) + }) switch { case res.Crashed || res.Panicked: q.Add(relPath, res.Err, mtime) diff --git a/internal/indexer/crash_isolation_test.go b/internal/indexer/crash_isolation_test.go index 8d2b0839d..124120d60 100644 --- a/internal/indexer/crash_isolation_test.go +++ b/internal/indexer/crash_isolation_test.go @@ -174,7 +174,7 @@ func TestSharedParsePool_Reused(t *testing.T) { g := graph.New() idx := newTestIndexer(g) idx.SetRootPath(t.TempDir()) - defer idx.CloseParsePool() + defer idx.Close() p1, _ := idx.sharedParsePool() require.NotNil(t, p1) @@ -197,7 +197,7 @@ func TestIndexFile_CrashIsolationReusesPool(t *testing.T) { idx := newTestIndexer(g) _, err := idx.Index(dir) // cold index also anchors rootPath require.NoError(t, err) - defer idx.CloseParsePool() + defer idx.Close() require.NoError(t, idx.IndexFile(filepath.Join(dir, "a.go"))) require.NotNil(t, idx.parsePool, "first IndexFile must create the shared pool") diff --git a/internal/indexer/deferred_enrich_gate_test.go b/internal/indexer/deferred_enrich_gate_test.go index 6c2a7e570..d18f15893 100644 --- a/internal/indexer/deferred_enrich_gate_test.go +++ b/internal/indexer/deferred_enrich_gate_test.go @@ -116,7 +116,7 @@ func TestMultiIndexer_RunDeferredEnrich_GatesUnchangedRepos(t *testing.T) { // Drive the same per-repo dispatch the warmup's parallel enrich loop uses. mi := newEmptyMultiIndexer(t, g) - mi.runDeferredEnrichParallel([]*Indexer{changed, unchanged}) + mi.runDeferredEnrichPool([]*Indexer{changed, unchanged}) assert.Equal(t, []string{"repo-changed"}, spy.invoked(), "only the changed repo should have its enrichment dispatched") diff --git a/internal/indexer/deferred_enrich_ledger_test.go b/internal/indexer/deferred_enrich_ledger_test.go index 938037aba..79040a0ac 100644 --- a/internal/indexer/deferred_enrich_ledger_test.go +++ b/internal/indexer/deferred_enrich_ledger_test.go @@ -278,7 +278,7 @@ func TestSeedPendingEnrichAll_ResumesNonGitRepo(t *testing.T) { assert.True(t, plain.pendingEnrich.Load()) assert.False(t, complete.pendingEnrich.Load()) - mi.runDeferredEnrichParallel([]*Indexer{complete, plain}) + mi.runDeferredEnrichPool([]*Indexer{complete, plain}) assert.Equal(t, []string{filepath.Join("plain", "watched.go")}, spy.enrichedFiles(), "only the resumed repo should have its enrichment dispatched") assert.Empty(t, spy.invoked(), diff --git a/internal/indexer/deferred_enrich_resume_test.go b/internal/indexer/deferred_enrich_resume_test.go index e75a0a3bb..cadefd117 100644 --- a/internal/indexer/deferred_enrich_resume_test.go +++ b/internal/indexer/deferred_enrich_resume_test.go @@ -161,7 +161,7 @@ func TestSeedPendingEnrichAll_ResumesOnlyIncompleteRepos(t *testing.T) { assert.False(t, complete.pendingEnrich.Load()) // The parallel enrich driver then dispatches only the re-armed repo. - mi.runDeferredEnrichParallel([]*Indexer{complete, incomplete}) + mi.runDeferredEnrichPool([]*Indexer{complete, incomplete}) assert.Equal(t, []string{"incomplete"}, spy.invoked(), "only the resumed repo should have its enrichment dispatched") } diff --git a/internal/indexer/deferred_semantic_state_test.go b/internal/indexer/deferred_semantic_state_test.go index 47e34abbb..e8a2b0ed5 100644 --- a/internal/indexer/deferred_semantic_state_test.go +++ b/internal/indexer/deferred_semantic_state_test.go @@ -50,7 +50,7 @@ func TestDeferredEnrichmentReleasesEveryRepoStateAfterDrain(t *testing.T) { }) } - scheduled := mi.RunDeferredPassesAll(context.Background()) + scheduled := mi.RunDeferredPassesAllResult(context.Background()).EnrichScheduled if scheduled != repoCount { t.Fatalf("scheduled enrichment = %d, want %d", scheduled, repoCount) } diff --git a/internal/indexer/deferred_work_test.go b/internal/indexer/deferred_work_test.go index 828ae873a..8534bd6e8 100644 --- a/internal/indexer/deferred_work_test.go +++ b/internal/indexer/deferred_work_test.go @@ -38,7 +38,7 @@ func TestRunDeferredPassesAllSkipsIdleRepositoriesBeforeStatsScan(t *testing.T) }, } - if scheduled := mi.RunDeferredPassesAll(context.Background()); scheduled != 0 { + if scheduled := mi.RunDeferredPassesAllResult(context.Background()).EnrichScheduled; scheduled != 0 { t.Fatalf("scheduled enrichments = %d, want 0", scheduled) } if store.repoStatsCalls != 0 { diff --git a/internal/indexer/diffusion_test.go b/internal/indexer/diffusion_test.go index 6db429519..ec51613ef 100644 --- a/internal/indexer/diffusion_test.go +++ b/internal/indexer/diffusion_test.go @@ -1,6 +1,7 @@ package indexer import ( + "context" "testing" "github.com/stretchr/testify/assert" @@ -327,7 +328,7 @@ func spokeID(i int) string { // TestDetectClonesAndEmitEdges_DiffusionWiring is an integration test // over the full clone+diffusion pass. It hand-builds a graph where two // function bodies are exact clones of a shared body and a third is a -// partial variant, then asserts detectClonesAndEmitEdges materialises +// partial variant, then asserts the live clone pass materialises // both similar_to and semantically_related edges and reports the // diffusion counts on CloneDetectionStats. func TestDetectClonesAndEmitEdges_DiffusionWiring(t *testing.T) { @@ -357,7 +358,7 @@ func TestDetectClonesAndEmitEdges_DiffusionWiring(t *testing.T) { Meta: map[string]any{cloneSigMetaKey: encAB}, }) - stats := detectClonesAndEmitEdges(g, "", 0) + stats, _ := detectClonesAndEmitEdgesWithBaselineCtx(context.Background(), g, "", 0) // A, B, C all share a signature: three direct clone pairs, so the // only diffusable pairs are themselves direct clones — diffusion // correctly emits nothing (partition invariant). diff --git a/internal/indexer/embed_pool_test.go b/internal/indexer/embed_pool_test.go index c6d261a99..b2e2f5f06 100644 --- a/internal/indexer/embed_pool_test.go +++ b/internal/indexer/embed_pool_test.go @@ -132,7 +132,7 @@ func TestEmbedAllChunks_PoolRespectsConcurrencyCap(t *testing.T) { idx := newPoolTestIndexer(t, emb, cap) texts := makeTexts(60) // 60 texts, batch 5 → 12 batches > cap - vectors, err := idx.embedAllChunks(texts, 5, passthroughEmbedFn(idx)) + vectors, err := idx.embedAllChunks(context.Background(), texts, 5, passthroughEmbedFn(idx)) require.NoError(t, err) require.Len(t, vectors, len(texts), "every text must be embedded") @@ -155,7 +155,7 @@ func TestEmbedAllChunks_AbortsOnError(t *testing.T) { emb := &poolEmbedder{delay: 5 * time.Millisecond, failOnText: "t37"} idx := newPoolTestIndexer(t, emb, 4) - vectors, err := idx.embedAllChunks(makeTexts(80), 5, passthroughEmbedFn(idx)) + vectors, err := idx.embedAllChunks(context.Background(), makeTexts(80), 5, passthroughEmbedFn(idx)) require.Error(t, err, "one failing chunk must fail the whole embed") assert.Nil(t, vectors, "a failed embed must return no partial result") assert.Contains(t, err.Error(), "t37") @@ -197,7 +197,7 @@ func TestEmbedAllChunks_DeterministicRegardlessOfOrder(t *testing.T) { idx := newPoolTestIndexer(t, emb, 6) texts := makeTexts(50) - vectors, err := idx.embedAllChunks(texts, 3, passthroughEmbedFn(idx)) + vectors, err := idx.embedAllChunks(context.Background(), texts, 3, passthroughEmbedFn(idx)) require.NoError(t, err) require.Len(t, vectors, len(texts)) for i := range texts { @@ -217,13 +217,69 @@ func TestEmbedAllChunks_SerialForNonConcurrentEmbedder(t *testing.T) { idx.SetEmbedder(emb) idx.SetEmbeddingAPIConcurrency(8) - vectors, err := idx.embedAllChunks(makeTexts(30), 3, passthroughEmbedFn(idx)) + vectors, err := idx.embedAllChunks(context.Background(), makeTexts(30), 3, passthroughEmbedFn(idx)) require.NoError(t, err) require.Len(t, vectors, 30) assert.Equal(t, 1, emb.peak, "an embedder without Concurrent() must be driven serially (peak in-flight 1)") } +func TestEmbedAllChunks_SerialHonorsParentCancellation(t *testing.T) { + emb := &serialOnlyEmbedder{} + idx := newPoolTestIndexer(t, emb, 8) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + + vectors, err := idx.embedAllChunks(ctx, makeTexts(3), 1, passthroughEmbedFn(idx)) + require.ErrorIs(t, err, context.Canceled) + assert.Nil(t, vectors) + assert.Equal(t, 0, emb.peak, "a cancelled owner must not start serial embedding") +} + +func TestEmbedAllChunks_ParallelHonorsCancellationWhenProviderReturnsLate(t *testing.T) { + idx := newPoolTestIndexer(t, &poolEmbedder{}, 2) + ctx, cancel := context.WithCancel(context.Background()) + started := make(chan struct{}) + release := make(chan struct{}) + var startedOnce sync.Once + + done := make(chan struct { + vectors [][]float32 + err error + }, 1) + go func() { + vectors, err := idx.embedAllChunks(ctx, makeTexts(4), 1, func(_ context.Context, items []string) ([][]float32, error) { + startedOnce.Do(func() { close(started) }) + <-release // Model a provider that ignores cancellation and returns late. + vectors := make([][]float32, len(items)) + for i := range vectors { + vectors[i] = []float32{1, 0, 0} + } + return vectors, nil + }) + done <- struct { + vectors [][]float32 + err error + }{vectors: vectors, err: err} + }() + + select { + case <-started: + case <-time.After(5 * time.Second): + t.Fatal("parallel embedding did not start") + } + cancel() + close(release) + + select { + case result := <-done: + require.ErrorIs(t, result.err, context.Canceled) + assert.Nil(t, result.vectors, "cancelled embedding must not return late vectors") + case <-time.After(5 * time.Second): + t.Fatal("parallel embedding did not stop after cancellation") + } +} + // TestEmbedAllChunks_ConcurrencyCapIsBinding proves the configured cap // is the *binding* constraint, not merely an upper bound that the // workload happens to stay under. A barrier embedder blocks each call @@ -242,7 +298,7 @@ func TestEmbedAllChunks_ConcurrencyCapIsBinding(t *testing.T) { texts := makeTexts(30) // batch 2 → 15 batches done := make(chan error, 1) go func() { - _, err := idx.embedAllChunks(texts, 2, passthroughEmbedFn(idx)) + _, err := idx.embedAllChunks(context.Background(), texts, 2, passthroughEmbedFn(idx)) done <- err }() @@ -276,7 +332,7 @@ func TestEmbedAllChunks_ZeroCapUsesDefault(t *testing.T) { // More batches than the default so the pool can saturate it. texts := makeTexts(40) // batch 2 → 20 batches - vectors, err := idx.embedAllChunks(texts, 2, passthroughEmbedFn(idx)) + vectors, err := idx.embedAllChunks(context.Background(), texts, 2, passthroughEmbedFn(idx)) require.NoError(t, err) require.Len(t, vectors, len(texts)) @@ -296,7 +352,7 @@ func TestSetEmbeddingAPIConcurrency_FlowsToPool(t *testing.T) { emb := &poolEmbedder{delay: 5 * time.Millisecond} idx := newPoolTestIndexer(t, emb, 1) // cap 1 → no overlap - vectors, err := idx.embedAllChunks(makeTexts(20), 2, passthroughEmbedFn(idx)) + vectors, err := idx.embedAllChunks(context.Background(), makeTexts(20), 2, passthroughEmbedFn(idx)) require.NoError(t, err) require.Len(t, vectors, 20) assert.Equal(t, 1, emb.peak, diff --git a/internal/indexer/extraction_lifecycle.go b/internal/indexer/extraction_lifecycle.go new file mode 100644 index 000000000..a08547ca2 --- /dev/null +++ b/internal/indexer/extraction_lifecycle.go @@ -0,0 +1,107 @@ +package indexer + +import ( + "errors" + "sync" + + "github.com/zzet/gortex/internal/config" + "github.com/zzet/gortex/internal/parser" +) + +// ErrIndexerClosed is returned when extraction is attempted after lifecycle +// teardown has begun. +var ErrIndexerClosed = errors.New("indexer is closed") + +// extractionLifecycle is the admission/close boundary shared by every parse +// route. Admission and close are serialized under mu so close cannot race a +// WaitGroup Add; close waits for timed-out background extracts as well as +// synchronous and crash-worker requests. +type extractionLifecycle struct { + mu sync.Mutex + cond *sync.Cond + closed bool + active int +} + +func (l *extractionLifecycle) admit() (func(), error) { + l.mu.Lock() + if l.cond == nil { + l.cond = sync.NewCond(&l.mu) + } + if l.closed { + l.mu.Unlock() + return nil, ErrIndexerClosed + } + l.active++ + l.mu.Unlock() + + var once sync.Once + return func() { + once.Do(func() { + l.mu.Lock() + l.active-- + if l.active == 0 { + l.cond.Broadcast() + } + l.mu.Unlock() + }) + }, nil +} + +func (l *extractionLifecycle) closeAndWait() { + l.mu.Lock() + if l.cond == nil { + l.cond = sync.NewCond(&l.mu) + } + l.closed = true + for l.active > 0 { + l.cond.Wait() + } + l.mu.Unlock() +} + +// initializeExtractionOptions loads repository-local parser configuration once +// per Indexer lifetime. The parser value is immutable and safe for concurrent +// extraction requests. +func (idx *Indexer) initializeExtractionOptions(root string) { + idx.extractionOptionsOnce.Do(func() { + opts := parser.NewExtractionOptions(config.LoadLocalTemporalEnvHelpers(root)) + idx.extractionOptions.Store(&opts) + }) +} + +func (idx *Indexer) extractionOptionsValue() parser.ExtractionOptions { + if opts := idx.extractionOptions.Load(); opts != nil { + return *opts + } + return parser.ExtractionOptions{} +} + +// ExtractBuffer parses an in-memory file through the same repository options +// and lifecycle admission used by disk indexing. MCP overlays use this path to +// maintain base/overlay parity. +func (idx *Indexer) ExtractBuffer(language, relPath string, src []byte) (*parser.ExtractionResult, error) { + if idx == nil || idx.registry == nil { + return nil, errors.New("indexer has no parser registry") + } + ext, ok := idx.registry.GetByLanguage(language) + if !ok || ext == nil { + return nil, errors.New("indexer has no extractor for language " + language) + } + release, err := idx.extractionLifecycle.admit() + if err != nil { + return nil, err + } + defer release() + return safeExtractWithOptions(ext, relPath, src, idx.extractionOptionsValue()) +} + +// Close rejects new extraction admissions, waits for every admitted request, +// then terminates the long-lived crash-isolation workers. It is idempotent. +func (idx *Indexer) Close() { + if idx == nil { + return + } + idx.extractionLifecycle.closeAndWait() + idx.closeParsePool() +} diff --git a/internal/indexer/extraction_lifecycle_test.go b/internal/indexer/extraction_lifecycle_test.go new file mode 100644 index 000000000..71d84ecda --- /dev/null +++ b/internal/indexer/extraction_lifecycle_test.go @@ -0,0 +1,130 @@ +package indexer + +import ( + "context" + "errors" + "sync" + "testing" + "time" + + "go.uber.org/zap" + + "github.com/zzet/gortex/internal/config" + "github.com/zzet/gortex/internal/graph" + "github.com/zzet/gortex/internal/parser" +) + +type blockingLifecycleExtractor struct { + started chan struct{} + release chan struct{} + once sync.Once +} + +func (e *blockingLifecycleExtractor) Language() string { return "blocking" } +func (e *blockingLifecycleExtractor) Extensions() []string { return []string{".block"} } +func (e *blockingLifecycleExtractor) Extract(string, []byte) (*parser.ExtractionResult, error) { + e.once.Do(func() { close(e.started) }) + <-e.release + return &parser.ExtractionResult{}, nil +} + +func TestIndexerCloseWaitsForTimedOutExtraction(t *testing.T) { + ext := &blockingLifecycleExtractor{started: make(chan struct{}), release: make(chan struct{})} + reg := parser.NewRegistry() + reg.Register(ext) + idx := New(graph.New(), reg, config.IndexConfig{MaxExtractMillis: 1}, zap.NewNop()) + + _, err := idx.extractWithTimeoutDone(ext, "slow.block", []byte("source"), nil) + if !errors.Is(err, errExtractTimeout) { + t.Fatalf("extract error = %v, want timeout", err) + } + select { + case <-ext.started: + case <-time.After(time.Second): + t.Fatal("extractor did not start") + } + + closed := make(chan struct{}) + go func() { + idx.Close() + close(closed) + }() + select { + case <-closed: + t.Fatal("Close returned while timed-out extraction was still active") + case <-time.After(25 * time.Millisecond): + } + close(ext.release) + select { + case <-closed: + case <-time.After(time.Second): + t.Fatal("Close did not finish after extraction released") + } + + if _, err := idx.ExtractBuffer("blocking", "after.block", nil); !errors.Is(err, ErrIndexerClosed) { + t.Fatalf("post-close extraction error = %v, want ErrIndexerClosed", err) + } + idx.Close() +} + +func TestMultiIndexerCloseDrainsMutationAndClosesIndexers(t *testing.T) { + reg := parser.NewRegistry() + ext := &blockingLifecycleExtractor{started: make(chan struct{}), release: make(chan struct{})} + reg.Register(ext) + idx := New(graph.New(), reg, config.IndexConfig{}, zap.NewNop()) + + coordinator := newRepositoryMutationCoordinator(nil) + mi := &MultiIndexer{ + indexers: map[string]*Indexer{"repo": idx}, + repos: map[string]*RepoMetadata{"repo": {RepoPrefix: "repo"}}, + repositoryMutations: map[string]*repositoryMutationCoordinator{"repo": coordinator}, + } + mutationStarted := make(chan struct{}) + mutationRelease := make(chan struct{}) + mutationDone := make(chan error, 1) + go func() { + mutationDone <- coordinator.runExclusive(context.Background(), func() error { + close(mutationStarted) + <-mutationRelease + return nil + }) + }() + <-mutationStarted + + closeDone := make(chan error, 1) + go func() { closeDone <- mi.Close(context.Background()) }() + deadline := time.After(time.Second) + for !mi.isClosed() { + select { + case <-deadline: + t.Fatal("MultiIndexer did not close mutation admission") + default: + time.Sleep(time.Millisecond) + } + } + if err := mi.repositoryMutationCoordinator("new").runExclusive(context.Background(), func() error { return nil }); !errors.Is(err, errRepositoryMutationCoordinatorClosed) { + t.Fatalf("new mutation error = %v", err) + } + select { + case err := <-closeDone: + t.Fatalf("Close returned before mutation drained: %v", err) + case <-time.After(25 * time.Millisecond): + } + + close(mutationRelease) + if err := <-mutationDone; err != nil { + t.Fatal(err) + } + if err := <-closeDone; err != nil { + t.Fatal(err) + } + if got := mi.GetIndexer("repo"); got != nil { + t.Fatalf("detached indexer = %#v", got) + } + if _, err := idx.ExtractBuffer("blocking", "after.block", nil); !errors.Is(err, ErrIndexerClosed) { + t.Fatalf("owned indexer was not closed: %v", err) + } + if err := mi.Close(context.Background()); err != nil { + t.Fatalf("repeated Close: %v", err) + } +} diff --git a/internal/indexer/extraction_options_test.go b/internal/indexer/extraction_options_test.go new file mode 100644 index 000000000..a588fb347 --- /dev/null +++ b/internal/indexer/extraction_options_test.go @@ -0,0 +1,129 @@ +package indexer + +import ( + "fmt" + "os" + "path/filepath" + "sync" + "testing" + + "go.uber.org/zap" + + "github.com/zzet/gortex/internal/config" + "github.com/zzet/gortex/internal/graph" + "github.com/zzet/gortex/internal/parser" + "github.com/zzet/gortex/internal/parser/languages" +) + +func writeIndexerTemporalAllowlist(t *testing.T, root, helper string) { + t.Helper() + dir := filepath.Join(root, ".gortex") + if err := os.MkdirAll(dir, 0o755); err != nil { + t.Fatal(err) + } + content := []byte("env_helpers:\n - " + helper + "\n") + if err := os.WriteFile(filepath.Join(dir, "temporal-allowlist.yaml"), content, 0o600); err != nil { + t.Fatal(err) + } +} + +func indexerTemporalSource(helper string) []byte { + return []byte(fmt.Sprintf(`package sample +import "go.temporal.io/sdk/workflow" +func Run(ctx workflow.Context) { + name := %s("ACTIVITY", "MyActivity") + workflow.ExecuteActivity(ctx, name) +} +`, helper)) +} + +func indexerTemporalMeta(result *parser.ExtractionResult) map[string]any { + if result == nil { + return nil + } + defer result.ReleaseTree() + for _, edge := range result.Edges { + if edge != nil && edge.Meta != nil && edge.Meta["via"] == "temporal.stub" { + return edge.Meta + } + } + return nil +} + +func TestRepositoryExtractionOptionsIsolationAndOverlayParity(t *testing.T) { + t.Setenv(config.LocalTemporalOptInEnv, "true") + rootA := t.TempDir() + rootB := t.TempDir() + writeIndexerTemporalAllowlist(t, rootA, "RepoAHelper") + writeIndexerTemporalAllowlist(t, rootB, "RepoBHelper") + + reg := parser.NewRegistry() + languages.RegisterAll(reg) + idxA := New(graph.New(), reg, config.IndexConfig{}, zap.NewNop()) + idxB := New(graph.New(), reg, config.IndexConfig{}, zap.NewNop()) + idxA.SetRootPath(rootA) + idxB.SetRootPath(rootB) + defer idxA.Close() + defer idxB.Close() + type outcome struct { + owner string + meta map[string]any + err error + } + out := make(chan outcome, 4) + var wg sync.WaitGroup + for _, testCase := range []struct { + idx *Indexer + owner string + helper string + }{ + {idx: idxA, owner: "A-own", helper: "RepoAHelper"}, + {idx: idxA, owner: "A-foreign", helper: "RepoBHelper"}, + {idx: idxB, owner: "B-own", helper: "RepoBHelper"}, + {idx: idxB, owner: "B-foreign", helper: "RepoAHelper"}, + } { + testCase := testCase + wg.Add(1) + go func() { + defer wg.Done() + result, err := testCase.idx.ExtractBuffer("go", testCase.owner+".go", indexerTemporalSource(testCase.helper)) + out <- outcome{owner: testCase.owner, meta: indexerTemporalMeta(result), err: err} + }() + } + wg.Wait() + close(out) + for result := range out { + if result.err != nil { + t.Errorf("%s: %v", result.owner, result.err) + continue + } + want := any(nil) + if result.owner == "A-own" || result.owner == "B-own" { + want = "allowlist" + } + if result.meta == nil { + t.Errorf("%s: Temporal edge missing", result.owner) + continue + } + if got := result.meta["temporal_env_source"]; got != want { + t.Errorf("%s source = %#v, want %#v", result.owner, got, want) + } + } + + ext, _ := reg.GetByLanguage("go") + base, err := idxA.extractWithTimeoutDone(ext, "base.go", indexerTemporalSource("RepoAHelper"), nil) + if err != nil { + t.Fatal(err) + } + overlay, err := idxA.ExtractBuffer("go", "overlay.go", indexerTemporalSource("RepoAHelper")) + if err != nil { + t.Fatal(err) + } + baseMeta := indexerTemporalMeta(base) + overlayMeta := indexerTemporalMeta(overlay) + for _, key := range []string{"temporal_name", "temporal_name_origin", "temporal_env_source", "temporal_kind"} { + if baseMeta[key] != overlayMeta[key] { + t.Fatalf("base/overlay %s mismatch: %#v != %#v", key, baseMeta[key], overlayMeta[key]) + } + } +} diff --git a/internal/indexer/file_meta.go b/internal/indexer/file_meta.go index 6a5ef8539..c9c4187f2 100644 --- a/internal/indexer/file_meta.go +++ b/internal/indexer/file_meta.go @@ -182,12 +182,9 @@ func persistFileMetaRows(target graph.Store, repoPrefix string, rows []graph.Fil const reparsePendingEnrichmentBatchSize = 256 type reparsePendingEnrichmentBatch struct { - byFile map[string]bool - deferResolverCatchup bool - deferredAffectedFiles map[string]struct{} - deferredAffectedPasses int64 - deferredAffectedResolved int64 - deferredAffectedDropped int64 + byFile map[string]bool + deferResolverCatchup bool + deferredAffectedFiles map[string]struct{} } func (b *reparsePendingEnrichmentBatch) add(graphPath string, pending bool) bool { diff --git a/internal/indexer/incremental_batch.go b/internal/indexer/incremental_batch.go index 0259a6b38..a62781733 100644 --- a/internal/indexer/incremental_batch.go +++ b/internal/indexer/incremental_batch.go @@ -1290,10 +1290,7 @@ func affectedByDeltaFromExtraction( } type affectedByBatchPlan struct { - files []string - passes int64 - resolved int64 - dropped int64 + files []string } func (b *reparsePendingEnrichmentBatch) mergeDeferredAffected(plan affectedByBatchPlan) { @@ -1308,9 +1305,6 @@ func (b *reparsePendingEnrichmentBatch) mergeDeferredAffected(plan affectedByBat b.deferredAffectedFiles[filePath] = struct{}{} } } - b.deferredAffectedPasses += plan.passes - b.deferredAffectedResolved += plan.resolved - b.deferredAffectedDropped += plan.dropped } func (b *reparsePendingEnrichmentBatch) deferredAffectedPlan() affectedByBatchPlan { @@ -1322,10 +1316,7 @@ func (b *reparsePendingEnrichmentBatch) deferredAffectedPlan() affectedByBatchPl files = append(files, filePath) } sort.Strings(files) - return affectedByBatchPlan{ - files: files, passes: b.deferredAffectedPasses, - resolved: b.deferredAffectedResolved, dropped: b.deferredAffectedDropped, - } + return affectedByBatchPlan{files: files} } func (idx *Indexer) planAffectedByStages(stages []*incrementalBatchStage) affectedByBatchPlan { @@ -1406,14 +1397,11 @@ func (idx *Indexer) planAffectedByStages(stages []*incrementalBatchStage) affect idx.logger.Debug("affected-by: re-resolve set truncated", zap.String("file", changedPath), zap.Int("affected", len(files)), zap.Int("cap", maxFiles), zap.Int("dropped", len(files)-maxFiles)) - plan.dropped += int64(len(files) - maxFiles) files = files[:maxFiles] } if len(files) == 0 { continue } - plan.passes++ - plan.resolved += int64(len(files)) for _, filePath := range files { union[filePath] = struct{}{} } @@ -1427,14 +1415,9 @@ func (idx *Indexer) planAffectedByStages(stages []*incrementalBatchStage) affect } func (idx *Indexer) executeAffectedByPlan(plan affectedByBatchPlan) { - if plan.dropped > 0 { - idx.affectedByDropped.Add(plan.dropped) - } if len(plan.files) == 0 { return } - idx.affectedByPasses.Add(plan.passes) - idx.affectedByFilesResolved.Add(plan.resolved) idx.observeIncrementalCatchup("affected_by", plan.files) idx.resolver.ResolveFilesAndIncoming(plan.files) resolver.SynthesizeExternalCallsForFiles(idx.graph, idx.externalCallSynthesisEnabled(), plan.files) diff --git a/internal/indexer/incremental_reindex_paths_test.go b/internal/indexer/incremental_reindex_paths_test.go index 433809464..c51d6331c 100644 --- a/internal/indexer/incremental_reindex_paths_test.go +++ b/internal/indexer/incremental_reindex_paths_test.go @@ -5,6 +5,7 @@ import ( "os" "path/filepath" "testing" + "time" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -194,7 +195,7 @@ func TestIncrementalDiscoverPaths_PreservesDeletedTrackedFile(t *testing.T) { require.NotEmpty(t, g.FindNodesByName("Original")) require.NoError(t, os.Remove(gone)) - res, err := idx.incrementalDiscoverPaths(dir, nil) + res, err := idx.incrementalReindexPathsMode(dir, nil, incrementalPathMode{detectDeletions: false}) require.NoError(t, err) require.NotNil(t, res) assert.Zero(t, res.DeletedFileCount) @@ -309,6 +310,26 @@ func TestIncrementalReindexPathsReportsRepoFileCountNotScopeSize(t *testing.T) { // TestIncrementalReindexPathsFileCountTracksAddsAndDeletes verifies the // repo-wide count still moves when the repo itself changes size, so the // fix reports a live total rather than a frozen one. +func TestIncrementalReindexPaths_DoesNotReembedTheWholeUnchangedCorpus(t *testing.T) { + root := t.TempDir() + path := filepath.Join(root, "main.py") + writeFile(t, path, "def alpha():\n return 1\n") + emb := &poolEmbedder{} + idx := newVectorPersistIndexer(t, graph.New(), emb) + _, err := idx.Index(root) + require.NoError(t, err) + before := emb.calls + require.Greater(t, before, int32(0)) + + writeFile(t, path, "def alpha():\n return 2\n") + future := time.Now().Add(time.Second) + require.NoError(t, os.Chtimes(path, future, future)) + _, err = idx.IncrementalReindexPaths(root, []string{path}) + require.NoError(t, err) + require.Equal(t, before, emb.calls, + "changed-file reconciliation must not trigger a paid full-corpus embedding pass") +} + func TestIncrementalReindexPathsFileCountTracksAddsAndDeletes(t *testing.T) { dir := t.TempDir() writeFile(t, filepath.Join(dir, "a.go"), "package main\n\nfunc A() {}\n") diff --git a/internal/indexer/indexer.go b/internal/indexer/indexer.go index 1a9655f6a..8ddef4e6c 100644 --- a/internal/indexer/indexer.go +++ b/internal/indexer/indexer.go @@ -1,7 +1,6 @@ package indexer import ( - "bytes" "context" "encoding/json" "encoding/xml" @@ -11,7 +10,6 @@ import ( "path/filepath" "regexp" "sort" - "strconv" "strings" "sync" "sync/atomic" @@ -223,6 +221,14 @@ type Indexer struct { parseQuar *crashpool.Quarantine parsePoolMu sync.Mutex + // extractionLifecycle rejects new parses once Close begins and waits for + // every admitted in-process, crash-worker, streaming, and overlay request. + extractionLifecycle extractionLifecycle + // extractionOptions is loaded once after the repository root is established. + // The pointed-to value is immutable for the Indexer's lifetime. + extractionOptionsOnce sync.Once + extractionOptions atomic.Pointer[parser.ExtractionOptions] + // Trigram code-search index, lazily built on first GrepText call // and rebuilt only when indexGen advances past the build it was // made from. indexGen is bumped by every full or incremental @@ -275,33 +281,14 @@ type Indexer struct { // embedder is the optional embedding provider for semantic search. embedder embedding.Provider - // skipVectorBuild, when true, makes buildSearchIndex populate only - // the text index and never run the embedding pass — even with an - // embedder set. The daemon flips it on for the warmup re-index loop - // when a snapshot already carries the workspace vector index, so - // the graph is not re-embedded only to have the cached index - // overwrite it. Off by default; a normal index always builds - // vectors when an embedder is present. - skipVectorBuild bool - - // bulkVectorSink holds the disk store captured at the bulk-load - // shadow swap, so buildSearchIndex can still persist the vector - // index to the backend while idx.graph points at the in-memory - // shadow (which does not implement graph.VectorSearcher). Without - // it the embedding pass under the bulk loader builds vectors only - // in the in-process HNSW — they never reach the `vectors` table and - // are lost on the next daemon restart, forcing a paid re-embed. - // Set during the shadow swap, cleared when idx.graph is restored. - bulkVectorSink graph.VectorSearcher - - // contentSink mirrors bulkVectorSink for the content full-text index: + // contentSink captures the durable content full-text index during a shadow: // the disk store captured at the shadow swap, so the per-file content // stream reaches content_fts on disk even while idx.graph points at the // in-memory shadow (which does not implement graph.ContentSearcher). // Set during the shadow swap, cleared when idx.graph is restored. contentSink graph.ContentSearcher - // contractStateSink mirrors bulkVectorSink for the contract-tier + // contractStateSink captures the durable contract-tier store during a shadow: // completion marker: the disk store captured at the shadow swap, so the // inline contract pass records the marker on the backend even while // idx.graph points at the in-memory shadow (which does not implement @@ -338,12 +325,6 @@ type Indexer struct { // semanticMgr is the optional semantic enrichment manager. semanticMgr *semantic.Manager - // resolverLSPHelper, when non-nil, is the resolve-time LSP - // helper installed on idx.resolver. Held here so MultiIndexer - // can mirror it onto the global post-pass resolver in - // RunDeferredPassesAll. See SetResolverLSPHelper. - resolverLSPHelper resolver.LSPHelper - // npmAliasOnce builds npmAlias lazily on the first resolve-time // import-rewrite request. Lazy because the repo root and prefix // are set after New(); by the time the resolver runs they are @@ -379,19 +360,6 @@ type Indexer struct { contractCache map[string]*contractCacheEntry contractCacheMu sync.RWMutex - // upgradeOnce gates the BM25→Bleve auto-upgrade to exactly one - // goroutine per indexer lifetime. Without this, every post-threshold - // IndexCtx — which fires once per tracked repo during multi-repo - // warmup — would spawn a fresh upgradeSearchToBleve goroutine. - // Each rebuilds ~N-doc Bleve indexes (≈32 KiB/doc), so overlapping - // upgrades peak memory far above steady-state and waste CPU on - // rebuilds that the next Swap immediately discards. Also counts - // scheduled upgrades so tests can observe gating decisions - // without relying on log scrapes or timing. - upgradeOnce sync.Once - upgradeSpawnedMu sync.Mutex - upgradeSpawned int - // deferResolve, when set, makes IndexCtx skip the cross-cutting passes // (per-repo ResolveAll / semantic enrichment / contract extraction + // commit) so the multi-repo orchestrator can run them serially after @@ -422,7 +390,7 @@ type Indexer struct { // graph-apply phase until the channel closes. Set by BeginDeferredPasses // before the pool launches (so the pool's compute may overlap the warmup // resolve without its applies starving the resolver) and cleared by - // FinishTail. + // FinishTailResult. deferredApplyGate <-chan struct{} // deferredEnrichFiles holds a deduplicated, repo-scoped Go frontier for @@ -467,14 +435,14 @@ type Indexer struct { // entire shared graph, so running them per-repo inside a batch loop // (warmup, ReconcileAll) is O(R · global_size) — quadratic for repo // counts in the hundreds. The batch caller is responsible for invoking - // RunGlobalGraphPasses exactly once at the end. Has no effect on the + // the shared global-pass pipeline exactly once at the end. Has no effect on the // deferResolve path (multi-repo IndexCtx already skips those passes). deferGlobalPasses atomic.Bool // skipResolveInDeferred, when set, makes RunDeferredPasses skip the // per-repo resolver.ResolveAll() call. ResolveAll walks the entire // shared graph, so paying it once per indexer across hundreds of - // repos is O(R · E). MultiIndexer.RunDeferredPassesAll sets this + // repos is O(R · E). MultiIndexer.RunDeferredPassesAllResult sets this // flag on every indexer and runs a single resolver.New(graph).ResolveAll // once at the end, which picks up every placeholder edge at once. // Has no effect on direct (non-batch) callers of RunDeferredPasses. @@ -506,16 +474,6 @@ type Indexer struct { preparedMu sync.Mutex prepared map[string]*preparedExtraction - // affectedByPasses / affectedByFilesResolved / affectedByDropped - // count the affected-by re-resolution activity (see affected_by.go): - // passes that found a signature delta and ran, referencing files - // re-resolved by them, and files dropped by the fan-out cap. - // Exposed via AffectedByCounts so tests and diagnostics can observe - // that a body-only edit triggered no fan-out. - affectedByPasses atomic.Int64 - affectedByFilesResolved atomic.Int64 - affectedByDropped atomic.Int64 - // repositoryMutation is the single discovery/parse/resolve/derived lane // for this repository. It is lazy so focused Indexer fixtures do not need // constructor changes, while every production entry point shares it. @@ -551,18 +509,18 @@ func New(g graph.Store, reg *parser.Registry, cfg config.IndexConfig, logger *za indexMemoryAdmission: processIndexMemoryAdmission, registry: reg, resolver: resolver.New(g), - // Wrap in Swappable so the auto-upgrade to Bleve at large - // corpus sizes can happen in a background goroutine without - // racing with concurrent searches. Subsequent reassignments to - // idx.search (Hybrid wrap, etc.) should use swap helpers below. + // Wrap in Swappable so the later Hybrid re-wrap (text + + // vector) can happen without racing with concurrent searches. + // Subsequent reassignments to idx.search should use the swap + // helpers below. // // When the backing store implements graph.SymbolSearcher // (today only store_sqlite), the initial backend is a thin // adapter that forwards Search to the store's native FTS. - // The in-process Bleve / BM25 build path is then bypassed - // entirely — saving ~100MB heap on a Vscode-scale repo and - // putting search in the same address space as the rest of - // the graph queries. + // The in-process BM25 build path is then bypassed entirely — + // saving ~100MB heap on a Vscode-scale repo and putting + // search in the same address space as the rest of the graph + // queries. search: search.NewSwappable(initialSearchBackend(g)), config: cfg, transforms: newTransformPipeline(cfg.Transforms, logger), @@ -710,10 +668,8 @@ func (d *vectorSearcherDelegate) SimilarTo(vec []float32, limit int) ([]graph.Ve // in its Swappable on construction. When the underlying store // implements graph.SymbolSearcher (today only store_sqlite), a // thin adapter routes Search calls through the store's native FTS -// — the in-process BM25 / Bleve build path is bypassed entirely. -// Otherwise falls through to search.NewAuto which picks BM25 for -// small corpora and auto-upgrades to Bleve once the size warrants -// it. +// — the in-process BM25 build path is bypassed entirely. Otherwise +// falls through to search.NewAuto's in-memory BM25 index. func initialSearchBackend(g graph.Store) search.Backend { if s, ok := g.(graph.SymbolSearcher); ok { return search.NewSymbolSearcherBackend(s) @@ -722,18 +678,19 @@ func initialSearchBackend(g graph.Store) search.Backend { } // isSymbolSearcherBackend reports whether the swappable's currently -// active backend is the SymbolSearcher adapter. Used to suppress -// the Bleve auto-upgrade goroutine — if the active backend is -// already a native FTS, upgrading to Bleve would re-index the same -// corpus into a parallel in-process Bleve and silently swap it in, -// defeating the FTS path and pinning the ~100MB heap the FTS +// active backend is the SymbolSearcher adapter. Used to suppress the +// in-process index builds — if the active backend is already a native +// FTS, re-indexing the same corpus into a parallel in-process index +// would defeat the FTS path and pin the ~100MB heap the FTS // integration was meant to release. func isSymbolSearcherBackend(b search.Backend) bool { switch backend := b.(type) { case *search.SymbolSearcherBackend: return true case *search.Swappable: - return isSymbolSearcherBackend(backend.Inner()) + inner, release := backend.AcquireBackend() + defer release() + return isSymbolSearcherBackend(inner) case *search.HybridBackend: return isSymbolSearcherBackend(backend.TextBackend()) default: @@ -863,13 +820,12 @@ func (idx *Indexer) populateSymbolFTS(reporter progress.Reporter) error { } // shouldIndexForSearch reports whether a node should be added to the -// text search index (BM25/Bleve). File and Import nodes are never +// text search index. File and Import nodes are never // searchable symbols. Beyond that, config.SkipSearch filters out // (language, kind) pairs that would only add noise — JSON/YAML/TOML -// keys, CSS tokens, Terraform blocks, shell/build variables. All three -// text-index call sites (buildSearchIndex bulk loop, indexFile -// incremental add, upgradeSearchToBleve repopulate) must go through -// this predicate so they can't drift. +// keys, CSS tokens, Terraform blocks, shell/build variables. Every +// text-index call site (buildSearchIndex bulk loop, indexFile +// incremental add) must go through this predicate so they can't drift. func (idx *Indexer) shouldIndexForSearch(n *graph.Node) bool { // Cross-daemon proxy-edge nodes stand in for remote symbols; they // are never surfaced in local name search. Inert until @@ -941,121 +897,6 @@ func (idx *Indexer) removeFromSearch(n *graph.Node) { idx.search.Remove(n.ID) } -// upgradeSearchToBleve constructs a Bleve backend from the current graph -// and atomically swaps it in. Designed to run in a background goroutine -// triggered by IndexCtx after the initial index completes. Does nothing -// if Bleve construction fails (caller already hit AutoThreshold but the -// in-memory backend keeps serving correctly, just with worse memory -// characteristics). -// bleveUpgradeEntry is one row of the snapshot the upgrade goroutine -// works from. Snapshotting (id, name, file, signature) up front in -// the foreground — before the goroutine starts reading them — keeps -// the goroutine race-free against subsequent Index calls' Meta-writing -// passes (reach.BuildIndex, ResolveTemporalCalls, ...). -type bleveUpgradeEntry struct { - id string - // fields is the BM25 text payload for the node, as produced by - // searchIndexFields: name + file + signature for a code symbol, - // name + file + section body for a KindDoc prose section. - fields []string -} - -// snapshotBleveEntries captures every node currently eligible for the -// search index plus its `signature` Meta string. Called synchronously -// from IndexCtx after every Node.Meta mutating pass has returned, so -// the read of n.Meta happens with no concurrent writer. -func (idx *Indexer) snapshotBleveEntries() []bleveUpgradeEntry { - nodes := graph.RepoCodeNodes(idx.graph, idx.repoPrefix) - out := make([]bleveUpgradeEntry, 0, len(nodes)) - for _, n := range nodes { - if !idx.shouldIndexForSearch(n) { - continue - } - out = append(out, bleveUpgradeEntry{id: n.ID, fields: searchIndexFields(n, idx.projectName)}) - } - return out -} - -func (idx *Indexer) upgradeSearchToBleve(snapshot []bleveUpgradeEntry) { - // Defensive early-return: if the active text backend is already - // Bleve, there is nothing to upgrade. IndexCtx's sync.Once guard - // prevents re-entry from the auto-upgrade path, but direct - // callers (tests, manual invocation from tooling) could still - // hit this function twice; a second run would pointlessly - // rebuild a full Bleve index and Swap it over an identical one. - inner := idx.swappable().Inner() - if _, ok := inner.(*search.BleveBackend); ok { - return - } - if hyb, ok := inner.(*search.HybridBackend); ok { - if _, ok := hyb.TextBackend().(*search.BleveBackend); ok { - return - } - } - - // Opt-in disk backend. Scorch stores the inverted index on disk - // (~10-20× less heap than upsidedown+gtreap) at the cost of file - // I/O during build. Users point GORTEX_BLEVE_DISK_DIR at a - // writable path; we manage the file lifecycle inside it. - diskDir := os.Getenv("GORTEX_BLEVE_DISK_DIR") - - var ( - blv *search.BleveBackend - err error - ) - if diskDir != "" { - blv, err = search.NewBleveDisk(diskDir) - if err != nil { - idx.logger.Warn("search: bleve disk construction failed, falling back to in-memory", - zap.String("dir", diskDir), zap.Error(err)) - } - } - if blv == nil { - blv, err = search.NewBleve() - if err != nil { - idx.logger.Warn("search: bleve construction failed, staying on in-memory", - zap.Error(err)) - return - } - } - - // Use the foreground snapshot the spawner captured rather than - // re-walking idx.graph here: the goroutine outlives the spawning - // IndexCtx call, and subsequent Index calls' Meta-writing passes - // (reach.BuildIndex, ResolveTemporalCalls, ...) mutate Node.Meta - // on the same Node objects. Reading sig from a live n.Meta here - // would race with those writes. - for _, e := range snapshot { - blv.Add(e.id, e.fields...) - } - - // Preserve the vector index if one is wired up. The previous inner - // is normally a HybridBackend wrapping text + vector + embedder; - // swapping in raw Bleve would let Swap's old.Close() run on the - // old Hybrid, which closes only its text side (hybrid.Close) but - // leaves the resulting inner — raw *BleveBackend — unwrapped, so - // every downstream hybrid/semantic query silently degrades to - // BM25 until the daemon restarts. Rewrap the fresh Bleve in a new - // Hybrid carrying the old vector + embedder. The vector backend - // itself is never closed by Hybrid.Close, so it stays alive even - // after the old Hybrid is torn down by Swap. - sw := idx.swappable() - var replacement search.Backend = blv - if oldHybrid, ok := sw.Inner().(*search.HybridBackend); ok { - replacement = search.NewHybrid(blv, oldHybrid.VectorIndex(), oldHybrid.Embedder()) - } - sw.Swap(replacement) - - mode := "memory" - if blv.DiskPath() != "" { - mode = "disk" - } - idx.logger.Info("search: upgraded to Bleve backend (background)", - zap.Int("symbols", blv.Count()), - zap.String("mode", mode), - zap.String("disk_path", blv.DiskPath())) -} - // Graph returns the underlying graph. func (idx *Indexer) Graph() graph.Store { return idx.graph } @@ -1101,120 +942,11 @@ func (idx *Indexer) SetSkipResolveInDeferred(v bool) { idx.skipResolveInDeferred // SetDeferGlobalPasses toggles whether the graph-wide derivation passes // (InferImplements, InferOverrides, markTestSymbolsAndEmitEdges) run // inline at the end of IndexCtx / IncrementalReindexPaths. Set true when the -// caller drives a batch (e.g. daemon warmup) and will invoke -// RunGlobalGraphPasses once at the end. See the deferGlobalPasses field +// caller drives a batch (e.g. daemon warmup) and will invoke the shared +// multi-repository global-pass pipeline once at the end. See the deferGlobalPasses field // comment. func (idx *Indexer) SetDeferGlobalPasses(v bool) { idx.deferGlobalPasses.Store(v) } -// RunGlobalGraphPasses runs the graph-wide derivation passes once -// against the indexer's shared graph. Safe to call against a graph that -// already has these edges — InferImplements / InferOverrides skip -// existing parents, and graph.AddEdge dedupes by edgeKey so EdgeTests -// re-emission is a no-op. Logs counts for telemetry. Use when batching -// multiple per-repo TrackRepoCtx / IncrementalReindexPaths calls under -// SetDeferGlobalPasses(true). -func (idx *Indexer) RunGlobalGraphPasses(ctx context.Context) { - if idx.graph == nil { - return - } - reporter := progress.FromContext(ctx) - if added := idx.resolver.InferImplements(); added > 0 { - idx.logger.Info("inferred implements (global)", zap.Int("added", added)) - } - if added := idx.resolver.InferOverrides(); added > 0 { - idx.logger.Info("inferred overrides (global)", zap.Int("added", added)) - } - marked, emitted := markTestSymbolsAndEmitEdges(idx.graph) - if marked > 0 || emitted > 0 { - idx.logger.Info("test edges emitted (global)", - zap.Int("test_symbols", marked), - zap.Int("edges", emitted), - ) - } - if ctrl := entrypoints.PropagateEntryPointsDownHierarchy(idx.graph); ctrl > 0 { - idx.logger.Info("entry-point hierarchy stamped (global)", zap.Int("stamped", ctrl)) - } - if re, ep, fa := synthesizeCapabilityEdges(idx.graph); re > 0 || ep > 0 || fa > 0 { - idx.logger.Info("capability edges emitted (global)", - zap.Int("reads_env", re), - zap.Int("executes_process", ep), - zap.Int("accesses_field", fa), - ) - } - reporter.Report("clone detection pass (global)", 0, 0) - cs, cloneBaseline := detectClonesAndEmitEdgesWithBaselineCtx(ctx, idx.graph, idx.repoPrefix, idx.cloneThreshold()) - if cs.Items > 0 { - idx.logger.Info("clone edges emitted (global)", - zap.Int("items", cs.Items), - zap.Int("clone_pairs", cs.Pairs), - zap.Int("edges", cs.Edges), - zap.Int("skipped_buckets", cs.SkippedBuckets), - zap.Int("skipped_bucket_items", cs.SkippedBucketItems), - zap.Int("diffused_pairs", cs.DiffusedPairs), - zap.Int("diffused_edges", cs.DiffusedEdges), - ) - } - // Adopt the freshly-finalized CMS/corpus seed so steady-state single-file - // edits go incremental without paging the same compact corpus again. - if idx.cloneIndex != nil { - idx.cloneIndex.AdoptBaselineOrRebuild(idx.graph, idx.repoPrefix, cloneBaseline) - } - // Framework dynamic-dispatch synthesis (gRPC stubs, Temporal - // workflow→activity, in-process / native event channels, native - // bridges). Runs after InferImplements/InferOverrides (the - // interface-satisfaction signals several synthesizers depend on) and - // before DetectCrossRepoEdges so a cross-repo synthesized call gets - // its parallel cross_repo_calls edge. - reporter.Report("framework dispatch synthesis (global)", 0, 0) - if rep := resolver.RunFrameworkSynthesizers(idx.graph); rep.Total > 0 { - idx.logger.Info("framework dispatch calls synthesized (global)", - zap.Int("edges", rep.Total), - zap.Any("per_synthesizer", rep.Per), - ) - } - // External-call placeholder synthesis (opt-in). Runs after the - // resolver and the gRPC/Temporal stub passes so every edge that - // could land on a real node already has; the leftover external - // terminals are then materialised into synthetic call-chain nodes. - reporter.Report("external-call synthesis (global)", 0, 0) - if extCalls := resolver.SynthesizeExternalCalls(idx.graph, idx.externalCallSynthesisEnabled()); extCalls > 0 { - idx.logger.Info("external-call placeholders synthesized (global)", - zap.Int("edges", extCalls), - ) - } - // Speculative dynamic-dispatch synthesis (opt-in, default off). Mints - // best-guess hidden-by-default call edges for blind-spot shapes. - if spec := resolver.ResolveSpeculativeDispatch(idx.graph, idx.speculativeDispatchEnabled()); spec > 0 { - idx.logger.Info("speculative dispatch edges synthesized (global)", - zap.Int("edges", spec), - ) - } - // Content -> code "why" links. Runs before DetectCrossRepoEdges so a - // chunk that motivates a symbol in another repo gets its parallel - // cross_repo_motivates edge minted by the cross-repo pass below. - reporter.Report("content links (global)", 0, 0) - idx.linkContentToCode() - // Cross-repo edge layer. Runs after InferImplements / InferOverrides - // so cross-repo implements / extends edges pick up their parallel - // cross_repo_* edges. No-op on single-repo graphs (no RepoPrefix). - reporter.Report("cross-repo edges (global)", 0, 0) - if crossRepoEdges := resolver.DetectCrossRepoEdges(idx.graph); crossRepoEdges > 0 { - idx.logger.Info("cross-repo edges emitted (global)", - zap.Int("edges", crossRepoEdges), - ) - } - // Reachability index — used to be precomputed here for every - // impact seed. The eager pass was retired because the breakeven - // math doesn't work: on a 200 k-seed graph (k8s) the build took - // ~2000 s of cold-index wall time to save ~10 ms per - // AnalyzeImpact call, requiring ~200 k queries to pay off — well - // beyond any realistic agent session. Lookups are now - // compute-on-first-use; we just invalidate the cache so any - // surviving stamps from a previous build don't shadow the fresh - // graph state. - reach.InvalidateIndex() -} - // cloneThreshold returns the configured Jaccard similarity cutoff for // clone detection (0 = use the clones package default). func (idx *Indexer) cloneThreshold() float64 { @@ -1230,7 +962,7 @@ func (idx *Indexer) cloneThreshold() float64 { // The graph-wide derivation passes (InferImplements, InferOverrides, // markTestSymbolsAndEmitEdges) intentionally do NOT run here. They walk // the entire shared graph, so the multi-repo orchestrator must invoke -// MultiIndexer.RunGlobalGraphPasses exactly once after every repo has +// its shared global-pass pipeline exactly once after every repo has // finished its deferred per-repo work. func (idx *Indexer) RunDeferredPasses(ctx context.Context) { if idx.pendingContractReg == nil { @@ -1594,6 +1326,7 @@ func (idx *Indexer) storeRootPath(absRoot string) { if idx.rootPath != absRoot { idx.rootPath = absRoot } + idx.initializeExtractionOptions(absRoot) } // populateCppIncludeDirs reconstructs each C/C++ source file's include search @@ -1751,13 +1484,6 @@ func (idx *Indexer) ProjectID() string { return idx.projectID } // When set, buildSearchIndex will create a HybridBackend with vector search. func (idx *Indexer) SetEmbedder(p embedding.Provider) { idx.embedder = p } -// SetSkipVectorBuild toggles the embedding pass in buildSearchIndex. -// When true, buildSearchIndex builds only the text index — used by the -// daemon warmup path when a snapshot already carries the workspace -// vector index, so the graph is not needlessly re-embedded. When false -// (the default) an indexer with an embedder set always builds vectors. -func (idx *Indexer) SetSkipVectorBuild(skip bool) { idx.skipVectorBuild = skip } - // SetEmbeddingChunkOptions tunes the AST sub-chunking applied to large // symbols before embedding (threshold and window line counts). The // zero value leaves the chunker on its built-in defaults. @@ -1797,66 +1523,12 @@ func (idx *Indexer) SemanticManager() *semantic.Manager { return idx.semanticMgr // // Pass nil to detach. Must be called before ResolveAll / ResolveFile; // the resolver caches no LSP state across passes, so mid-pass swaps -// are racy and not supported. +// are racy and not supported. The resolver owns the helper — this is a +// pass-through, not a second place it is stored. func (idx *Indexer) SetResolverLSPHelper(h resolver.LSPHelper) { if idx.resolver != nil { idx.resolver.SetLSPHelper(h) } - idx.resolverLSPHelper = h -} - -// ResolverLSPHelper returns the currently installed resolver-time LSP -// helper, or nil. Exported so MultiIndexer can mirror the helper onto -// the global post-pass resolver in RunDeferredPassesAll. -func (idx *Indexer) ResolverLSPHelper() resolver.LSPHelper { return idx.resolverLSPHelper } - -// ExportVectorIndex returns the serialized vector index bytes, dims, and count. -// Returns nil, 0, 0 if no vector index is active. -func (idx *Indexer) ExportVectorIndex() ([]byte, int, int) { - hybrid, ok := idx.swappable().Inner().(*search.HybridBackend) - if !ok { - return nil, 0, 0 - } - vec := hybrid.VectorIndex() - if vec == nil || vec.Count() == 0 { - return nil, 0, 0 - } - - var buf bytes.Buffer - if err := vec.Save(&buf); err != nil { - idx.logger.Warn("failed to export vector index", zap.Error(err)) - return nil, 0, 0 - } - return buf.Bytes(), vec.Dims(), vec.Count() -} - -// ImportVectorIndex restores a vector index from serialized data and wraps -// the current text search backend into a HybridBackend. -func (idx *Indexer) ImportVectorIndex(data []byte, dims, count int) error { - if len(data) == 0 || idx.embedder == nil { - return nil - } - - // Validate dimensions match the current embedder to avoid mismatches - // when switching providers (e.g., GloVe 50d → ONNX 384d). - embedderDims := idx.embedder.Dimensions() - if embedderDims > 0 && embedderDims != dims { - idx.logger.Info("vector index dims mismatch, will re-embed", - zap.Int("cached_dims", dims), zap.Int("embedder_dims", embedderDims)) - return nil // skip import, buildSearchIndex will re-embed - } - - vec := search.NewVector(dims) - if err := vec.LoadFrom(bytes.NewReader(data)); err != nil { - return fmt.Errorf("import vector index: %w", err) - } - vec.SetCount(count) - - sw := idx.swappable() - sw.Swap(search.NewHybrid(sw.Inner(), vec, idx.embedder)) - idx.logger.Info("restored vector index from cache", - zap.Int("vectors", count), zap.Int("dims", dims)) - return nil } // prefixPath prepends the repoPrefix to a relative path when in multi-repo mode. @@ -2817,6 +2489,7 @@ func (idx *Indexer) indexCtxRaw(ctx context.Context, root string) (result *Index // state. var diskTarget graph.Store var inMemShadow *graph.Graph + var deferredVectorPlan *preparedVectorPlan var shadowEstimate graph.RepoMemoryEstimate var shadowEstimateReady bool bl, blOK := idx.graph.(graph.BulkLoader) @@ -2988,14 +2661,8 @@ func (idx *Indexer) indexCtxRaw(ctx context.Context, root string) (result *Index // and edge through as it is parsed. idx.indexCount.Add(1) diskTarget = idx.graph - inMemShadow = graph.New() + inMemShadow = idx.newStructuralIntegrityShadow(diskTarget, graph.StructuralPathShadowCold) idx.graph = inMemShadow - // Capture the disk store as the vector sink: buildSearchIndex runs - // while idx.graph is the shadow (no VectorSearcher), so without this - // the embedded vectors never land on disk. The `vectors` table has no - // FK to `nodes`, so upserting before FlushBulk persists the nodes is - // safe. Cleared when idx.graph is restored below. - idx.bulkVectorSink, _ = diskTarget.(graph.VectorSearcher) // Same capture for the content index: the per-file content stream // must reach content_fts on disk while idx.graph is the shadow. idx.contentSink, _ = diskTarget.(graph.ContentSearcher) @@ -3014,8 +2681,11 @@ func (idx *Indexer) indexCtxRaw(ctx context.Context, root string) (result *Index } defer func() { if retErr != nil { + if deferredVectorPlan != nil { + deferredVectorPlan.Release() + deferredVectorPlan = nil + } idx.graph = diskTarget - idx.bulkVectorSink = nil idx.contentSink = nil idx.contractStateSink = nil if idx.resolver != nil { @@ -3170,17 +2840,26 @@ func (idx *Indexer) indexCtxRaw(ctx context.Context, root string) (result *Index } reporter.Report("persisting bulk graph", 1, 1) idx.graph = diskTarget - idx.bulkVectorSink = nil idx.contentSink = nil idx.contractStateSink = nil // Mirror of the SetGraph(inMemShadow) above: the resolver - // must follow the graph pointer back to the disk store, or - // every post-index per-file resolve (the watcher save path, - // incremental reindex) reads the drained — now empty — - // shadow and silently resolves nothing. + // must follow the graph pointer back to the disk store before vector + // ownership validation and every later incremental operation. if idx.resolver != nil { idx.resolver.SetGraph(diskTarget) } + if deferredVectorPlan != nil { + if retErr == nil { + plan := deferredVectorPlan + deferredVectorPlan = nil + if err := idx.installVectorPlan(ctx, diskTarget, plan); err != nil { + retErr = fmt.Errorf("indexer: publish vector corpus after shadow drain: %w", err) + } + } else { + deferredVectorPlan.Release() + deferredVectorPlan = nil + } + } }() } else if diskTarget == nil && idx.graph.NodeCount() == 0 && idx.graph.EdgeCount() == 0 { if _, isBulk := idx.graph.(graph.BulkLoader); isBulk && firstIndex && (!belowShadowMax || !belowShadowBytes || !shadowTaken) { @@ -3831,7 +3510,7 @@ func (idx *Indexer) indexCtxRaw(ctx context.Context, root string) (result *Index zap.Int("chunk_size", chunkSize)) for chunkStart := 0; chunkStart < len(files); chunkStart += chunkSize { chunkEnd := min(chunkStart+chunkSize, len(files)) - chunkShadow := graph.New() + chunkShadow := idx.newStructuralIntegrityShadow(streamingDisk, graph.StructuralPathShadowStreaming) idx.graph = chunkShadow parseChunk(files[chunkStart:chunkEnd]) if err := ctx.Err(); err != nil { @@ -4126,8 +3805,20 @@ func (idx *Indexer) indexCtxRaw(ctx context.Context, root string) (result *Index } reporter.Report("building search index", 0, 0) - // Build search index. - idx.buildSearchIndex() + // Prepare embeddings exactly once. Cold-shadow publication is deferred until + // the graph has drained and ownership validation can see durable nodes; + // direct and streaming paths already point at the durable store here. + vectorPlan, vectorErr := idx.prepareSearchIndexForPublication(ctx) + if vectorErr != nil { + return nil, vectorErr + } + if shadowTaken { + deferredVectorPlan = vectorPlan + } else if vectorPlan != nil { + if err := idx.installVectorPlan(ctx, idx.graph, vectorPlan); err != nil { + return nil, fmt.Errorf("indexer: publish vector corpus: %w", err) + } + } if !idx.deferResolve.Load() { // Contracts were already extracted inline during parse (per file, @@ -4184,7 +3875,7 @@ func (idx *Indexer) indexCtxRaw(ctx context.Context, root string) (result *Index // Framework dynamic-dispatch synthesis — runs once the call // graph and interface inference are final. Skipped under // deferGlobalPasses; the batch caller folds it into - // RunGlobalGraphPasses. + // shared multi-repository global-pass pipeline. reporter.Report("framework dispatch synthesis", 0, 0) if rep := resolver.RunFrameworkSynthesizers(idx.graph); rep.Total > 0 { idx.logger.Info("framework dispatch calls synthesized", @@ -4220,40 +3911,6 @@ func (idx *Indexer) indexCtxRaw(ctx context.Context, root string) (result *Index } } - // Auto-upgrade to Bleve if above threshold. Run in the background - // so the foreground IndexCtx returns immediately — populating - // Bleve with 50k+ symbols takes 30-60s and adding that to the - // initial-index latency was the dominant tail. Searches against - // idx.search keep hitting the in-memory backend until the swap - // completes; nothing observes a half-built Bleve. - // - // upgradeOnce gates the spawn so multi-repo warmup, which calls - // IndexCtx once per tracked repo, doesn't launch one upgrade - // goroutine per post-threshold repo. One per indexer lifetime. - // - // Skip the upgrade when the active search backend is the - // SymbolSearcher adapter: the disk store's native FTS is - // already serving search at engine-native latency, and - // spawning a parallel Bleve build would (a) waste ~100MB heap - // re-indexing the same corpus and (b) silently swap the - // adapter out for Bleve on completion — defeating the whole - // FTS path. The Swappable's current backend tells us which - // branch we're on. - if !isSymbolSearcherBackend(idx.search) && idx.search.Count() >= search.AutoThreshold { - idx.upgradeOnce.Do(func() { - reporter.Report("scheduling search backend upgrade", 0, 0) - idx.upgradeSpawnedMu.Lock() - idx.upgradeSpawned++ - idx.upgradeSpawnedMu.Unlock() - // Snapshot upfront so the background goroutine doesn't - // read Node.Meta concurrently with subsequent Index - // calls' Meta-writing passes (reach.BuildIndex, - // ResolveTemporalCalls, ...). - snapshot := idx.snapshotBleveEntries() - go idx.upgradeSearchToBleve(snapshot) - }) - } - // Persist the parser quarantine so a file that crashed the parser // stays skipped across daemon restarts until its content changes. if quarantine != nil { @@ -4323,13 +3980,31 @@ func (idx *Indexer) repoNodeEdgeCount() (int, int) { // incremental pipeline after ChangedSinceMtimes has already proved the tree // unchanged. It preserves the one necessary side effect for non-persistent // search backends without repeating filesystem discovery. -func (idx *Indexer) cleanCensusResult(detected int, started time.Time) *IndexResult { +func (idx *Indexer) cleanCensusResult(ctx context.Context, detected int, started time.Time) (*IndexResult, error) { if idx.totalDetected == 0 { idx.totalDetected = detected } - if !isSymbolSearcherBackend(idx.search) { - idx.buildSearchIndex() + + // A populated durable corpus can be republished from cheap statistics with + // no paid embedding pass. An empty corpus (including the v10 migration that + // deliberately discarded legacy vectors) triggers one vector-only rebuild + // from the already-persisted graph. + restored, restoreErr := idx.restoreDurableVectorBackend(ctx, idx.graph) + if restoreErr != nil { + if ctxErr := ctx.Err(); ctxErr != nil { + return nil, ctxErr + } + idx.lastVectorBuildErr = restoreErr + idx.logger.Warn("restore durable vector corpus failed; rebuilding", zap.Error(restoreErr)) + } + if restored { + idx.rebuildTextSearchIndex() + } else if !isSymbolSearcherBackend(idx.search) || idx.embedder != nil { + if err := idx.buildSearchIndexCtx(ctx); err != nil { + return nil, err + } } + nodes, edges := idx.repoNodeEdgeCount() fileCount := idx.trackedFileCount() if fileCount == 0 { @@ -4342,7 +4017,7 @@ func (idx *Indexer) cleanCensusResult(detected int, started time.Time) *IndexRes DurationMs: time.Since(started).Milliseconds(), } idx.warnIfEdgeSanityViolated(result) - return result + return result, nil } // warnIfEdgeSanityViolated logs a loud warning when an index pass @@ -5281,11 +4956,10 @@ func (idx *Indexer) restubIncomingRefs(graphPath string) { // embeddingDimsOrDefault returns the embedder's reported vector width, // falling back to a neutral placeholder only when the provider cannot // state its width yet (Dimensions() == 0, the APIProvider-before-first- -// call case). The fallback is never persisted: buildSearchIndex and -// ImportVectorIndex both overwrite it with the true width taken from a -// real vector / the cached header. Kept as a named helper so the -// vector-dimension default has one definition instead of a scattered -// magic number. +// call case). The fallback is never persisted: buildSearchIndex +// overwrites it with the true width taken from a real vector. Kept as a +// named helper so the vector-dimension default has one definition +// instead of a scattered magic number. func embeddingDimsOrDefault(p embedding.Provider) int { if p == nil { return 0 @@ -5466,10 +5140,14 @@ type embedChunkBatch struct { // embedFn already layers the deadline-halving retry on top of each // batch. func (idx *Indexer) embedAllChunks( + parent context.Context, texts []string, batchSize int, embedFn func(ctx context.Context, items []string) ([][]float32, error), ) ([][]float32, error) { + if parent == nil { + parent = context.Background() + } if len(texts) == 0 { return nil, nil } @@ -5507,10 +5185,13 @@ func (idx *Indexer) embedAllChunks( } if !apiBacked || concurrency <= 1 { - // Serial path — unchanged behaviour for in-process embedders. - ctx := context.Background() + // Serial path — unchanged behaviour for in-process embedders, now + // honoring cancellation from the owning index operation. for _, b := range batches { - vecs, err := embedFn(ctx, b.texts) + if err := parent.Err(); err != nil { + return nil, err + } + vecs, err := embedFn(parent, b.texts) if err != nil { return nil, err } @@ -5523,7 +5204,7 @@ func (idx *Indexer) embedAllChunks( // A cancellable group context means the first failure stops every // in-flight worker; the indexer's existing per-batch retry still // runs underneath embedFn. - ctx, cancel := context.WithCancel(context.Background()) + ctx, cancel := context.WithCancel(parent) defer cancel() jobs := make(chan embedChunkBatch) @@ -5574,6 +5255,12 @@ func (idx *Indexer) embedAllChunks( close(jobs) wg.Wait() + // A provider is expected to honor ctx, but cancellation still belongs to + // the owning index operation even if a provider returns successfully after + // its context was cancelled. Never publish a partial/late vector batch. + if err := parent.Err(); err != nil { + return nil, err + } if firstErr != nil { return nil, firstErr } @@ -5594,316 +5281,14 @@ func flattenEmbedResults(results [][][]float32) [][]float32 { return out } -// buildSearchIndex populates the search backend from the current graph. -// When an embedder is set, also builds a vector index and wraps both -// in a HybridBackend with RRF fusion. -// -// In multi-repo mode the search backend is shared across every repo -// (Indexer.search is wired to MultiIndexer.search at construction). -// Re-reading every graph node would mean each freshly-tracked repo pays an -// O(workspace) re-index pass over all -// previously-tracked repos' nodes — quadratic in repo count and the -// dominant cost of warming up a 260-repo workspace. So when this -// indexer carries a non-empty repoPrefix we walk only that repo's -// byRepo bucket; the other repos' entries are already in the shared -// backend from when they were tracked. Single-repo mode uses the store's -// non-content projection with an empty repo namespace. +// buildSearchIndex is the compatibility entry point for direct and focused +// callers. Full indexing uses the context-aware preparation/publication split +// so a cold shadow can defer publication until its durable drain succeeds. func (idx *Indexer) buildSearchIndex() { - // Start every build from a clean vector-build error: the degraded vector - // paths below set it, and a successful build (or a benign skip / no embedder) - // leaves it nil, so LastVectorBuildError always reflects the current pass. - idx.lastVectorBuildErr = nil - - // Install the learned sub-word boundary table before populating an - // in-process BM25 backend. Capability detection happens before graph - // enumeration, so native SQLite FTS and Bleve do no boundary census. - search.BuildAndInstallNgramBoundaries(idx.search, idx.graph) - - nativeText := isSymbolSearcherBackend(idx.search) - buildVectors := idx.embedder != nil && !idx.skipVectorBuild - if nativeText && !buildVectors { - // SQLite maintains symbol_fts in the graph mutation path. Backend.Add - // only adjusts a process-local approximate counter, so walking and - // decoding every node here cannot add searchable data. - return - } - - // Code-only enumeration: content (data_class=content) sections live in - // the content index, never the symbol search or the vector store, so the - // FTS loop below and collectEmbedTexts both skip them anyway. Fetching - // the non-content set up front means a content-heavy repo's hundreds of - // thousands of sections never enter memory here (the disk backend filters - // them in SQL), instead of being materialised only to be skipped. - nodes := graph.RepoCodeNodes(idx.graph, idx.repoPrefix) - - // Build the text index only for in-process backends. Native SQLite FTS is - // already updated transactionally by graph mutations; its Add method is a - // counting compatibility shim rather than an indexing operation. - if !nativeText { - for _, n := range nodes { - if !idx.shouldIndexForSearch(n) { - continue - } - idx.search.Add(n.ID, searchIndexFields(n, idx.projectName)...) - } - } - - // With no requested vector build, text indexing is complete. This covers - // both a missing embedder and snapshot warmup's skipVectorBuild path. - if !buildVectors { - return - } - - // Provisional dimensionality: trust the embedder's own report. - // A provider that can't state its width yet (an APIProvider before - // its first call returns 0) gets a neutral placeholder — the value - // is overwritten below from the first real vector, so it never - // reaches the persisted index. The old hard-coded 300 was wrong for - // the default static GloVe provider (50d) and misrepresented the - // index width in the interim; deriving from Dimensions() keeps it - // honest for every provider. - dims := embeddingDimsOrDefault(idx.embedder) - - // Collect texts and IDs for batch embedding. Nodes matching - // Semantic.SkipEmbed (e.g. CSS custom properties, terraform blocks, - // YAML/TOML/shell config vars) are kept in the text index but - // excluded from the vector index — embedding them is pure cost - // with no semantic payoff and on big monorepos dominates RAM. - // - // A symbol whose source span exceeds the chunk threshold is split - // into AST windows: each window is embedded as its own vector under - // a synthetic ID ("#chunkK"), and chunkMap records the - // chunk → parent mapping so query-time de-chunking maps a chunk hit - // back to the symbol. A small symbol stays a single metadata-only - // vector under its own ID. chunkMap is empty when nothing was split. - texts, ids, chunkMap, skipped := idx.collectEmbedTexts(nodes) - if skipped > 0 { - idx.logger.Info("skipped embedding for low-value nodes", - zap.Int("count", skipped), - zap.Int("embedded", len(texts))) - } - - if len(texts) == 0 { - return - } - - // Embedding scaling guards. Hard-cap the vector index for repos - // big enough that the cost no longer pays off — BM25 alone is a - // fine fallback and an OOM during initial index is much worse than - // missing the semantic boost. Chunk the EmbedBatch calls so any - // single API request stays small (matters for hosted embedders - // with per-request token limits). - // - // embedChunkTimeout is generous because ONNX inference (Hugot) has - // long tail latency: a 60s budget made one in ~30 chunks miss its - // deadline, which under the old fail-fast policy threw away every - // already-embedded chunk and silently degraded to BM25 with no - // signal to the user. 5 minutes covers observed worst-case spikes - // without changing steady-state behaviour. On a true hang the - // caller can still cancel the parent indexing call. - const ( - defaultEmbedMaxSymbols = 100_000 - embedChunkSize = 500 - embedChunkTimeout = 5 * time.Minute - ) - - // The cap is over the embeddable-text count, which with AST - // sub-chunking can exceed the symbol count. embedding.max_symbols - // overrides the built-in default for users with memory headroom. - embedMaxSymbols := defaultEmbedMaxSymbols - if idx.embedMaxSymbols > 0 { - embedMaxSymbols = idx.embedMaxSymbols - } - // Env override wins over both the default and the config-wired cap — - // a reliable knob independent of the config plumbing. Lets an operator - // lift the vector-index size guard for a large repo without editing - // (or debugging) the layered config. - if env := os.Getenv("GORTEX_EMBEDDINGS_MAX_SYMBOLS"); env != "" { - if n, err := strconv.Atoi(strings.TrimSpace(env)); err == nil && n > 0 { - embedMaxSymbols = n - } - } - if len(texts) > embedMaxSymbols { - idx.logger.Warn("vector index disabled — embedding text count exceeds threshold", - zap.Int("texts", len(texts)), - zap.Int("threshold", embedMaxSymbols), - zap.String("hint", "BM25 text search remains active; raise embedding.max_symbols if you have memory headroom")) - idx.lastVectorBuildErr = fmt.Errorf("embedding text count %d exceeds threshold %d (raise embedding.max_symbols)", len(texts), embedMaxSymbols) - return - } - - // embedWithRetry runs one chunk under ctx; on a context-deadline - // failure it splits the chunk in half and retries each half once. - // A single slow batch shouldn't throw away every already-embedded - // chunk and silently demote the backend to BM25. ctx is the group - // context, so once one chunk fails everywhere the in-flight retries - // here see the cancellation and stop too. - var embedWithRetry func(ctx context.Context, items []string) ([][]float32, error) - embedWithRetry = func(ctx context.Context, items []string) ([][]float32, error) { - chunkCtx, cancel := context.WithTimeout(ctx, embedChunkTimeout) - out, err := idx.embedder.EmbedBatch(chunkCtx, items) - cancel() - if err == nil { - return out, nil - } - // Only retry on deadline-style failures; auth/protocol errors - // won't get better with smaller batches. A cancellation from - // the group context (a sibling chunk already failed) is not a - // retry case either. - if ctx.Err() != nil || !errors.Is(err, context.DeadlineExceeded) || len(items) <= 1 { - return nil, err - } - idx.logger.Warn("embed chunk timed out, retrying with halved batch", - zap.Int("size", len(items)), - zap.Error(err)) - mid := len(items) / 2 - left, lerr := embedWithRetry(ctx, items[:mid]) - if lerr != nil { - return nil, lerr - } - right, rerr := embedWithRetry(ctx, items[mid:]) - if rerr != nil { - return nil, rerr - } - return append(left, right...), nil - } - - // Embed every chunk. For an API-backed embedder the chunks are run - // through a bounded worker pool (a hosted round-trip dominates - // indexing time, so overlapping requests is a real win); local - // in-process backends serialise on an inference mutex, so they keep - // the simple serial path. Either way the abort-on-any-error - // contract holds — one chunk failure means no vector index ships. - vectors, err := idx.embedAllChunks(texts, embedChunkSize, embedWithRetry) - if err != nil { - // A partial vector index would mis-score later queries (some - // symbols semantically findable, others not) — bail to - // text-only search rather than ship an inconsistent hybrid - // backend. - idx.logger.Warn("vector index aborted on chunk failure", zap.Error(err)) - idx.lastVectorBuildErr = fmt.Errorf("chunk embedding failed: %w", err) - return - } - - // Detect actual dimensions from first vector. - if len(vectors) > 0 && len(vectors[0]) > 0 { - dims = len(vectors[0]) - } - - vecBackend := search.NewVector(dims) - // VectorSearcher capability bridging: if the underlying store - // has a native HNSW, install it as the in-process backend's - // delegate — Add becomes a no-op, Search forwards to the - // engine, and we don't allocate `dim × 4 × N` bytes of heap - // for a parallel in-process HNSW. The indexer still drives - // the writes (BulkUpsertEmbeddings below) so the engine - // index lands with the same corpus the in-process one would - // have built. - vecSearcher, _ := idx.graph.(graph.VectorSearcher) - if vecSearcher != nil { - vecBackend.SetDelegate(&vectorSearcherDelegate{s: vecSearcher}) - } - // persistSink is where the embedded vectors are written to the backend - // so they survive a restart. Normally it is the active graph (a disk - // store). Under the bulk loader idx.graph is the in-memory shadow, which - // does not implement VectorSearcher; fall back to the disk store captured - // at the shadow swap (bulkVectorSink) so the vector index still reaches - // the `vectors` table instead of living only in the in-process HNSW and - // vanishing on the next restart (which would force a paid re-embed). - persistSink := vecSearcher - if persistSink == nil { - persistSink = idx.bulkVectorSink - } - var backendItems []graph.VectorItem - if persistSink != nil { - backendItems = make([]graph.VectorItem, 0, len(vectors)) - } - // Add only well-formed vectors. A nil or wrong-width vector would poison - // the index — a mis-scored or panicking query — so drop it, count it, and - // keep a sample of offending node IDs for the log. - var droppedVectors int - var droppedSample []string - for i, vec := range vectors { - if len(vec) != dims { - droppedVectors++ - if len(droppedSample) < 5 { - droppedSample = append(droppedSample, ids[i]) - } - continue - } - vecBackend.Add(ids[i], vec) - if persistSink != nil { - backendItems = append(backendItems, graph.VectorItem{ - NodeID: ids[i], - Vec: vec, - }) - } - } - // If every vector was invalid there is nothing to search on. Ship text-only - // rather than a silently empty vector index that mis-scores every query — - // the same all-or-nothing contract as the chunk-failure abort above. - if len(vectors) > 0 && vecBackend.Count() == 0 { - idx.logger.Warn("vector index aborted — all embeddings invalid", - zap.Int("dropped", droppedVectors), - zap.Int("dimensions", dims), - zap.Strings("sample_ids", droppedSample)) - idx.lastVectorBuildErr = fmt.Errorf("all %d embedding vectors were invalid (want width %d)", droppedVectors, dims) - return - } - if persistSink != nil && len(backendItems) > 0 { - if err := persistSink.BulkUpsertEmbeddings(backendItems); err != nil { - idx.logger.Warn("indexer: backend vector bulk upsert failed", - zap.Error(err)) - } else if err := persistSink.BuildVectorIndex(dims); err != nil { - idx.logger.Warn("indexer: backend vector index build failed", - zap.Error(err)) - } - } - // Install the chunk → parent-symbol mapping so HybridBackend can - // de-chunk vector hits back to symbols at query time. Empty when no - // symbol was large enough to split. - if len(chunkMap) > 0 { - vecBackend.SetChunkMap(chunkMap) + if err := idx.buildSearchIndexCtx(context.Background()); err != nil { + idx.lastVectorBuildErr = err + idx.logger.Warn("vector index build canceled", zap.Error(err)) } - - // Wrap text + vector into hybrid backend, swapping it in atomically - // so any concurrent searches keep seeing a coherent backend. - // - // Unwrap any existing HybridBackend to its text side before - // re-wrapping. Without this, buildSearchIndex called again (e.g. - // once per tracked repo during daemon warmup) would stack a fresh - // Hybrid on top of the previous one — nested Hybrids retain all - // their stale vector indexes, ballooning live memory by an order - // of magnitude. The text backend (BM25 or Bleve) has already been - // updated with every node via idx.search.Add above; a single - // Hybrid wrapping it + the latest vecBackend is all we need. - sw := idx.swappable() - inner := sw.Inner() - if hyb, ok := inner.(*search.HybridBackend); ok { - inner = hyb.TextBackend() - } - sw.Swap(search.NewHybrid(inner, vecBackend, idx.embedder)) - if droppedVectors > 0 { - idx.logger.Warn("indexer: dropped invalid embedding vectors", - zap.Int("dropped", droppedVectors), - zap.Int("dimensions", dims), - zap.Strings("sample_ids", droppedSample)) - } - fields := []zap.Field{ - zap.Int("vectors", vecBackend.Count()), - zap.Int("chunk_vectors", len(chunkMap)), - zap.Int("dimensions", dims), - zap.Int("dropped", droppedVectors), - } - // Surface the actual token spend of a paid embedding pass when the - // backend reports usage (API providers do; in-process ones don't). - // Without this the cost of an embedding run is invisible after the fact. - if acc, ok := idx.embedder.(interface{ TokensUsed() int64 }); ok { - if tokens := acc.TokensUsed(); tokens > 0 { - fields = append(fields, zap.Int64("embed_tokens", tokens)) - } - } - idx.logger.Info("vector index built", fields...) } // dirIgnoreFiles are the per-directory ignore-file basenames honored by @@ -6196,15 +5581,6 @@ func (idx *Indexer) IncrementalReindexPaths(root string, paths []string) (*Index return idx.coordinateRepositoryReindex(context.Background(), canonical) } -// incrementalDiscoverPaths discovers and refreshes files beneath paths without -// treating absent tracked files as deletions. Watcher directory-create scans -// use this mode because their job is to recover creates that happened before a -// nested watch was attached. A concurrent file-delete event must remain the -// sole owner of eviction so it can publish the pre-delete symbol snapshot. -func (idx *Indexer) incrementalDiscoverPaths(root string, paths []string) (*IndexResult, error) { - return idx.incrementalReindexPaths(root, paths, false) -} - // incrementalPathMode keeps forced point semantics private to the caller that // owns an explicit filesystem receipt. Reconcile, storm, and discovery callers // retain their historical mtime/Merkle filtering. @@ -6299,17 +5675,6 @@ func (idx *Indexer) incrementalPathOwned(absPath string) bool { return ok } -func (idx *Indexer) incrementalReindexPaths( - root string, - paths []string, - detectDeletions bool, - markerBatches ...*reparsePendingEnrichmentBatch, -) (*IndexResult, error) { - return idx.incrementalReindexPathsMode(root, paths, incrementalPathMode{ - detectDeletions: detectDeletions, - }, markerBatches...) -} - func (idx *Indexer) incrementalReindexPathsMode( root string, paths []string, diff --git a/internal/indexer/lifecycle_acceptance_test.go b/internal/indexer/lifecycle_acceptance_test.go new file mode 100644 index 000000000..59911baca --- /dev/null +++ b/internal/indexer/lifecycle_acceptance_test.go @@ -0,0 +1,116 @@ +package indexer + +import ( + "context" + "errors" + "testing" + + "github.com/stretchr/testify/require" + "go.uber.org/zap" + + "github.com/zzet/gortex/internal/config" + "github.com/zzet/gortex/internal/graph" + "github.com/zzet/gortex/internal/parser" + "github.com/zzet/gortex/internal/search" +) + +type lifecycleFailingBulkStore struct { + graph.Store +} + +func (*lifecycleFailingBulkStore) BeginBulkLoad() {} + +func (*lifecycleFailingBulkStore) FlushBulk() error { + return errors.New("injected lifecycle bulk flush failure") +} + +func newLifecycleTestMultiIndexer(t *testing.T) *MultiIndexer { + t.Helper() + return NewMultiIndexer( + graph.New(), + newTestRegistry(), + search.NewBM25(), + newTestConfigManager(t), + zap.NewNop(), + ) +} + +func closeLifecycleTestMultiIndexer(t *testing.T, mi *MultiIndexer) { + t.Helper() + require.NoError(t, mi.Close(context.Background())) +} + +func TestLifecycleIndexRepoClosesReplacedIndexer(t *testing.T) { + t.Setenv(crashWorkerEnv, "1") + repo := setupRepoDir(t, "repo") + mi := newLifecycleTestMultiIndexer(t) + t.Cleanup(func() { closeLifecycleTestMultiIndexer(t, mi) }) + + _, err := mi.TrackRepo(config.RepoEntry{Path: repo, Name: "repo"}) + require.NoError(t, err) + old := mi.GetIndexer("repo") + require.NotNil(t, old) + pool, _ := old.sharedParsePool() + require.NotNil(t, pool, "replacement precondition requires a real crashpool") + + _, err = mi.IndexRepo("repo") + require.NoError(t, err) + require.NotSame(t, old, mi.GetIndexer("repo")) + require.Nil(t, old.parsePool, "replacement must release the old crashpool") + _, err = old.ExtractBuffer("go", "after.go", []byte("package after\n")) + require.ErrorIs(t, err, ErrIndexerClosed) +} + +func TestLifecycleFailedReplacementClosesCandidateIndexerAndPool(t *testing.T) { + t.Setenv(crashWorkerEnv, "1") + repo := setupRepoDir(t, "repo") + mi := newLifecycleTestMultiIndexer(t) + t.Cleanup(func() { closeLifecycleTestMultiIndexer(t, mi) }) + + _, err := mi.TrackRepo(config.RepoEntry{Path: repo, Name: "repo"}) + require.NoError(t, err) + old := mi.GetIndexer("repo") + require.NotNil(t, old) + mi.graph = &lifecycleFailingBulkStore{Store: mi.graph} + + var candidate *Indexer + mi.newIndexer = func(g graph.Store, reg *parser.Registry, cfg config.IndexConfig, logger *zap.Logger) *Indexer { + candidate = New(g, reg, cfg, logger) + candidate.SetRootPath(repo) + pool, _ := candidate.sharedParsePool() + require.NotNil(t, pool, "failure precondition requires a real crashpool") + return candidate + } + + _, err = mi.IndexRepo("repo") + require.Error(t, err) + require.NotNil(t, candidate) + require.Same(t, old, mi.GetIndexer("repo"), "failed replacement must preserve the live Indexer") + require.Nil(t, candidate.parsePool, "failed replacement must release the candidate crashpool") + _, err = candidate.ExtractBuffer("go", "after.go", []byte("package after\n")) + require.ErrorIs(t, err, ErrIndexerClosed) + + _, err = old.ExtractBuffer("go", "still-live.go", []byte("package live\n")) + require.NoError(t, err, "failed replacement must not close the live Indexer") +} + +func TestLifecycleUntrackClosesIndexerAndCrashPool(t *testing.T) { + t.Setenv(crashWorkerEnv, "1") + repo := setupRepoDir(t, "repo") + mi := newLifecycleTestMultiIndexer(t) + t.Cleanup(func() { closeLifecycleTestMultiIndexer(t, mi) }) + + _, err := mi.TrackRepo(config.RepoEntry{Path: repo, Name: "repo"}) + require.NoError(t, err) + owned := mi.GetIndexer("repo") + require.NotNil(t, owned) + pool, _ := owned.sharedParsePool() + require.NotNil(t, pool, "untrack precondition requires a real crashpool") + + nodesRemoved, _ := mi.UntrackRepo("repo") + require.Positive(t, nodesRemoved) + require.Nil(t, mi.GetIndexer("repo")) + require.Nil(t, owned.parsePool, "untrack must release the repository crashpool") + _, err = owned.ExtractBuffer("go", "after.go", []byte("package after\n")) + require.ErrorIs(t, err, ErrIndexerClosed) +} diff --git a/internal/indexer/multi.go b/internal/indexer/multi.go index 0d2ca7de0..cf6ce0ef2 100644 --- a/internal/indexer/multi.go +++ b/internal/indexer/multi.go @@ -1,7 +1,6 @@ package indexer import ( - "bytes" "context" "fmt" "os" @@ -63,13 +62,22 @@ type MultiIndexer struct { indexers map[string]*Indexer // repoPrefix → per-repo indexer configMgr *config.ConfigManager logger *zap.Logger - mu sync.RWMutex + // newIndexer is instance-local so lifecycle tests can observe a constructor + // failure without publishing the candidate Indexer. Production instances set + // it to New; every per-repository construction flows through this factory. + newIndexer func(graph.Store, *parser.Registry, config.IndexConfig, *zap.Logger) *Indexer + mu sync.RWMutex // repositoryMutations owns one stable mutation lane per repository prefix. // The slot survives Indexer replacement so an explicit re-index cannot race // an old watcher instance on a second lane. repositoryMutationMu sync.Mutex repositoryMutations map[string]*repositoryMutationCoordinator + // lifecycleClosed is guarded by repositoryMutationMu so closing admission + // is atomic with stable-lane creation. closeMu serializes idempotent teardown. + lifecycleClosed bool + lifecycleComplete bool + closeMu sync.Mutex // batchMutationGate makes a batch-mode transition atomic with respect to // complete repository mutation pipelines. Stable coordinators take the read @@ -119,7 +127,7 @@ type MultiIndexer struct { // deferGlobalPasses, when set, propagates SetDeferGlobalPasses(true) // to every per-repo Indexer constructed by this MultiIndexer. Batch // warmup callers flip it on around their loop and - // invoke RunGlobalGraphPasses once at the end so the O(global) walks + // invoke the shared global-pass pipeline once at the end so the O(global) walks // (InferImplements / InferOverrides / markTestSymbolsAndEmitEdges) // don't run R times against an R-repo graph. deferGlobalPasses bool @@ -129,13 +137,13 @@ type MultiIndexer struct { // parallel warmup path: per-repo ResolveAll / contract extract / // semantic enrich mutate the shared graph, so running them in // parallel across repos races. With this flag the parallel loop - // just parses; RunDeferredPassesAll runs the per-repo passes + // just parses; RunDeferredPassesAllResult runs the per-repo passes // serially after the loop. Independent of deferGlobalPasses — that // flag covers a separate (cheaper) set of O(global) walks. deferResolve bool // batchChangedPrefixes scopes the per-repo clone-detection and - // clone-index Rebuild passes in RunGlobalGraphPasses to the repos that + // clone-index Rebuild passes in the shared global-pass pipeline to the repos that // actually re-indexed in the current batch. nil — the default, and what // every one-off EndBatch caller leaves it as — means "run the clone // passes for every tracked repo", the prior whole-workspace behaviour. @@ -145,7 +153,7 @@ type MultiIndexer struct { // disk. Clone edges are per-repo (no cross-repo pair is ever formed), // so an unchanged repo's persisted edges stay valid; its in-memory // incremental clone index is reseeded lazily on its first later edit. - // Consumed and cleared by RunGlobalGraphPasses. Guarded by mi.mu. + // Consumed and cleared by the shared global-pass pipeline. Guarded by mi.mu. batchChangedPrefixes map[string]struct{} // batchCensusEligible is the daemon's one-shot full-coverage attestation // for the armed batch scope — see ArmBatchCensusEligible. @@ -153,7 +161,7 @@ type MultiIndexer struct { // resolverLSPHelper is the resolve-time LSP helper propagated // onto every per-repo Indexer and onto the global post-pass - // resolver in RunDeferredPassesAll. nil means no LSP hot-path + // resolver in RunDeferredPassesAllResult. nil means no LSP hot-path // (heuristic-only resolution, the pre-N5 behaviour). The // daemon installs the helper via SetResolverLSPHelper after // constructing the LSP router; a multi-repo composite helper @@ -168,16 +176,6 @@ type MultiIndexer struct { // participate in the N5 hot path without daemon restart. onRepoTracked func(prefix, absPath string) - // skipVectorBuild, when set, propagates SetSkipVectorBuild(true) to - // every per-repo Indexer this MultiIndexer constructs, so their - // buildSearchIndex passes populate only the text index and never - // embed. The daemon flips it on for the warmup loop when a snapshot - // already carries the workspace vector index — re-embedding 30k+ - // symbols only to overwrite them with the cached index is the - // dominant restart cost. After warmup it restores the cached index - // once via ImportVectorIndex and clears the flag. - skipVectorBuild bool - // embedChunkOpts is the AST sub-chunking configuration propagated // to every per-repo Indexer this MultiIndexer constructs. The zero // value leaves the chunker on its built-in defaults. @@ -323,7 +321,13 @@ func (mi *MultiIndexer) newPerRepoIndexerGuardedWithMode( cfg config.IndexConfig, mode multiIndexerBatchMode, ) *Indexer { - idx := New(mi.graph, mi.registry, cfg, mi.logger) + factory := mi.newIndexer + if factory == nil { + // Preserve hand-built zero-value MultiIndexer fixtures while keeping + // every production instance on the constructor installed above. + factory = New + } + idx := factory(mi.graph, mi.registry, cfg, mi.logger) idx.shadowAdmission = mi.shadowAdmission idx.parseAdmission.Store(mi.parseAdmission.Load()) idx.nativeParseAdmission.Store(mi.nativeParseAdmission.Load()) @@ -333,7 +337,6 @@ func (mi *MultiIndexer) newPerRepoIndexerGuardedWithMode( idx.SetEmbedder(mi.embedder) } applyMultiIndexerBatchMode(idx, mode) - idx.SetSkipVectorBuild(mi.skipVectorBuild) idx.SetEmbeddingChunkOptions(mi.embedChunkOpts) idx.SetEmbeddingMaxSymbols(mi.embedMaxSymbols) idx.SetEmbeddingAPIConcurrency(mi.embedAPIConcurrency) @@ -416,29 +419,9 @@ func (mi *MultiIndexer) SetSemanticManager(m *semantic.Manager) { } } -// SetSkipVectorBuild controls whether per-repo Indexers constructed -// from now on skip the embedding pass in buildSearchIndex (text index -// only). The daemon enables it for the warmup loop when a snapshot -// already carries the workspace vector index, then disables it and -// restores the cached index once warmup finishes. It also re-applies -// the flag to every per-repo Indexer already constructed so a flag -// flip mid-lifecycle takes effect everywhere. -func (mi *MultiIndexer) SetSkipVectorBuild(skip bool) { - mi.mu.Lock() - mi.skipVectorBuild = skip - live := make([]*Indexer, 0, len(mi.indexers)) - for _, idx := range mi.indexers { - live = append(live, idx) - } - mi.mu.Unlock() - for _, idx := range live { - idx.SetSkipVectorBuild(skip) - } -} - // SetResolverLSPHelper installs the resolve-time LSP helper used by // every per-repo Indexer this MultiIndexer constructs from now on, -// and by the global post-pass resolver in RunDeferredPassesAll. Pass +// and by the global post-pass resolver in RunDeferredPassesAllResult. Pass // nil to detach. Safe to call zero or one times; subsequent calls // silently replace and propagate to every existing per-repo indexer. func (mi *MultiIndexer) SetResolverLSPHelper(h resolver.LSPHelper) { @@ -491,7 +474,7 @@ func (mi *MultiIndexer) BeginBatch() { // indexing loop across goroutines (warmup) — the parallel parsers // must not race each other inside ResolveAll / contract extract / // semantic enrich, which all mutate the shared graph. Pair with -// EndBatch; call RunDeferredPassesAll between the parallel parse and +// EndBatch; call RunDeferredPassesAllResult between the parallel parse and // EndBatch to run the deferred per-repo passes serially. func (mi *MultiIndexer) BeginParallelBatch() { mi.batchMutationGate.Lock() @@ -507,26 +490,10 @@ func (mi *MultiIndexer) BeginParallelBatch() { } } -// RunDeferredPassesAll drains the deferred per-repo passes (semantic -// enrich / contract extract+commit) serially across the indexers the -// parallel parse populated. Pairs with BeginParallelBatch: the parallel -// loop parses with deferResolve on; this serial loop runs the passes that -// would otherwise race on the shared graph. The references-completeness -// resolve runs ahead of this in RunPreEnrichResolve (so the daemon can mark -// itself queryable before enrichment); the per-repo resolver pass is -// suppressed here because resolver.ResolveAll walks the entire shared graph -// — paying it R times is O(R · E). One master resolver.New(graph).ResolveAll -// runs at the end to lift the placeholder edges enrichment + contracts added. -// -// Returns the number of repos whose deferred semantic enrichment was -// actually dispatched (pendingEnrich set, or forced via -// GORTEX_WARMUP_FORCE_ENRICH) rather than skipped as unchanged. Sampled -// before runDeferredEnrichParallel runs, since a successful non-partial -// pass clears the flag it reads. // SeedPendingEnrichAll re-arms the deferred-enrichment gate for every tracked // repo whose persisted enrichment is known-incomplete at its current clean HEAD // (see Indexer.MaybeSeedPendingEnrich). The daemon warmup calls it after the -// parallel parse and before RunDeferredPassesAll so a repo left partial or +// parallel parse and before RunDeferredPassesAllResult so a repo left partial or // abandoned by a prior process resumes even when no file changed this run. // Returns the number of repos that will enrich (already pending plus newly // seeded) — the caller uses a non-zero count to run the deferred passes on a @@ -576,13 +543,9 @@ func (mi *MultiIndexer) GraphMutationRevision() (uint64, bool) { return revisioner.MutationRevision(), true } -func (mi *MultiIndexer) RunDeferredPassesAll(ctx context.Context) int { - return mi.RunDeferredPassesAllResult(ctx).EnrichScheduled -} - // RunDeferredPassesAllResult is the result-bearing orchestration form used by // lifecycle callers that must decide whether a later global safety sweep is -// still required. RunDeferredPassesAll preserves the compatibility surface. +// still required. func (mi *MultiIndexer) RunDeferredPassesAllResult(ctx context.Context) DeferredPassesResult { return mi.BeginDeferredPasses(ctx, nil).FinishTailResult() } @@ -654,11 +617,11 @@ func (r *DeferredPassesRun) BeginApplyMutationReceipt() { // BeginDeferredPasses selects the repos with deferred work, prepares the // mutation-receipt window, materialises go.mod dependencies, and launches the -// enrichment pool on its own goroutine. The caller must call FinishTail (which +// enrichment pool on its own goroutine. The caller must call FinishTailResult (which // joins the pool) exactly once. applyGate, when non-nil, parks every // provider's graph-apply phase until the caller closes it — the caller MUST // first call BeginApplyMutationReceipt after its resolve phase, then close the -// gate, or FinishTail deadlocks. +// gate, or FinishTailResult deadlocks. // // Without an apply gate the receipt opens here. With overlap, delaying it until // the apply boundary excludes resolver writes while still observing every @@ -744,11 +707,6 @@ func (mi *MultiIndexer) BeginDeferredPasses(_ context.Context, applyGate <-chan // Wait blocks until every enrichment lane has drained. func (r *DeferredPassesRun) Wait() { <-r.poolDone } -// FinishTail preserves the compatibility result used by existing callers. -func (r *DeferredPassesRun) FinishTail() int { - return r.FinishTailResult().EnrichScheduled -} - // FinishTailResult joins the enrichment pool, runs the contract passes, closes // the receipt window, and performs the deferred-mutation catch-up resolve. func (r *DeferredPassesRun) FinishTailResult() DeferredPassesResult { @@ -1113,15 +1071,6 @@ func (mi *MultiIndexer) RunDeferredGoModAll() { mi.runDeferredGoModAll() } -// runDeferredEnrichParallel runs each indexer's semantic enrichment in a -// bounded worker pool. Concurrency is capped so at most a few LSP servers -// background-index at once (the memory-sensitive part). The manager pins each -// repo's LSP provider in-use for the duration of its pass, so the router's -// LRU evictor never closes a provider another repo is still enriching against. -func (mi *MultiIndexer) runDeferredEnrichParallel(indexers []*Indexer) { - mi.runDeferredEnrichPool(indexers) -} - // runDeferredEnrichPool drains per-repo semantic enrichment through a // bounded worker pool with no inter-repo barriers. The old fixed batches // made every batch wait for its slowest member and gave each heavy-Go repo @@ -1409,7 +1358,7 @@ func enrichConcurrency(repos int) int { // the per-Indexer flag too so a subsequent one-off TrackRepoCtx call // runs the passes inline as expected. // ArmBatchScope records the prefixes of the repos that re-indexed in the -// batch about to be ended, so the next RunGlobalGraphPasses runs the +// batch about to be ended, so the next shared global-pass run executes the // per-repo clone-detection + clone-index Rebuild passes only for those // repos instead of for every tracked repo. An empty set, or scoped global // passes being disabled, leaves the scope nil (run all). Only the daemon @@ -1502,8 +1451,8 @@ func (mi *MultiIndexer) EndBatch() { // derivation passes. It is the warm-restart fast-path counterpart to // EndBatch: when the warmup reconcile loop observed zero changed files // across every repo, the persistent backend already holds every resolved -// and derived edge from the prior run, so RunGlobalGraphPasses (plus the -// RunDeferredPassesAll / RunGlobalResolve the caller also skips) would +// and derived edge from the prior run, so the shared global-pass pipeline (plus +// RunDeferredPassesAllResult / RunGlobalResolve, which the caller also skips) would // only recompute what's already on disk — the work that turns a warm // restart into a 30s–500s stall. The per-Indexer SetDeferGlobalPasses // flag is still restored so a later watch-triggered TrackRepoCtx / @@ -1524,21 +1473,6 @@ func (mi *MultiIndexer) ResetBatch() { } } -// RunGlobalGraphPasses runs the graph-wide derivation passes once -// against the shared graph: InferImplements (structural interface -// satisfaction), InferOverrides (method-level overrides on -// extends/implements/composes parents), and markTestSymbolsAndEmitEdges -// (test→subject EdgeTests). Idempotent — graph.AddEdge dedupes by -// edgeKey and the resolver passes skip already-present parents. -func (mi *MultiIndexer) RunGlobalGraphPasses(ctx context.Context) { - // Direct callers own no transition gate. Exclude repository mutation - // pipelines for the complete unscoped derivation run; callers already under - // a batch gate use runGlobalGraphPasses directly. - mi.batchMutationGate.Lock() - defer mi.batchMutationGate.Unlock() - mi.runGlobalGraphPasses(ctx, nil, false) -} - // runGlobalGraphPasses owns the reachability topology writer for callers that // already hold the appropriate batch transition gate. Incremental pipelines // that already own topology call runGlobalGraphPassesTopologyHeld directly. @@ -1882,6 +1816,7 @@ func NewMultiIndexer( indexers: make(map[string]*Indexer), configMgr: cm, logger: logger, + newIndexer: New, shadowAdmission: processShadowAdmission, } } @@ -2115,14 +2050,20 @@ func (mi *MultiIndexer) indexMultiRepo(repos []config.RepoEntry) (map[string]*In // race against each other across goroutines on the shared // graph. They run serially below via RunDeferredPasses after // wg.Wait(). The graph-wide derivation passes run once after - // the loop via mi.RunGlobalGraphPasses(). + // the loop via the shared global-pass pipeline. idx.SetDeferResolve(true) result, err := idx.indexCtxRaw(context.Background(), r.absPath) if err != nil { + idx.Close() resultCh <- repoResult{prefix: r.prefix, err: fmt.Errorf("indexing %s: %w", r.absPath, err)} return } + if result == nil { + idx.Close() + resultCh <- repoResult{prefix: r.prefix, err: fmt.Errorf("indexing %s returned a nil result", r.absPath)} + return + } result.RepoPrefix = r.prefix meta := &RepoMetadata{ @@ -2165,12 +2106,19 @@ func (mi *MultiIndexer) indexMultiRepo(repos []config.RepoEntry) (map[string]*In completed = append(completed, rr) results[rr.prefix] = rr.result } + oldIndexers := make([]*Indexer, 0, len(completed)) mi.mu.Lock() for _, rr := range completed { + if old := mi.indexers[rr.prefix]; old != nil && old != rr.idx { + oldIndexers = append(oldIndexers, old) + } mi.repos[rr.prefix] = rr.meta mi.indexers[rr.prefix] = rr.idx } mi.mu.Unlock() + for _, old := range oldIndexers { + old.Close() + } if coordinatedBulkActive { if err := coordinatedBulk.EndCoordinatedBulkLoad(); err != nil { return nil, fmt.Errorf("multi-repo bulk-load finalize: %w", err) @@ -2266,6 +2214,7 @@ func (mi *MultiIndexer) IndexRepo(repoPrefix string) (*IndexResult, error) { func (mi *MultiIndexer) indexRepoRaw(repoPrefix string) (*IndexResult, error) { mi.mu.RLock() meta, ok := mi.repos[repoPrefix] + oldIdx := mi.indexers[repoPrefix] mi.mu.RUnlock() if !ok { return nil, fmt.Errorf("repository not found: %s", repoPrefix) @@ -2279,6 +2228,12 @@ func (mi *MultiIndexer) indexRepoRaw(repoPrefix string) (*IndexResult, error) { mi.configMgr.LoadWorkspaceConfig(repoPrefix, meta.RootPath) cfg := mi.configMgr.GetRepoConfig(repoPrefix) idx := mi.newPerRepoIndexerGuarded(cfg.Index) + installed := false + defer func() { + if !installed { + idx.Close() + } + }() // Always stamp the repo prefix, even when this is the only tracked repo. // The multi-repo cold path (indexMultiRepo) already prefixes // unconditionally; gating the single-repo re-index on repo count left the @@ -2317,6 +2272,10 @@ func (mi *MultiIndexer) indexRepoRaw(repoPrefix string) (*IndexResult, error) { } mi.indexers[repoPrefix] = idx mi.mu.Unlock() + installed = true + if oldIdx != nil && oldIdx != idx { + oldIdx.Close() + } // TODO: After re-indexing, run CrossRepoResolver.ResolveForRepo(repoPrefix) // to update cross-repo edges. This will be implemented in Task 7.1. @@ -2767,6 +2726,7 @@ func (mi *MultiIndexer) TrackRepoCtx(ctx context.Context, entry config.RepoEntry idx.SetProjectID(resolveProjectID(&entryCopy, cfg, prefix)) var result *IndexResult + installed := false err = mi.coordinateRepositoryTopologyMutation(ctx, idx, func() error { // Construction can precede a queued batch transition. Once the stable // lane and transition generation are held, reapply the authoritative mode. @@ -2821,6 +2781,7 @@ func (mi *MultiIndexer) TrackRepoCtx(ctx context.Context, entry config.RepoEntry } mi.indexers[prefix] = idx mi.mu.Unlock() + installed = true // Add to global config. entry.Path = absPath @@ -2838,6 +2799,9 @@ func (mi *MultiIndexer) TrackRepoCtx(ctx context.Context, entry config.RepoEntry } return nil }) + if !installed { + idx.Close() + } if err != nil { return nil, err } @@ -2912,6 +2876,7 @@ func (mi *MultiIndexer) ReconcileRepoCtx(ctx context.Context, entry config.RepoE idx.SetFileMtimes(priorMtimes) var result *IndexResult + installed := false err = mi.coordinateRepositoryTopologyMutation(ctx, idx, func() error { // Construction can precede a queued batch transition. Once the stable // lane and transition generation are held, reapply the authoritative mode. @@ -2990,7 +2955,7 @@ func (mi *MultiIndexer) ReconcileRepoCtx(ctx context.Context, entry config.RepoE result, err = fullRetrack() case churn == 0 && !idx.merkleEnabled(): route = "census_noop" - result = idx.cleanCensusResult(detected, start) + result, err = idx.cleanCensusResult(ctx, detected, start) case churn == 0: // The mtime census cannot prove a Merkle-enabled repository clean: // a missing baseline or extractor-salt change still requires the @@ -3045,6 +3010,7 @@ func (mi *MultiIndexer) ReconcileRepoCtx(ctx context.Context, entry config.RepoE } mi.indexers[prefix] = idx mi.mu.Unlock() + installed = true entry.Path = absPath if err := mi.configMgr.Global().AddRepo(entry); err != nil { @@ -3086,6 +3052,9 @@ func (mi *MultiIndexer) ReconcileRepoCtx(ctx context.Context, entry config.RepoE topologyChanged = incrementalTopologyChanged(result) return nil }) + if !installed { + idx.Close() + } if err != nil { return nil, err } @@ -3182,6 +3151,9 @@ func (mi *MultiIndexer) ReconcileAllCtx(ctx context.Context) map[string]*IndexRe // UntrackRepo evicts a repo from the graph and removes it from config. func (mi *MultiIndexer) UntrackRepo(repoPrefix string) (int, int) { + if mi.isClosed() { + return 0, 0 + } // Snapshot the exact live registry generation first. Legacy restores and // direct-map fixtures may not have a stable lane yet; backfill one only // while both metadata and Indexer pointers still match this generation. @@ -3235,6 +3207,12 @@ func (mi *MultiIndexer) UntrackRepo(repoPrefix string) (int, int) { delete(mi.indexers, repoPrefix) mi.mu.Unlock() + // The stable mutation lane is drained and this exact generation is detached; + // now wait for any overlay/direct extraction before terminating its workers. + if idx != nil { + idx.Close() + } + // The process-wide trigram budget otherwise retains the removed Indexer // (and its full-text cache) until an unrelated search happens to evict it. if idx != nil { @@ -3242,33 +3220,43 @@ func (mi *MultiIndexer) UntrackRepo(repoPrefix string) (int, int) { idx.trigramBudget().forget(idx) } - // Every repo's nodes live in its byRepo bucket, so the sidecar-aware - // purge covers all of them. Single-repo-mode nodes used to carry an - // empty RepoPrefix, never entered that bucket, and needed a - // file-by-file EvictFile loop — which took the branch BELOW the - // capability probe and so skipped PurgeRepo entirely, leaking fifteen - // repo_prefix-keyed sidecar tables on every solo untrack. + // Every repo's nodes live in its byRepo bucket. Serialize the complete + // sidecar/vector purge and aggregate vector publication with sibling repo + // installs; otherwise an older stats snapshot can be published after a newer + // corpus commit. The callback holds no mi.mu and releases every SQLite write + // transaction before ReplaceHybridVector waits for pinned search readers. var nodesRemoved, edgesRemoved int - if purger, ok := mi.graph.(interface{ PurgeRepo(string) error }); ok { - // Prefer the full sidecar-aware purge. EvictRepo drops only - // nodes+edges and leaves fifteen repo_prefix-keyed sidecar tables - // (file_mtimes, *_enrichment, symbol_fts, content_fts, ...) behind, - // which accumulate across untrack/retrack cycles until they dominate - // a long-lived store. PurgeRepo clears them in one transaction. It - // returns no counts, so report the repo's last-index metadata as the - // removed estimate; fall back to EvictRepo (real counts) on error. - if err := purger.PurgeRepo(repoPrefix); err != nil { - mi.logger.Warn("purge repo failed; falling back to node/edge eviction", - zap.String("prefix", repoPrefix), zap.Error(err)) - nodesRemoved, edgesRemoved = mi.graph.EvictRepo(repoPrefix) - } else { - nodesRemoved, edgesRemoved = meta.NodeCount, meta.EdgeCount + purgeRepo := func() { + if purger, ok := mi.graph.(interface{ PurgeRepo(string) error }); ok { + // Prefer the full sidecar-aware purge. It returns no counts, so report + // the last-index metadata as the estimate; fall back to EvictRepo on + // error. The subsequent empty corpus replacement also cleans legacy + // synthetic chunk rows that are not graph node IDs. + if err := purger.PurgeRepo(repoPrefix); err != nil { + mi.logger.Warn("purge repo failed; falling back to node/edge eviction", + zap.String("prefix", repoPrefix), zap.Error(err)) + nodesRemoved, edgesRemoved = mi.graph.EvictRepo(repoPrefix) + } else { + nodesRemoved, edgesRemoved = meta.NodeCount, meta.EdgeCount + } + return } - } else { - // Backends without the purge capability (the in-memory store has no - // sidecars, so EvictRepo is already complete there). + // Backends without sidecars are complete after ordinary eviction. nodesRemoved, edgesRemoved = mi.graph.EvictRepo(repoPrefix) } + refresh := func(sw *search.Swappable) error { + purgeRepo() + return mi.publishVectorCorpusAfterRepoRemoval(context.Background(), repoPrefix, sw) + } + if sw, ok := mi.search.(*search.Swappable); ok { + if err := sw.SerializeVectorUpdate(func() error { return refresh(sw) }); err != nil { + mi.logger.Warn("refresh vector corpus after untrack failed", + zap.String("prefix", repoPrefix), zap.Error(err)) + } + } else if err := refresh(nil); err != nil { + mi.logger.Warn("remove vector corpus after untrack failed", + zap.String("prefix", repoPrefix), zap.Error(err)) + } // Remove from global config. if meta.RootPath != "" { @@ -4420,127 +4408,3 @@ func (mi *MultiIndexer) applyRemoteStitch(cr *resolver.CrossRepoResolver) { func (mi *MultiIndexer) Search() search.Backend { return mi.search } - -// ExportVectorIndex serializes the workspace-global semantic-search -// vector index — there is one shared HNSW index across every tracked -// repo, not one per repo. Returns nil, 0, 0 when no vector index is -// active (embeddings disabled, or the backend is still text-only). -// Used by the daemon snapshot path so a default-on daemon does not -// re-embed the whole graph on every restart. -func (mi *MultiIndexer) ExportVectorIndex() ([]byte, int, int) { - sw, ok := mi.search.(*search.Swappable) - if !ok { - return nil, 0, 0 - } - hybrid, ok := sw.Inner().(*search.HybridBackend) - if !ok { - return nil, 0, 0 - } - vec := hybrid.VectorIndex() - if vec == nil || vec.Count() == 0 { - return nil, 0, 0 - } - var buf bytes.Buffer - if err := vec.Save(&buf); err != nil { - mi.logger.Warn("failed to export vector index", zap.Error(err)) - return nil, 0, 0 - } - return buf.Bytes(), vec.Dims(), vec.Count() -} - -// ImportVectorIndex restores a previously-exported vector index into -// the shared search backend, wrapping the current text backend in a -// HybridBackend. It is a no-op when embeddings are disabled (no -// configured embedder) or when the cached index's dimensionality does -// not match the active embedder — a provider switch (GloVe 50d → ONNX -// 384d) makes the cached vectors meaningless, so the indexer re-embeds -// instead. Returns an error only on a structurally corrupt index blob. -func (mi *MultiIndexer) ImportVectorIndex(data []byte, dims, count int) error { - if len(data) == 0 || mi.embedder == nil { - return nil - } - if embedderDims := mi.embedder.Dimensions(); embedderDims > 0 && embedderDims != dims { - mi.logger.Info("vector index dims mismatch, will re-embed", - zap.Int("cached_dims", dims), zap.Int("embedder_dims", embedderDims)) - return nil - } - sw, ok := mi.search.(*search.Swappable) - if !ok { - return nil - } - vec := search.NewVector(dims) - if err := vec.LoadFrom(bytes.NewReader(data)); err != nil { - return fmt.Errorf("import vector index: %w", err) - } - vec.SetCount(count) - - // Unwrap an existing HybridBackend to its text side before - // re-wrapping so we never nest Hybrids (each retains a stale - // vector index — see buildSearchIndex for the memory rationale). - inner := sw.Inner() - if hyb, ok := inner.(*search.HybridBackend); ok { - inner = hyb.TextBackend() - } - sw.Swap(search.NewHybrid(inner, vec, mi.embedder)) - mi.logger.Info("restored vector index from snapshot", - zap.Int("vectors", count), zap.Int("dims", dims)) - return nil -} - -// AutoDetectRepos walks immediate subdirectories of parentPath looking for -// .git directories. If parentPath itself is a Git repo, it returns a single -// entry (the caller should index it as single-repo). If zero Git repos are -// found, it returns nil so the caller can fall back to single-repo mode. -// This is gated by the workspace.auto_detect config flag. -func (mi *MultiIndexer) AutoDetectRepos(parentPath string) []config.RepoEntry { - absPath, err := filepath.Abs(parentPath) - if err != nil { - mi.logger.Warn("auto-detect: failed to resolve path", zap.String("path", parentPath), zap.Error(err)) - return nil - } - - // If the path itself is a Git repo, return it as a single repo. - if isGitRepo(absPath) { - return []config.RepoEntry{{ - Path: absPath, - Name: filepath.Base(absPath), - }} - } - - // Walk immediate subdirectories (not recursive) for .git dirs. - entries, err := os.ReadDir(absPath) - if err != nil { - mi.logger.Warn("auto-detect: failed to read directory", zap.String("path", absPath), zap.Error(err)) - return nil - } - - var repos []config.RepoEntry - for _, entry := range entries { - if !entry.IsDir() { - continue - } - subDir := filepath.Join(absPath, entry.Name()) - if isGitRepo(subDir) { - repos = append(repos, config.RepoEntry{ - Path: subDir, - Name: entry.Name(), // Derive RepoPrefix from subdirectory name. - }) - } - } - - // If zero Git repos found, return nil — caller falls back to single-repo. - if len(repos) == 0 { - return nil - } - - return repos -} - -// isGitRepo checks whether the given directory contains a .git subdirectory. -func isGitRepo(dir string) bool { - info, err := os.Stat(filepath.Join(dir, ".git")) - if err != nil { - return false - } - return info.IsDir() -} diff --git a/internal/indexer/multi_close.go b/internal/indexer/multi_close.go new file mode 100644 index 000000000..821a456e2 --- /dev/null +++ b/internal/indexer/multi_close.go @@ -0,0 +1,81 @@ +package indexer + +import ( + "context" + "errors" +) + +var errMultiIndexerClosed = errors.New("multi-indexer is closed") + +func (mi *MultiIndexer) isClosed() bool { + if mi == nil { + return true + } + mi.repositoryMutationMu.Lock() + closed := mi.lifecycleClosed + mi.repositoryMutationMu.Unlock() + return closed +} + +// Close rejects new repository mutations, drains every stable mutation lane, +// detaches the owned Indexers under mi.mu, and closes each Indexer outside that +// lock. Calls are idempotent; a context-canceled call may be retried to finish +// draining and teardown. +func (mi *MultiIndexer) Close(ctx context.Context) error { + if mi == nil { + return nil + } + if ctx == nil { + ctx = context.Background() + } + + mi.closeMu.Lock() + defer mi.closeMu.Unlock() + if mi.lifecycleComplete { + return nil + } + + mi.repositoryMutationMu.Lock() + mi.lifecycleClosed = true + coordinators := make([]*repositoryMutationCoordinator, 0, len(mi.repositoryMutations)) + for _, coordinator := range mi.repositoryMutations { + if coordinator != nil { + coordinators = append(coordinators, coordinator) + } + } + mi.repositoryMutationMu.Unlock() + + // Close every admission boundary before waiting on any one lane. This + // prevents another repository from admitting work while Close is blocked on + // an earlier in-flight mutation. + for _, coordinator := range coordinators { + coordinator.closeAdmission() + } + for _, coordinator := range coordinators { + if err := coordinator.wait(ctx); err != nil { + return err + } + } + + mi.mu.Lock() + seen := make(map[*Indexer]struct{}, len(mi.indexers)) + indexers := make([]*Indexer, 0, len(mi.indexers)) + for _, idx := range mi.indexers { + if idx == nil { + continue + } + if _, exists := seen[idx]; exists { + continue + } + seen[idx] = struct{}{} + indexers = append(indexers, idx) + } + mi.indexers = make(map[string]*Indexer) + mi.mu.Unlock() + + for _, idx := range indexers { + idx.Close() + } + mi.lifecycleComplete = true + return nil +} diff --git a/internal/indexer/multi_cold_orchestration_test.go b/internal/indexer/multi_cold_orchestration_test.go index fac3be143..2ca5cfe9e 100644 --- a/internal/indexer/multi_cold_orchestration_test.go +++ b/internal/indexer/multi_cold_orchestration_test.go @@ -248,7 +248,7 @@ func TestRunDeferredPassesAllPoolsEnrichmentBounded(t *testing.T) { t.Fatalf("queue dropped repos: %d of %d", len(queue), len(indexers)) } - scheduled := mi.RunDeferredPassesAll(t.Context()) + scheduled := mi.RunDeferredPassesAllResult(t.Context()).EnrichScheduled require.Equal(t, 8, scheduled) assert.Equal(t, int32(8), provider.calls.Load()) // The pool bounds concurrent per-repo passes at enrichConcurrency @@ -262,7 +262,7 @@ func TestRunDeferredPassesAllFailureKeepsGenerationPending(t *testing.T) { provider := &coldBatchProvider{language: "go", fail: true} mi := deferredBatchFixture(t, "go", 1, provider) - require.Equal(t, 1, mi.RunDeferredPassesAll(t.Context())) + require.Equal(t, 1, mi.RunDeferredPassesAllResult(t.Context()).EnrichScheduled) idx := mi.indexers["batch-00"] require.NotNil(t, idx) assert.True(t, idx.pendingEnrich.Load(), @@ -297,7 +297,7 @@ func serialColdReference(t *testing.T, repos []config.RepoEntry) *MultiIndexer { mi.indexers[prefix].RunDeferredPasses(t.Context()) } mi.runCrossRepoResolve(true) - mi.RunGlobalGraphPasses(t.Context()) + mi.runGlobalGraphPasses(t.Context(), nil, false) return mi } @@ -476,7 +476,7 @@ func TestIndexMultiRepoRefFactFailureStopsGlobalConsumers(t *testing.T) { // TestBeginDeferredPassesOverlapSplitsPoolFromTail pins the overlap contract: // BeginDeferredPasses drains the enrichment pool on its own goroutine while // the caller is free to run the resolve phase; contracts and the catch-up -// resolve happen only in FinishTail, and the batch-only resolve gate is +// resolve happen only in FinishTailResult, and the batch-only resolve gate is // restored afterwards. func TestBeginDeferredPassesOverlapSplitsPoolFromTail(t *testing.T) { provider := &coldBatchProvider{language: "python", delay: 15 * time.Millisecond} @@ -485,13 +485,13 @@ func TestBeginDeferredPassesOverlapSplitsPoolFromTail(t *testing.T) { run := mi.BeginDeferredPasses(t.Context(), nil) run.Wait() assert.Equal(t, int32(4), provider.calls.Load(), - "the pool must drain without FinishTail being called") + "the pool must drain without FinishTailResult being called") for prefix, idx := range mi.indexers { assert.True(t, idx.skipResolveInDeferred, "%s must keep the batch-only resolve gate until the tail runs", prefix) } - scheduled := run.FinishTail() + scheduled := run.FinishTailResult().EnrichScheduled assert.Equal(t, 4, scheduled) for prefix, idx := range mi.indexers { assert.False(t, idx.skipResolveInDeferred, diff --git a/internal/indexer/multi_global_passes_test.go b/internal/indexer/multi_global_passes_test.go index d0b657077..c2869d845 100644 --- a/internal/indexer/multi_global_passes_test.go +++ b/internal/indexer/multi_global_passes_test.go @@ -113,10 +113,10 @@ func TestMultiIndexer_IndexAll_GlobalPassesProduceEdges(t *testing.T) { assert.GreaterOrEqual(t, testFuncs, 2, "is_test should be stamped on TestRunGreet in each repo") } -// TestMultiIndexer_RunGlobalGraphPasses_Idempotent verifies that running -// the global passes a second time does not mutate edge counts (graph -// dedup + resolver passes skip already-present edges). -func TestMultiIndexer_RunGlobalGraphPasses_Idempotent(t *testing.T) { +// TestMultiIndexer_GlobalGraphPassPipeline_Idempotent verifies that running +// the live global-pass pipeline a second time does not mutate edge counts +// (graph dedup + resolver passes skip already-present edges). +func TestMultiIndexer_GlobalGraphPassPipeline_Idempotent(t *testing.T) { repoA := setupRepoWithTestAndIface(t, "repo-a") repoB := setupRepoWithTestAndIface(t, "repo-b") @@ -145,8 +145,8 @@ func TestMultiIndexer_RunGlobalGraphPasses_Idempotent(t *testing.T) { require.Greater(t, testsBefore, 0) // Re-run the global passes. None of the three should add duplicates. - mi.RunGlobalGraphPasses(context.Background()) - mi.RunGlobalGraphPasses(context.Background()) + mi.runGlobalGraphPasses(context.Background(), nil, false) + mi.runGlobalGraphPasses(context.Background(), nil, false) assert.Equal(t, implsBefore, countEdges(g, graph.EdgeImplements), "InferImplements re-emission should be idempotent") diff --git a/internal/indexer/multi_test.go b/internal/indexer/multi_test.go index 37e363bbf..a9e9c09ec 100644 --- a/internal/indexer/multi_test.go +++ b/internal/indexer/multi_test.go @@ -819,8 +819,6 @@ guards: require.NoError(t, err) require.NotNil(t, cfg) - // New fields should have defaults. - assert.False(t, cfg.Multi.AutoDetect) // Existing fields should be loaded. assert.Equal(t, 4, cfg.Index.Workers) assert.True(t, cfg.Watch.Enabled) diff --git a/internal/indexer/native_parse_admission_test.go b/internal/indexer/native_parse_admission_test.go index 07bcdb8ba..182c36837 100644 --- a/internal/indexer/native_parse_admission_test.go +++ b/internal/indexer/native_parse_admission_test.go @@ -103,8 +103,8 @@ func TestGeneratedParserProjectionBypassesNativeAdmission(t *testing.T) { src := generatedParserAdmissionFixture() ctx, cancel := context.WithTimeout(context.Background(), 20*time.Millisecond) defer cancel() - result, skipped, err := idx.extractFileCtx( - ctx, admission, nil, nil, + result, skipped, err := idx.extractFileCtxWithRawLease( + ctx, admission, nil, nil, nil, "/tmp/parser.c", "parser.c", "c", ext, src, ) if err != nil { @@ -141,8 +141,8 @@ func TestTimedOutNativeExtractionRetainsAdmission(t *testing.T) { } outcome := make(chan extractionOutcome, 1) go func() { - result, skipped, err := idx.extractFileCtx( - context.Background(), admission, nil, nil, + result, skipped, err := idx.extractFileCtxWithRawLease( + context.Background(), admission, nil, nil, nil, "/tmp/blocked.c", "blocked.c", "c", blocking, []byte("int x;"), ) outcome <- extractionOutcome{result: result, skipped: skipped, err: err} @@ -161,8 +161,8 @@ func TestTimedOutNativeExtractionRetainsAdmission(t *testing.T) { second := &nativeAdmissionTestExtractor{} ctx, cancel := context.WithTimeout(context.Background(), 30*time.Millisecond) defer cancel() - _, _, err := idx.extractFileCtx( - ctx, admission, nil, nil, + _, _, err := idx.extractFileCtxWithRawLease( + ctx, admission, nil, nil, nil, "/tmp/second.c", "second.c", "c", second, []byte("int y;"), ) if !errors.Is(err, context.DeadlineExceeded) { @@ -180,8 +180,8 @@ func TestTimedOutNativeExtractionRetainsAdmission(t *testing.T) { } ctx, cancel = context.WithTimeout(context.Background(), time.Second) defer cancel() - _, skipped, err := idx.extractFileCtx( - ctx, admission, nil, nil, + _, skipped, err := idx.extractFileCtxWithRawLease( + ctx, admission, nil, nil, nil, "/tmp/third.c", "third.c", "c", second, []byte("int z;"), ) if err != nil || skipped { diff --git a/internal/indexer/reconcile_clean_census_test.go b/internal/indexer/reconcile_clean_census_test.go index 2b8d4217b..3b8b48f74 100644 --- a/internal/indexer/reconcile_clean_census_test.go +++ b/internal/indexer/reconcile_clean_census_test.go @@ -1,6 +1,7 @@ package indexer import ( + "context" "fmt" "os" "path/filepath" @@ -70,8 +71,8 @@ func TestCleanCensusResultBootstrapsNonPersistentSearch(t *testing.T) { idx.search = search.NewBM25() idx.SetFileMtimes(map[string]int64{"a.go": 1}) - result := idx.cleanCensusResult(1, time.Now()) - + result, err := idx.cleanCensusResult(t.Context(), 1, time.Now()) + require.NoError(t, err) require.NotNil(t, result) assert.Equal(t, 1, result.FileCount) assert.Equal(t, 1, result.NodeCount) @@ -80,6 +81,85 @@ func TestCleanCensusResultBootstrapsNonPersistentSearch(t *testing.T) { require.NotEmpty(t, idx.search.Search("Alpha", 10)) } +func TestCleanCensusRestoresDurableVectorsWithoutEmbedding(t *testing.T) { + root := vectorPersistFixture(t, 1) + store, err := store_sqlite.Open(filepath.Join(t.TempDir(), "store.sqlite")) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, store.Close()) }) + + seedEmbedder := &poolEmbedder{} + seed := newVectorPersistIndexer(t, store, seedEmbedder) + _, err = seed.IndexCtx(context.Background(), root) + require.NoError(t, err) + require.Greater(t, seedEmbedder.calls, int32(0)) + + restartEmbedder := &poolEmbedder{failOnText: "every call would fail"} + restarted := newVectorPersistIndexer(t, store, restartEmbedder) + result, err := restarted.cleanCensusResult(context.Background(), 1, time.Now()) + require.NoError(t, err) + require.NotNil(t, result) + require.Zero(t, restartEmbedder.calls, + "warm restoration must publish durable corpus statistics without a paid embedding pass") + assertDelegatedVectorPublication(t, restarted) +} + +func TestCleanCensusRebuildsMigrationClearedVectorCorpus(t *testing.T) { + root := vectorPersistFixture(t, 1) + store, err := store_sqlite.Open(filepath.Join(t.TempDir(), "store.sqlite")) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, store.Close()) }) + + seed := newVectorPersistIndexer(t, store, &poolEmbedder{}) + _, err = seed.IndexCtx(context.Background(), root) + require.NoError(t, err) + _, err = store.ReplaceVectorCorpus(context.Background(), "", 3, nil) + require.NoError(t, err) + + rebuildEmbedder := &poolEmbedder{} + restarted := newVectorPersistIndexer(t, store, rebuildEmbedder) + _, err = restarted.cleanCensusResult(context.Background(), 1, time.Now()) + require.NoError(t, err) + require.Greater(t, rebuildEmbedder.calls, int32(0), + "an empty durable sidecar with live graph nodes must be rebuilt once") + stats, err := store.VectorCorpusStatsForRepo(context.Background(), "", 3) + require.NoError(t, err) + require.Greater(t, stats.RepositoryVectorCount, 0) + assertDelegatedVectorPublication(t, restarted) +} + +func TestRestoreDurableVectorBackendRequiresCurrentRepositoryCorpus(t *testing.T) { + store, err := store_sqlite.Open(filepath.Join(t.TempDir(), "store.sqlite")) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, store.Close()) }) + store.AddBatch([]*graph.Node{ + {ID: "A/alpha", Kind: graph.KindFunction, Name: "alpha", FilePath: "A/a.go", RepoPrefix: "A"}, + {ID: "B/beta", Kind: graph.KindFunction, Name: "beta", FilePath: "B/b.go", RepoPrefix: "B"}, + }, nil) + _, err = store.ReplaceVectorCorpus(context.Background(), "B", 3, []graph.VectorCorpusItem{ + {NodeID: "B/beta", Vec: []float32{1, 0, 0}}, + }) + require.NoError(t, err) + + emb := &poolEmbedder{} + idx := newVectorPersistIndexer(t, store, emb) + idx.SetRepoPrefix("A") + restored, err := idx.restoreDurableVectorBackend(context.Background(), store) + require.NoError(t, err) + require.False(t, restored, + "another repository's vectors must not suppress this repository's rebuild") + require.Zero(t, emb.calls) + + _, err = store.ReplaceVectorCorpus(context.Background(), "A", 3, []graph.VectorCorpusItem{ + {NodeID: "A/alpha", Vec: []float32{0, 1, 0}}, + }) + require.NoError(t, err) + restored, err = idx.restoreDurableVectorBackend(context.Background(), store) + require.NoError(t, err) + require.True(t, restored) + require.Zero(t, emb.calls, "repository-scoped restore must never invoke embedding") + assertDelegatedVectorPublication(t, idx) +} + func TestReconcileRepoCtxUsesCleanCensusNoOp(t *testing.T) { root := t.TempDir() writeFile(t, filepath.Join(root, "main.go"), "package sample\nfunc Alpha() {}\n") @@ -140,7 +220,9 @@ func BenchmarkCleanReconcileDiscovery(b *testing.B) { if err != nil || len(changed) != 0 || len(deleted) != 0 { b.Fatalf("clean census: changed=%d deleted=%d err=%v", len(changed), len(deleted), err) } - idx.cleanCensusResult(detected, time.Now()) + if _, err := idx.cleanCensusResult(context.Background(), detected, time.Now()); err != nil { + b.Fatalf("clean result: %v", err) + } } }) diff --git a/internal/indexer/repository_mutation_coordinator.go b/internal/indexer/repository_mutation_coordinator.go index 0814987b8..3be528380 100644 --- a/internal/indexer/repository_mutation_coordinator.go +++ b/internal/indexer/repository_mutation_coordinator.go @@ -280,14 +280,16 @@ func (c *repositoryMutationCoordinator) runExclusiveMode( return fn() } -func (c *repositoryMutationCoordinator) closeAndWait(ctx context.Context) error { - if ctx == nil { - ctx = context.Background() - } +func (c *repositoryMutationCoordinator) closeAdmission() { c.mu.Lock() c.closed = true c.mu.Unlock() +} +func (c *repositoryMutationCoordinator) wait(ctx context.Context) error { + if ctx == nil { + ctx = context.Background() + } done := make(chan struct{}) go func() { c.work.Wait() @@ -301,6 +303,11 @@ func (c *repositoryMutationCoordinator) closeAndWait(ctx context.Context) error } } +func (c *repositoryMutationCoordinator) closeAndWait(ctx context.Context) error { + c.closeAdmission() + return c.wait(ctx) +} + type repositoryMutationCoordinatorStats struct { RequestedGeneration uint64 CompletedGeneration uint64 @@ -324,6 +331,13 @@ func (c *repositoryMutationCoordinator) stats() repositoryMutationCoordinatorSta func (mi *MultiIndexer) repositoryMutationCoordinator(repoPrefix string) *repositoryMutationCoordinator { mi.repositoryMutationMu.Lock() defer mi.repositoryMutationMu.Unlock() + if mi.lifecycleClosed { + coordinator := newRepositoryMutationCoordinator(func([]string) (*IndexResult, error) { + return nil, errMultiIndexerClosed + }) + coordinator.closeAdmission() + return coordinator + } if mi.repositoryMutations == nil { mi.repositoryMutations = make(map[string]*repositoryMutationCoordinator) } @@ -487,7 +501,7 @@ func (idx *Indexer) ensureRepositoryMutationRoot(root string) error { idx.repositoryMutationMu.Lock() defer idx.repositoryMutationMu.Unlock() if idx.rootPath == "" { - idx.rootPath = absRoot + idx.storeRootPath(absRoot) return nil } current, err := filepath.Abs(idx.rootPath) diff --git a/internal/indexer/scoped_global_passes_test.go b/internal/indexer/scoped_global_passes_test.go index 62486f25a..ae9679535 100644 --- a/internal/indexer/scoped_global_passes_test.go +++ b/internal/indexer/scoped_global_passes_test.go @@ -111,7 +111,7 @@ func TestCloneRepoNodes_ScopedNeverMaterialisesOtherRepo(t *testing.T) { cs := newIdxCountingStore(twoRepoFuncGraph()) // Detect for repoA: finalise + detect both walk cloneRepoNodes(repoA). - detectClonesAndEmitEdgesCtx(context.Background(), cs, "repoA", 0.8) + detectClonesAndEmitEdgesWithBaselineCtx(context.Background(), cs, "repoA", 0.8) // Incremental index rebuild for repoA reseeds from the same repo's nodes. ci := newIncrementalCloneIndex() ci.Rebuild(cs, "repoA") @@ -131,7 +131,7 @@ func TestCloneRepoNodes_ScopedNeverMaterialisesOtherRepo(t *testing.T) { // single-repo/shadow regime does not regress to a graph-wide snapshot. func TestCloneRepoNodes_EmptyPrefixUsesExactRepoProjection(t *testing.T) { cs := newIdxCountingStore(twoRepoFuncGraph()) - detectClonesAndEmitEdgesCtx(context.Background(), cs, "", 0.8) + detectClonesAndEmitEdgesWithBaselineCtx(context.Background(), cs, "", 0.8) if cs.allNodes != 0 { t.Errorf("empty-prefix clone detect must not call AllNodes(); got %d calls", cs.allNodes) } diff --git a/internal/indexer/skip_telemetry.go b/internal/indexer/skip_telemetry.go index 621a8bb59..43464bf6d 100644 --- a/internal/indexer/skip_telemetry.go +++ b/internal/indexer/skip_telemetry.go @@ -30,18 +30,23 @@ func (e *extractorPanicError) Error() string { return fmt.Sprintf("extractor panic on %s: %v", e.file, e.value) } -// safeExtract runs ext.Extract guarded by a recover so a panic on a -// single malformed file becomes an error instead of crashing the whole -// indexing run. This is the in-process last line of defence behind the -// subprocess crash-isolation pool (which only runs when enabled). -func safeExtract(ext parser.Extractor, relPath string, src []byte) (result *parser.ExtractionResult, err error) { +// safeExtractWithOptions runs the central extraction dispatcher guarded by a +// recover so a panic on one malformed file becomes an error instead of +// crashing the whole indexing run. The request options are immutable and +// repository-scoped. +func safeExtractWithOptions( + ext parser.Extractor, + relPath string, + src []byte, + opts parser.ExtractionOptions, +) (result *parser.ExtractionResult, err error) { defer func() { if rec := recover(); rec != nil { result = nil err = &extractorPanicError{file: relPath, value: rec, stack: debug.Stack()} } }() - return ext.Extract(relPath, src) + return parser.Extract(ext, relPath, src, opts) } // skippedFile records a file dropped by the size cap or a full-index @@ -100,15 +105,12 @@ func largeFileReadParallelism(workers int) int { return min(2, workers) } -// extractWithTimeout runs ext.Extract under the per-file extraction -// budget. With no budget configured it calls Extract directly. On -// timeout it returns errExtractTimeout; the slow extraction runs on to -// completion in its goroutine (tree-sitter's own 5s parse cap bounds -// the worst case) and its result is discarded. -func (idx *Indexer) extractWithTimeout(ext parser.Extractor, relPath string, src []byte) (*parser.ExtractionResult, error) { - return idx.extractWithTimeoutDone(ext, relPath, src, nil) -} - +// extractWithTimeoutDone runs ext.Extract under the per-file extraction +// budget and invokes done when extraction actually finishes. With no budget +// configured it calls Extract directly. On timeout it returns +// errExtractTimeout; the slow extraction runs on to completion in its goroutine +// (tree-sitter's own 5s parse cap bounds the worst case) and its result is +// discarded. func (idx *Indexer) extractWithTimeoutDone( ext parser.Extractor, relPath string, @@ -118,10 +120,20 @@ func (idx *Indexer) extractWithTimeoutDone( if done == nil { done = func() {} } + releaseLifecycle, err := idx.extractionLifecycle.admit() + if err != nil { + done() + return nil, err + } + finish := func() { + defer releaseLifecycle() + done() + } + opts := idx.extractionOptionsValue() budget := effectiveExtractBudget(idx.config.MaxExtractMillis, len(src)) if budget <= 0 { - defer done() - return safeExtract(ext, relPath, src) + defer finish() + return safeExtractWithOptions(ext, relPath, src, opts) } type outcome struct { result *parser.ExtractionResult @@ -129,8 +141,8 @@ func (idx *Indexer) extractWithTimeoutDone( } ch := make(chan outcome, 1) go func() { - r, err := safeExtract(ext, relPath, src) - done() + r, err := safeExtractWithOptions(ext, relPath, src, opts) + finish() ch <- outcome{result: r, err: err} }() timer := time.NewTimer(time.Duration(budget) * time.Millisecond) diff --git a/internal/indexer/skip_telemetry_test.go b/internal/indexer/skip_telemetry_test.go index dd0098b6b..5c2f734d2 100644 --- a/internal/indexer/skip_telemetry_test.go +++ b/internal/indexer/skip_telemetry_test.go @@ -32,7 +32,7 @@ func (s *slowExtractor) Extract(filePath string, _ []byte) (*parser.ExtractionRe func TestExtractWithTimeout_NoBudget(t *testing.T) { idx := newTestIndexer(graph.New()) // MaxExtractMillis = 0 - r, err := idx.extractWithTimeout(&slowExtractor{delay: 5 * time.Millisecond}, "x.slow", []byte("x")) + r, err := idx.extractWithTimeoutDone(&slowExtractor{delay: 5 * time.Millisecond}, "x.slow", []byte("x"), nil) require.NoError(t, err) require.NotNil(t, r) } @@ -40,7 +40,7 @@ func TestExtractWithTimeout_NoBudget(t *testing.T) { func TestExtractWithTimeout_FastFileUnderBudget(t *testing.T) { idx := newTestIndexer(graph.New()) idx.config.MaxExtractMillis = 2000 - r, err := idx.extractWithTimeout(&slowExtractor{delay: 5 * time.Millisecond}, "x.slow", []byte("x")) + r, err := idx.extractWithTimeoutDone(&slowExtractor{delay: 5 * time.Millisecond}, "x.slow", []byte("x"), nil) require.NoError(t, err) require.NotNil(t, r) } @@ -48,7 +48,7 @@ func TestExtractWithTimeout_FastFileUnderBudget(t *testing.T) { func TestExtractWithTimeout_SlowFileTimesOut(t *testing.T) { idx := newTestIndexer(graph.New()) idx.config.MaxExtractMillis = 50 - _, err := idx.extractWithTimeout(&slowExtractor{delay: 800 * time.Millisecond}, "x.slow", []byte("x")) + _, err := idx.extractWithTimeoutDone(&slowExtractor{delay: 800 * time.Millisecond}, "x.slow", []byte("x"), nil) require.ErrorIs(t, err, errExtractTimeout) } diff --git a/internal/indexer/streaming.go b/internal/indexer/streaming.go index 55c432d2d..8a9cd030c 100644 --- a/internal/indexer/streaming.go +++ b/internal/indexer/streaming.go @@ -14,6 +14,12 @@ import ( // and recovers any per-document panic (mirroring safeExtract) so a malformed // asset isolates to a failed file rather than crashing the pass. func (idx *Indexer) extractStreaming(se parser.StreamingExtractor, path, relPath string) (res *parser.ExtractionResult, err error) { + releaseLifecycle, admissionErr := idx.extractionLifecycle.admit() + if admissionErr != nil { + return nil, admissionErr + } + defer releaseLifecycle() + f, oerr := os.Open(path) if oerr != nil { return nil, oerr diff --git a/internal/indexer/structural_integrity.go b/internal/indexer/structural_integrity.go new file mode 100644 index 000000000..8514f4d91 --- /dev/null +++ b/internal/indexer/structural_integrity.go @@ -0,0 +1,7 @@ +package indexer + +import "github.com/zzet/gortex/internal/graph" + +func (idx *Indexer) newStructuralIntegrityShadow(owner graph.Store, path graph.StructuralDropPath) *graph.Graph { + return graph.NewStructuralIntegrityShadow(owner, idx.RepoPrefix(), path) +} diff --git a/internal/indexer/structural_integrity_test.go b/internal/indexer/structural_integrity_test.go new file mode 100644 index 000000000..69641fb9f --- /dev/null +++ b/internal/indexer/structural_integrity_test.go @@ -0,0 +1,143 @@ +package indexer + +import ( + "go/ast" + "go/parser" + "go/token" + "os" + "path/filepath" + "runtime" + "testing" + + "github.com/zzet/gortex/internal/graph" +) + +func TestIndexerShadowsForwardRejectedAttemptsToDurableOwner(t *testing.T) { + tests := []struct { + name string + path graph.StructuralDropPath + }{ + {name: "cold", path: graph.StructuralPathShadowCold}, + {name: "streaming", path: graph.StructuralPathShadowStreaming}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + owner := graph.New() + idx := &Indexer{repoPrefix: "repo-shadow"} + shadow := idx.newStructuralIntegrityShadow(owner, tt.path) + shadow.AddBatch(nil, []*graph.Edge{{ + From: "source", To: "target#param:x", Kind: graph.EdgeImplements, Origin: "LSP", + }}) + // Simulate cancellation before the throwaway shadow crosses its durable + // drain boundary: the shadow is discarded, but the attempted rejection + // must already belong to the logical durable store. + shadow = nil + _ = shadow + + snapshot := owner.StructuralIntegritySnapshot(graph.StructuralIntegritySnapshotOptions{IncludeAttribution: true, IncludeSamples: true}) + if snapshot.Totals.WriteRejected != 1 || len(snapshot.Attribution) != 1 || len(snapshot.Samples) != 1 { + t.Fatalf("rejected attempt did not survive shadow discard: %+v", snapshot) + } + got := snapshot.Attribution[0] + if got.Repo != "repo-shadow" || got.Path != tt.path || got.Count != 1 { + t.Fatalf("shadow attribution mismatch: %+v", got) + } + }) + } +} + +// TestIndexCtxRawUsesIntegrityAwareColdAndStreamingShadows guards the actual +// staging call sites. The end-to-end indexCtxRaw paths require the full cold +// indexing pipeline (and streaming mode is intentionally disabled by default), +// so this AST assertion makes a regression back to graph.New fail at the two +// precise shadow assignments while the behavioral test above covers forwarding +// and aborted-attempt retention. +func TestIndexCtxRawUsesIntegrityAwareColdAndStreamingShadows(t *testing.T) { + _, testFile, _, ok := runtime.Caller(0) + if !ok { + t.Fatal("resolve test source path") + } + indexerPath := filepath.Join(filepath.Dir(testFile), "indexer.go") + source, err := os.ReadFile(indexerPath) + if err != nil { + t.Fatalf("read indexer source: %v", err) + } + file, err := parser.ParseFile(token.NewFileSet(), indexerPath, source, 0) + if err != nil { + t.Fatalf("parse indexer source: %v", err) + } + + type expectedCall struct { + owner string + path string + count int + } + expected := map[string]*expectedCall{ + "inMemShadow": {owner: "diskTarget", path: "StructuralPathShadowCold"}, + "chunkShadow": {owner: "streamingDisk", path: "StructuralPathShadowStreaming"}, + } + methodFound := false + for _, declaration := range file.Decls { + method, ok := declaration.(*ast.FuncDecl) + if !ok || method.Name.Name != "indexCtxRaw" || method.Body == nil { + continue + } + methodFound = true + ast.Inspect(method.Body, func(node ast.Node) bool { + assignment, ok := node.(*ast.AssignStmt) + if !ok { + return true + } + for i, rhs := range assignment.Rhs { + if i >= len(assignment.Lhs) { + continue + } + lhs, ok := assignment.Lhs[i].(*ast.Ident) + if !ok { + continue + } + want, ok := expected[lhs.Name] + if !ok { + continue + } + call, ok := rhs.(*ast.CallExpr) + if ok && integrityShadowCallMatches(call, want.owner, want.path) { + want.count++ + } + } + return true + }) + } + if !methodFound { + t.Fatal("indexCtxRaw method not found") + } + for assignment, want := range expected { + if want.count != 1 { + t.Fatalf("%s must be assigned exactly once from idx.newStructuralIntegrityShadow(%s, graph.%s); found %d", assignment, want.owner, want.path, want.count) + } + } +} + +func integrityShadowCallMatches(call *ast.CallExpr, owner, path string) bool { + if len(call.Args) != 2 { + return false + } + method, ok := call.Fun.(*ast.SelectorExpr) + if !ok || method.Sel.Name != "newStructuralIntegrityShadow" { + return false + } + receiver, ok := method.X.(*ast.Ident) + if !ok || receiver.Name != "idx" { + return false + } + ownerArg, ok := call.Args[0].(*ast.Ident) + if !ok || ownerArg.Name != owner { + return false + } + pathArg, ok := call.Args[1].(*ast.SelectorExpr) + if !ok || pathArg.Sel.Name != path { + return false + } + pathPackage, ok := pathArg.X.(*ast.Ident) + return ok && pathPackage.Name == "graph" +} diff --git a/internal/indexer/transform.go b/internal/indexer/transform.go index 6f6bb006f..21e12e033 100644 --- a/internal/indexer/transform.go +++ b/internal/indexer/transform.go @@ -80,13 +80,6 @@ func newTransformPipeline(rules []config.TransformRule, logger *zap.Logger) *tra return p } -// addPrePass registers an offset-preserving pre-parse transform. Pre-parse -// transforms run before the offset-shifting ones and their length is enforced -// by run. -func (p *transformPipeline) addPrePass(t preParseTransform) { - p.prePass = append(p.prePass, t) -} - // run applies every matching transform to src in order. A transform // that errors is logged and skipped — the bytes from the previous // stage are kept, so one failing processor never drops a file. diff --git a/internal/indexer/transform_offset_test.go b/internal/indexer/transform_offset_test.go index 2635fe963..2769dbb0d 100644 --- a/internal/indexer/transform_offset_test.go +++ b/internal/indexer/transform_offset_test.go @@ -63,7 +63,7 @@ func TestTransformPipeline_OffsetPreserving(t *testing.T) { t.Run("length-changing pre-parse transform is rejected", func(t *testing.T) { src := []byte("hello world") p := newTransformPipeline(nil, nil) - p.addPrePass(shrinkingPreParse{}) + p.prePass = append(p.prePass, shrinkingPreParse{}) out := p.run("x.txt", src) if !bytes.Equal(out, src) { t.Errorf("a length-changing pre-parse transform must be dropped; got %q", out) diff --git a/internal/indexer/upgrade_gating_test.go b/internal/indexer/upgrade_gating_test.go deleted file mode 100644 index aaf4725d9..000000000 --- a/internal/indexer/upgrade_gating_test.go +++ /dev/null @@ -1,125 +0,0 @@ -package indexer - -import ( - "os" - "path/filepath" - "testing" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - "go.uber.org/zap" - - "github.com/zzet/gortex/internal/config" - "github.com/zzet/gortex/internal/graph" - "github.com/zzet/gortex/internal/parser" - "github.com/zzet/gortex/internal/parser/languages" - "github.com/zzet/gortex/internal/search" -) - -// upgradeSpawnCount is an in-package accessor kept as a function so -// the test doesn't have to export a field just to read the counter. -func upgradeSpawnCount(idx *Indexer) int { - idx.upgradeSpawnedMu.Lock() - defer idx.upgradeSpawnedMu.Unlock() - return idx.upgradeSpawned -} - -// Second and later calls to upgradeSearchToBleve must be a no-op when -// the text backend is already Bleve. Under the bug, each call would -// rebuild a full Bleve index and Swap it in — observable as a new -// Bleve pointer identity. The defensive early-return at the top of -// the function keeps the pointer stable. -func TestUpgradeSearchToBleve_IdempotentOnSecondCall(t *testing.T) { - dir := t.TempDir() - require.NoError(t, os.WriteFile(filepath.Join(dir, "main.go"), []byte(`package main - -func Alpha() {} -`), 0o644)) - - g := graph.New() - reg := parser.NewRegistry() - reg.Register(languages.NewGoExtractor()) - - cfg := config.Default().Index - cfg.Workers = 1 - cfg.SkipSearch = config.DefaultSkipSearch() - - idx := New(g, reg, cfg, zap.NewNop()) - idx.SetEmbedder(stubEmbedder{}) - - _, err := idx.Index(dir) - require.NoError(t, err) - - sw, ok := idx.Search().(*search.Swappable) - require.True(t, ok) - - // First upgrade: text side transitions BM25 → Bleve. We don't - // care about identity here — just capture the post-upgrade - // Bleve pointer so we can compare to the second run. - idx.upgradeSearchToBleve(idx.snapshotBleveEntries()) - firstHybrid, ok := sw.Inner().(*search.HybridBackend) - require.True(t, ok, "after first upgrade, inner must be Hybrid(Bleve, Vector)") - firstBleve, ok := firstHybrid.TextBackend().(*search.BleveBackend) - require.True(t, ok) - - // Second upgrade must early-return. No new Bleve is built, no - // Swap happens. - idx.upgradeSearchToBleve(idx.snapshotBleveEntries()) - secondHybrid, ok := sw.Inner().(*search.HybridBackend) - require.True(t, ok, "inner should still be Hybrid after no-op second call") - secondBleve, ok := secondHybrid.TextBackend().(*search.BleveBackend) - require.True(t, ok) - - assert.Same(t, firstHybrid, secondHybrid, - "second upgrade must not replace the Hybrid wrapper") - assert.Same(t, firstBleve, secondBleve, - "second upgrade must not rebuild Bleve") -} - -// Only one upgrade goroutine may spawn per Indexer lifetime, even if -// IndexCtx runs many times past AutoThreshold (as it does in -// multi-repo warmup). Uses a low override threshold so a tiny test -// corpus can trigger it. -func TestIndexCtx_SpawnsUpgradeAtMostOnce(t *testing.T) { - dir := t.TempDir() - // Two Go files, each with several funcs — enough that one index - // pass produces a handful of search docs. The threshold overrides - // below ensure we cross it. - require.NoError(t, os.WriteFile(filepath.Join(dir, "a.go"), []byte(`package main - -func A1() {} -func A2() {} -func A3() {} -`), 0o644)) - require.NoError(t, os.WriteFile(filepath.Join(dir, "b.go"), []byte(`package main - -func B1() {} -func B2() {} -func B3() {} -`), 0o644)) - - g := graph.New() - reg := parser.NewRegistry() - reg.Register(languages.NewGoExtractor()) - - cfg := config.Default().Index - cfg.Workers = 1 - cfg.SkipSearch = config.DefaultSkipSearch() - - idx := New(g, reg, cfg, zap.NewNop()) - idx.SetEmbedder(stubEmbedder{}) - - // Temporarily force the threshold to 1 so any non-empty index - // triggers the upgrade branch. Restored after the test. - orig := search.AutoThreshold - defer func() { search.AutoThreshold = orig }() - search.AutoThreshold = 1 - - for range 5 { - _, err := idx.Index(dir) - require.NoError(t, err) - } - - assert.Equal(t, 1, upgradeSpawnCount(idx), - "upgradeOnce must gate all post-threshold Index calls to one spawn") -} diff --git a/internal/indexer/upgrade_hybrid_test.go b/internal/indexer/upgrade_hybrid_test.go deleted file mode 100644 index d436ee1a5..000000000 --- a/internal/indexer/upgrade_hybrid_test.go +++ /dev/null @@ -1,97 +0,0 @@ -package indexer - -import ( - "context" - "os" - "path/filepath" - "testing" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - "go.uber.org/zap" - - "github.com/zzet/gortex/internal/config" - "github.com/zzet/gortex/internal/graph" - "github.com/zzet/gortex/internal/parser" - "github.com/zzet/gortex/internal/parser/languages" - "github.com/zzet/gortex/internal/search" -) - -// stubEmbedder is a deterministic minimal embedder that lets the -// indexer wire up a HybridBackend in tests without pulling in the -// static-vector asset or a real ONNX runtime. It emits a 4-dim -// vector whose first element is the text length — enough for the -// vector backend to accept the adds and for Search to return -// something non-empty. -type stubEmbedder struct{} - -func (stubEmbedder) Embed(_ context.Context, text string) ([]float32, error) { - return []float32{float32(len(text)), 0, 0, 0}, nil -} - -func (stubEmbedder) EmbedBatch(_ context.Context, texts []string) ([][]float32, error) { - out := make([][]float32, len(texts)) - for i, t := range texts { - out[i] = []float32{float32(len(t)), 0, 0, 0} - } - return out, nil -} - -func (stubEmbedder) Dimensions() int { return 4 } -func (stubEmbedder) Close() error { return nil } - -// upgradeSearchToBleve must rewrap the new Bleve in a HybridBackend -// carrying the pre-existing vector + embedder. Without this, swapping -// in a raw *BleveBackend causes Swap to Close() the old Hybrid (which -// closes only the text side) and leaves every downstream query -// silently degraded to BM25-only, with vectorBytes reporting as 0 in -// the daemon status. Regression guard for the vector-index -// destruction bug that went live in v0.10.0. -func TestUpgradeSearchToBleve_PreservesVectorIndex(t *testing.T) { - dir := t.TempDir() - require.NoError(t, os.WriteFile(filepath.Join(dir, "main.go"), []byte(`package main - -func Alpha() {} -func Beta() {} -`), 0o644)) - - g := graph.New() - reg := parser.NewRegistry() - reg.Register(languages.NewGoExtractor()) - - cfg := config.Default().Index - cfg.Workers = 1 - cfg.SkipSearch = config.DefaultSkipSearch() - - idx := New(g, reg, cfg, zap.NewNop()) - idx.SetEmbedder(stubEmbedder{}) - - _, err := idx.Index(dir) - require.NoError(t, err) - - // Sanity: post-index, inner is Hybrid(BM25, Vector). - sw, ok := idx.Search().(*search.Swappable) - require.True(t, ok, "indexer search is always a Swappable") - preHybrid, ok := sw.Inner().(*search.HybridBackend) - require.True(t, ok, "buildSearchIndex should have produced a HybridBackend") - require.NotNil(t, preHybrid.VectorIndex(), "vector backend must be attached") - preVector := preHybrid.VectorIndex() - preVectorBytes := preHybrid.VectorSizeBytes() - require.Greater(t, preVectorBytes, uint64(0), "vector index should have content") - - // Force the upgrade directly — don't go through the AutoThreshold - // check. Under the bug the resulting inner is a raw *BleveBackend. - // Under the fix it is a new Hybrid wrapping the same vector. - idx.upgradeSearchToBleve(idx.snapshotBleveEntries()) - - postHybrid, ok := sw.Inner().(*search.HybridBackend) - require.True(t, ok, "post-upgrade inner must still be *HybridBackend") - assert.Same(t, preVector, postHybrid.VectorIndex(), - "vector backend pointer must be preserved across the upgrade") - assert.Equal(t, preVectorBytes, postHybrid.VectorSizeBytes(), - "vector byte count must be unchanged — Hybrid.Close does not touch the vector") - - // Text side must be the new Bleve, not the old BM25. - _, isBleve := postHybrid.TextBackend().(*search.BleveBackend) - assert.True(t, isBleve, "text side must be the upgraded Bleve backend") -} diff --git a/internal/indexer/vector_multirepo_test.go b/internal/indexer/vector_multirepo_test.go new file mode 100644 index 000000000..c40a81da6 --- /dev/null +++ b/internal/indexer/vector_multirepo_test.go @@ -0,0 +1,90 @@ +package indexer + +import ( + "context" + "path/filepath" + "testing" + + "github.com/stretchr/testify/require" + "go.uber.org/zap" + + "github.com/zzet/gortex/internal/config" + "github.com/zzet/gortex/internal/graph/store_sqlite" + "github.com/zzet/gortex/internal/search" +) + +func requirePublishedVectorStats( + t *testing.T, + sw *search.Swappable, + wantVectorCount int, + wantChunks bool, +) { + t.Helper() + backend, release := sw.AcquireBackend() + defer release() + hybrid, ok := backend.(*search.HybridBackend) + require.True(t, ok, "shared search backend must publish one hybrid layer") + require.NotNil(t, hybrid.VectorIndex()) + require.Equal(t, wantVectorCount, hybrid.VectorIndex().Count()) + require.Equal(t, wantChunks, hybrid.VectorIndex().HasChunks()) + require.Zero(t, hybrid.VectorSizeBytes(), "multi-repo publication must never retain a repo-local heap HNSW") +} + +func TestIndexMultiRepoPublishesCombinedDurableVectorCorpus(t *testing.T) { + repos := coldOrchestrationRepos(t, 2) + store, err := store_sqlite.Open(filepath.Join(t.TempDir(), "multi-vectors.sqlite")) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, store.Close()) }) + sw := search.NewSwappable(initialSearchBackend(store)) + mi := NewMultiIndexer(store, coldOrchestrationRegistry(), sw, nil, zap.NewNop()) + mi.SetEmbedder(&poolEmbedder{}) + + results, err := mi.indexMultiRepo(repos) + require.NoError(t, err) + require.Len(t, results, len(repos)) + stats, err := store.VectorCorpusStats(context.Background(), 3) + require.NoError(t, err) + require.Greater(t, stats.VectorCount, 0) + for _, entry := range repos { + prefix := config.ResolvePrefix(entry) + repoStats, statsErr := store.VectorCorpusStatsForRepo(context.Background(), prefix, 3) + require.NoError(t, statsErr) + require.Greater(t, repoStats.RepositoryVectorCount, 0, "%s corpus missing", prefix) + } + requirePublishedVectorStats(t, sw, stats.VectorCount, stats.ChunkCount > 0) +} + +func TestUntrackRepoRemovesVectorsAndRepublishesAggregateStats(t *testing.T) { + repos := coldOrchestrationRepos(t, 2) + cm := newTestConfigManager(t) + cm.Global().Repos = append([]config.RepoEntry(nil), repos...) + store, err := store_sqlite.Open(filepath.Join(t.TempDir(), "untrack-vectors.sqlite")) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, store.Close()) }) + sw := search.NewSwappable(initialSearchBackend(store)) + mi := NewMultiIndexer(store, coldOrchestrationRegistry(), sw, cm, zap.NewNop()) + mi.SetEmbedder(&poolEmbedder{}) + _, err = mi.indexMultiRepo(repos) + require.NoError(t, err) + + prefixA := config.ResolvePrefix(repos[0]) + prefixB := config.ResolvePrefix(repos[1]) + mi.UntrackRepo(prefixA) + statsA, err := store.VectorCorpusStatsForRepo(context.Background(), prefixA, 3) + require.NoError(t, err) + require.Zero(t, statsA.RepositoryVectorCount) + statsB, err := store.VectorCorpusStatsForRepo(context.Background(), prefixB, 3) + require.NoError(t, err) + require.Greater(t, statsB.RepositoryVectorCount, 0) + requirePublishedVectorStats(t, sw, statsB.VectorCount, statsB.ChunkCount > 0) + + mi.UntrackRepo(prefixB) + empty, err := store.VectorCorpusStats(context.Background(), 3) + require.NoError(t, err) + require.Zero(t, empty.VectorCount) + requirePublishedVectorStats(t, sw, 0, false) + backend, release := sw.AcquireBackend() + hybrid := backend.(*search.HybridBackend) + require.Empty(t, hybrid.VectorIndex().Search([]float32{1, 0, 0}, 10)) + release() +} diff --git a/internal/indexer/vector_persist_test.go b/internal/indexer/vector_persist_test.go index c270ad918..d8049ebb9 100644 --- a/internal/indexer/vector_persist_test.go +++ b/internal/indexer/vector_persist_test.go @@ -2,8 +2,12 @@ package indexer import ( "context" + "errors" "path/filepath" + "sync" + "sync/atomic" "testing" + "time" "github.com/stretchr/testify/require" "go.uber.org/zap" @@ -13,6 +17,7 @@ import ( "github.com/zzet/gortex/internal/graph/store_sqlite" "github.com/zzet/gortex/internal/parser" "github.com/zzet/gortex/internal/parser/languages" + "github.com/zzet/gortex/internal/search" ) // TestBulkLoad_PersistsVectorsToBackend guards the fix where the embedding @@ -22,10 +27,9 @@ import ( // reached the in-process HNSW and never the sqlite `vectors` table. They were // lost on the next daemon restart, forcing a full (paid) re-embed every time. // -// The fix captures the disk store at the shadow swap (idx.bulkVectorSink) and -// persists the vectors against it. This test indexes a tiny repo into a sqlite -// store with an embedder wired in, then asserts the backend's vectors table is -// populated — it was empty before the fix. +// The fix prepares an immutable vector plan while the shadow is live, then +// installs it only after the graph has drained and idx.graph points back to the +// durable store. This test asserts both persistence and delegated publication. func TestBulkLoad_PersistsVectorsToBackend(t *testing.T) { dir := t.TempDir() writeFile(t, filepath.Join(dir, "app.py"), ` @@ -68,4 +72,224 @@ def store(value): require.NotEmpty(t, embs, "embedded vectors must be persisted to the sqlite backend under the bulk loader "+ "(regression: vectors lived only in the in-process HNSW and were lost on restart)") + assertDelegatedVectorPublication(t, idx) +} + +type vectorPublicationProbeStore struct { + *store_sqlite.Store + flushStarted chan struct{} + flushRelease chan struct{} + replaceCalled chan struct{} + flushOnce sync.Once + replaceOnce sync.Once + flushErr error + replaceErr error + replaceCalls atomic.Int32 +} + +func newVectorPublicationProbeStore(store *store_sqlite.Store) *vectorPublicationProbeStore { + return &vectorPublicationProbeStore{ + Store: store, + flushStarted: make(chan struct{}), + replaceCalled: make(chan struct{}), + } +} + +func (s *vectorPublicationProbeStore) FlushBulk() error { + s.flushOnce.Do(func() { close(s.flushStarted) }) + if s.flushRelease != nil { + <-s.flushRelease + } + if s.flushErr != nil { + return s.flushErr + } + return s.Store.FlushBulk() +} + +func (s *vectorPublicationProbeStore) ReplaceVectorCorpus( + ctx context.Context, + repoPrefix string, + dims int, + items []graph.VectorCorpusItem, +) (graph.VectorCorpusStats, error) { + s.replaceCalls.Add(1) + s.replaceOnce.Do(func() { close(s.replaceCalled) }) + if s.replaceErr != nil { + return graph.VectorCorpusStats{}, s.replaceErr + } + return s.Store.ReplaceVectorCorpus(ctx, repoPrefix, dims, items) +} + +func vectorPersistFixture(t *testing.T, files int) string { + t.Helper() + dir := t.TempDir() + for i := 0; i < files; i++ { + writeFile(t, filepath.Join(dir, "app"+string(rune('a'+i))+".py"), ` +def fetch(url): + return url + +def store(value): + return value +`) + } + return dir +} + +func newVectorPersistIndexer(t *testing.T, store graph.Store, emb *poolEmbedder) *Indexer { + t.Helper() + reg := parser.NewRegistry() + reg.Register(languages.NewPythonExtractor()) + cfg := config.Default().Index + cfg.Workers = 2 + idx := New(store, reg, cfg, zap.NewNop()) + idx.SetEmbedder(emb) + return idx +} + +func assertDelegatedVectorPublication(t *testing.T, idx *Indexer) { + t.Helper() + backend, release := idx.swappable().AcquireBackend() + defer release() + hybrid, ok := backend.(*search.HybridBackend) + require.True(t, ok, "vector publication must install a hybrid backend") + require.NotNil(t, hybrid.VectorIndex()) + require.Greater(t, hybrid.VectorIndex().Count(), 0) + require.Zero(t, hybrid.VectorSizeBytes(), "durable delegation must retain no heap HNSW") +} + +func waitForVectorPublicationSignal(t *testing.T, signal <-chan struct{}, label string) { + t.Helper() + select { + case <-signal: + case <-time.After(10 * time.Second): + t.Fatalf("timed out waiting for %s", label) + } +} + +func TestBulkLoad_PublishesVectorsOnlyAfterDurableFlush(t *testing.T) { + t.Setenv("GORTEX_SHADOW_MAX_FILES", "1000000") + t.Setenv("GORTEX_SHADOW_MAX_BYTES", "1073741824") + t.Setenv("GORTEX_STREAMING_FLUSH", "0") + dir := vectorPersistFixture(t, 1) + base, err := store_sqlite.Open(filepath.Join(t.TempDir(), "store.sqlite")) + require.NoError(t, err) + t.Cleanup(func() { _ = base.Close() }) + probe := newVectorPublicationProbeStore(base) + probe.flushRelease = make(chan struct{}) + idx := newVectorPersistIndexer(t, probe, &poolEmbedder{}) + + before, release := idx.swappable().AcquireBackend() + release() + done := make(chan error, 1) + go func() { + _, indexErr := idx.IndexCtx(context.Background(), dir) + done <- indexErr + }() + + waitForVectorPublicationSignal(t, probe.flushStarted, "bulk flush") + select { + case <-probe.replaceCalled: + t.Fatal("vector corpus was installed before the shadow graph completed its durable flush") + default: + } + during, release := idx.swappable().AcquireBackend() + release() + if during != before { + t.Fatal("active search backend changed before durable graph flush") + } + + close(probe.flushRelease) + select { + case indexErr := <-done: + require.NoError(t, indexErr) + case <-time.After(10 * time.Second): + t.Fatal("index did not finish after releasing the durable flush") + } + waitForVectorPublicationSignal(t, probe.replaceCalled, "vector corpus replacement") + require.Equal(t, int32(1), probe.replaceCalls.Load()) + assertDelegatedVectorPublication(t, idx) +} + +func TestBulkLoad_FlushFailureDoesNotPublishVectors(t *testing.T) { + t.Setenv("GORTEX_SHADOW_MAX_FILES", "1000000") + t.Setenv("GORTEX_SHADOW_MAX_BYTES", "1073741824") + t.Setenv("GORTEX_STREAMING_FLUSH", "0") + dir := vectorPersistFixture(t, 1) + base, err := store_sqlite.Open(filepath.Join(t.TempDir(), "store.sqlite")) + require.NoError(t, err) + t.Cleanup(func() { _ = base.Close() }) + probe := newVectorPublicationProbeStore(base) + probe.flushErr = errors.New("injected flush failure") + idx := newVectorPersistIndexer(t, probe, &poolEmbedder{}) + before, release := idx.swappable().AcquireBackend() + release() + + _, err = idx.IndexCtx(context.Background(), dir) + require.ErrorIs(t, err, probe.flushErr) + require.Zero(t, probe.replaceCalls.Load(), "failed graph persistence must not install vectors") + after, release := idx.swappable().AcquireBackend() + release() + if after != before { + t.Fatal("failed shadow drain changed the active search backend") + } +} + +func TestBulkLoad_NativeVectorFailureFallsBackWithoutReembedding(t *testing.T) { + t.Setenv("GORTEX_SHADOW_MAX_FILES", "1000000") + t.Setenv("GORTEX_SHADOW_MAX_BYTES", "1073741824") + t.Setenv("GORTEX_STREAMING_FLUSH", "0") + dir := vectorPersistFixture(t, 1) + base, err := store_sqlite.Open(filepath.Join(t.TempDir(), "store.sqlite")) + require.NoError(t, err) + t.Cleanup(func() { _ = base.Close() }) + probe := newVectorPublicationProbeStore(base) + probe.replaceErr = errors.New("injected vector replacement failure") + emb := &poolEmbedder{} + idx := newVectorPersistIndexer(t, probe, emb) + + _, err = idx.IndexCtx(context.Background(), dir) + require.NoError(t, err, "single-repository indexing may degrade to the prepared heap fallback") + require.Equal(t, int32(1), emb.calls, "fallback must reuse the prepared vectors without a second paid pass") + require.ErrorIs(t, idx.LastVectorBuildError(), probe.replaceErr) + + backend, release := idx.swappable().AcquireBackend() + defer release() + hybrid, ok := backend.(*search.HybridBackend) + require.True(t, ok) + require.Greater(t, hybrid.VectorIndex().Count(), 0) + require.Greater(t, hybrid.VectorSizeBytes(), uint64(0), "fallback must be the process-local heap backend") +} + +func TestDirectSQLiteIndexPublishesDelegatedVectorCorpus(t *testing.T) { + t.Setenv("GORTEX_SHADOW_MAX_FILES", "0") + t.Setenv("GORTEX_STREAMING_FLUSH", "0") + dir := vectorPersistFixture(t, 1) + base, err := store_sqlite.Open(filepath.Join(t.TempDir(), "store.sqlite")) + require.NoError(t, err) + t.Cleanup(func() { _ = base.Close() }) + probe := newVectorPublicationProbeStore(base) + idx := newVectorPersistIndexer(t, probe, &poolEmbedder{}) + + _, err = idx.IndexCtx(context.Background(), dir) + require.NoError(t, err) + require.Equal(t, int32(1), probe.replaceCalls.Load()) + assertDelegatedVectorPublication(t, idx) +} + +func TestStreamingFlush_PublishesOneDelegatedVectorCorpus(t *testing.T) { + t.Setenv("GORTEX_SHADOW_MAX_FILES", "0") + t.Setenv("GORTEX_STREAMING_FLUSH", "1") + t.Setenv("GORTEX_STREAMING_CHUNK_SIZE", "1") + dir := vectorPersistFixture(t, 2) + base, err := store_sqlite.Open(filepath.Join(t.TempDir(), "store.sqlite")) + require.NoError(t, err) + t.Cleanup(func() { _ = base.Close() }) + probe := newVectorPublicationProbeStore(base) + idx := newVectorPersistIndexer(t, probe, &poolEmbedder{}) + + _, err = idx.IndexCtx(context.Background(), dir) + require.NoError(t, err) + require.Equal(t, int32(1), probe.replaceCalls.Load(), + "streaming chunks must prepare once and publish only after the final chunk") + assertDelegatedVectorPublication(t, idx) } diff --git a/internal/indexer/vector_plan.go b/internal/indexer/vector_plan.go new file mode 100644 index 000000000..50f66a12c --- /dev/null +++ b/internal/indexer/vector_plan.go @@ -0,0 +1,478 @@ +package indexer + +import ( + "context" + "errors" + "fmt" + "math" + "os" + "strconv" + "strings" + "time" + + "github.com/zzet/gortex/internal/graph" + "github.com/zzet/gortex/internal/search" + "go.uber.org/zap" +) + +// preparedVectorPlan is the immutable hand-off between embedding and vector +// publication. It owns only IDs and copied vectors: graph nodes and source +// buffers are deliberately not retained across a cold-shadow drain. +type preparedVectorPlan struct { + repoPrefix string + dims int + items []graph.VectorCorpusItem + chunkMap map[string]string + dropped int + droppedIDs []string + embeddedText int +} + +// Release drops every potentially large reference held by the plan. Heap +// vector publication copies the chunk map first; durable publication copies +// vectors into its transaction before this method runs. +func (p *preparedVectorPlan) Release() { + if p == nil { + return + } + for i := range p.items { + p.items[i].NodeID = "" + p.items[i].ParentID = "" + p.items[i].Vec = nil + } + p.items = nil + clear(p.chunkMap) + p.chunkMap = nil + clear(p.droppedIDs) + p.droppedIDs = nil + p.repoPrefix = "" + p.dims = 0 + p.dropped = 0 + p.embeddedText = 0 +} + +// prepareSearchIndex updates the process-local text index, then prepares the +// vector corpus without mutating either the durable vector store or the active +// vector backend. The caller decides when publication is safe. +func (idx *Indexer) prepareSearchIndex(ctx context.Context) (*preparedVectorPlan, error) { + if ctx == nil { + ctx = context.Background() + } + idx.lastVectorBuildErr = nil + + // Install learned sub-word boundaries before populating an in-process BM25 + // backend. Native SQLite FTS needs neither the census nor duplicate Adds. + search.BuildAndInstallNgramBoundaries(idx.search, idx.graph) + nativeText := isSymbolSearcherBackend(idx.search) + buildVectors := idx.embedder != nil + if nativeText && !buildVectors { + return nil, nil + } + + // Content sections belong to content search, not symbol/vector search. + nodes := graph.RepoCodeNodes(idx.graph, idx.repoPrefix) + if !nativeText { + for _, n := range nodes { + if idx.shouldIndexForSearch(n) { + idx.search.Add(n.ID, searchIndexFields(n, idx.projectName)...) + } + } + } + if !buildVectors { + return nil, nil + } + return idx.prepareVectorPlan(ctx, nodes) +} + +// rebuildTextSearchIndex restores only the non-persistent text channel. It is +// used on warm startup after a durable vector corpus has been published, so a +// paid embedding pass is not repeated merely to reconstruct BM25. +func (idx *Indexer) rebuildTextSearchIndex() { + search.BuildAndInstallNgramBoundaries(idx.search, idx.graph) + if isSymbolSearcherBackend(idx.search) { + return + } + for _, n := range graph.RepoCodeNodes(idx.graph, idx.repoPrefix) { + if idx.shouldIndexForSearch(n) { + idx.search.Add(n.ID, searchIndexFields(n, idx.projectName)...) + } + } +} + +func (idx *Indexer) prepareVectorPlan(ctx context.Context, nodes []*graph.Node) (*preparedVectorPlan, error) { + if err := ctx.Err(); err != nil { + return nil, err + } + texts, ids, collectedChunks, skipped := idx.collectEmbedTexts(nodes) + if skipped > 0 { + idx.logger.Info("skipped embedding for low-value nodes", + zap.Int("count", skipped), + zap.Int("embedded", len(texts))) + } + + const ( + defaultEmbedMaxSymbols = 100_000 + embedChunkSize = 500 + embedChunkTimeout = 5 * time.Minute + ) + embedMaxSymbols := defaultEmbedMaxSymbols + if idx.embedMaxSymbols > 0 { + embedMaxSymbols = idx.embedMaxSymbols + } + if env := os.Getenv("GORTEX_EMBEDDINGS_MAX_SYMBOLS"); env != "" { + if n, err := strconv.Atoi(strings.TrimSpace(env)); err == nil && n > 0 { + embedMaxSymbols = n + } + } + if len(texts) > embedMaxSymbols { + return nil, fmt.Errorf("embedding text count %d exceeds threshold %d (raise embedding.max_symbols)", len(texts), embedMaxSymbols) + } + + // An empty successful plan is authoritative: installing it removes stale + // vectors for a repository that no longer has embeddable symbols. + if len(texts) == 0 { + return &preparedVectorPlan{ + repoPrefix: idx.repoPrefix, + dims: embeddingDimsOrDefault(idx.embedder), + }, nil + } + + var embedWithRetry func(context.Context, []string) ([][]float32, error) + embedWithRetry = func(parent context.Context, items []string) ([][]float32, error) { + chunkCtx, cancel := context.WithTimeout(parent, embedChunkTimeout) + out, err := idx.embedder.EmbedBatch(chunkCtx, items) + cancel() + if err == nil { + return out, nil + } + if parent.Err() != nil || !errors.Is(err, context.DeadlineExceeded) || len(items) <= 1 { + return nil, err + } + idx.logger.Warn("embed chunk timed out, retrying with halved batch", + zap.Int("size", len(items)), zap.Error(err)) + mid := len(items) / 2 + left, leftErr := embedWithRetry(parent, items[:mid]) + if leftErr != nil { + return nil, leftErr + } + right, rightErr := embedWithRetry(parent, items[mid:]) + if rightErr != nil { + return nil, rightErr + } + return append(left, right...), nil + } + + vectors, err := idx.embedAllChunks(ctx, texts, embedChunkSize, embedWithRetry) + if err != nil { + return nil, fmt.Errorf("chunk embedding failed: %w", err) + } + if len(vectors) != len(ids) { + return nil, fmt.Errorf("embedding provider returned %d vectors for %d texts", len(vectors), len(ids)) + } + + // A provider-reported width is authoritative. Providers that learn their + // dimensions on first use are inferred from the first non-empty result. + dims := idx.embedder.Dimensions() + if dims <= 0 { + for _, vec := range vectors { + if len(vec) > 0 { + dims = len(vec) + break + } + } + } + if dims <= 0 { + dims = embeddingDimsOrDefault(idx.embedder) + } + + plan := &preparedVectorPlan{ + repoPrefix: idx.repoPrefix, + dims: dims, + items: make([]graph.VectorCorpusItem, 0, len(vectors)), + embeddedText: len(texts), + } + for i, vec := range vectors { + valid := len(vec) == dims + if valid { + for _, value := range vec { + f := float64(value) + if math.IsNaN(f) || math.IsInf(f, 0) { + valid = false + break + } + } + } + if !valid { + plan.dropped++ + if len(plan.droppedIDs) < 5 { + plan.droppedIDs = append(plan.droppedIDs, ids[i]) + } + continue + } + parentID := collectedChunks[ids[i]] + plan.items = append(plan.items, graph.VectorCorpusItem{ + NodeID: ids[i], + ParentID: parentID, + Vec: append([]float32(nil), vec...), + }) + if parentID != "" { + if plan.chunkMap == nil { + plan.chunkMap = make(map[string]string) + } + plan.chunkMap[ids[i]] = parentID + } + } + if len(plan.items) == 0 { + dropped := plan.dropped + plan.Release() + return nil, fmt.Errorf("all %d embedding vectors were invalid (want width %d)", dropped, dims) + } + return plan, nil +} + +// prepareSearchIndexForPublication preserves the historical text-only +// degradation policy for provider/cap/validation failures. Parent cancellation +// is different: it aborts the index operation and must not publish anything. +func (idx *Indexer) prepareSearchIndexForPublication(ctx context.Context) (*preparedVectorPlan, error) { + plan, err := idx.prepareSearchIndex(ctx) + if err == nil { + return plan, nil + } + idx.lastVectorBuildErr = err + idx.logger.Warn("vector index preparation failed; retaining previous vector publication", zap.Error(err)) + if ctxErr := ctx.Err(); ctxErr != nil { + return nil, ctxErr + } + return nil, nil +} + +// installVectorPlan publishes one complete plan. Expensive embedding happens +// before SerializeVectorUpdate; the shared lane spans only durable replacement +// and the matching process-local publication. +func (idx *Indexer) installVectorPlan(ctx context.Context, durable graph.Store, plan *preparedVectorPlan) error { + if plan == nil { + return nil + } + defer plan.Release() + if ctx == nil { + ctx = context.Background() + } + + sw := idx.swappable() + return sw.SerializeVectorUpdate(func() error { + if err := ctx.Err(); err != nil { + idx.lastVectorBuildErr = err + return err + } + + installer, hasInstaller := durable.(graph.AtomicVectorCorpusInstaller) + vectorSearcher, hasVectorSearcher := durable.(graph.VectorSearcher) + var nativeErr error + if hasInstaller && hasVectorSearcher { + stats, err := installer.ReplaceVectorCorpus(ctx, plan.repoPrefix, plan.dims, plan.items) + if err == nil { + dims := stats.Dims + if dims <= 0 { + dims = plan.dims + } + delegated := search.NewDelegatedVector( + dims, + &vectorSearcherDelegate{s: vectorSearcher}, + stats.VectorCount, + stats.ChunkCount, + ) + sw.ReplaceHybridVector(delegated, idx.embedder) + idx.lastVectorBuildErr = nil + idx.logVectorPublication("durable vector corpus published", plan, stats.VectorCount, stats.ChunkCount, true) + return nil + } + if ctxErr := ctx.Err(); ctxErr != nil { + idx.lastVectorBuildErr = ctxErr + return ctxErr + } + if errors.Is(err, context.Canceled) || errors.Is(err, context.DeadlineExceeded) { + idx.lastVectorBuildErr = err + return err + } + nativeErr = fmt.Errorf("durable vector corpus install failed: %w", err) + } else if hasInstaller != hasVectorSearcher { + nativeErr = fmt.Errorf("durable vector backend exposes an incomplete atomic corpus capability") + } + + // A repository-local heap cannot replace a process-global vector channel: + // it would hide every sibling repository. Preserve the last committed + // delegate in multi-repo mode and surface the degraded install instead. + if idx.repositoryMutationOwner != nil { + if nativeErr == nil { + nativeErr = fmt.Errorf("durable atomic vector corpus installation is unavailable in multi-repository mode") + } + idx.lastVectorBuildErr = nativeErr + idx.logger.Warn("vector publication retained previous global corpus", + zap.String("repo", plan.repoPrefix), zap.Error(nativeErr)) + return nil + } + + heap := search.NewVector(plan.dims) + for _, item := range plan.items { + heap.Add(item.NodeID, item.Vec) + } + if len(plan.chunkMap) > 0 { + chunks := make(map[string]string, len(plan.chunkMap)) + for child, parent := range plan.chunkMap { + chunks[child] = parent + } + heap.SetChunkMap(chunks) + } + sw.ReplaceHybridVector(heap, idx.embedder) + if nativeErr != nil { + idx.lastVectorBuildErr = nativeErr + idx.logger.Warn("durable vector install failed; published process-local fallback", + zap.String("repo", plan.repoPrefix), zap.Error(nativeErr), + zap.String("restart_behavior", "durable corpus remains at its previous generation")) + } else { + idx.lastVectorBuildErr = nil + } + idx.logVectorPublication("process-local vector corpus published", plan, heap.Count(), len(plan.chunkMap), false) + return nil + }) +} + +func (idx *Indexer) logVectorPublication(message string, plan *preparedVectorPlan, vectors, chunks int, delegated bool) { + fields := []zap.Field{ + zap.String("repo", plan.repoPrefix), + zap.Int("vectors", vectors), + zap.Int("chunk_vectors", chunks), + zap.Int("dimensions", plan.dims), + zap.Int("dropped", plan.dropped), + zap.Bool("delegated", delegated), + } + if len(plan.droppedIDs) > 0 { + fields = append(fields, zap.Strings("dropped_sample_ids", plan.droppedIDs)) + } + if acc, ok := idx.embedder.(interface{ TokensUsed() int64 }); ok { + if tokens := acc.TokensUsed(); tokens > 0 { + fields = append(fields, zap.Int64("embed_tokens", tokens)) + } + } + idx.logger.Info(message, fields...) +} + +// restoreDurableVectorBackend republishes an existing complete durable corpus +// without invoking the embedding provider. False means the corpus is absent or +// the store lacks the atomic capability, so the caller may rebuild it. +func (idx *Indexer) restoreDurableVectorBackend(ctx context.Context, durable graph.Store) (bool, error) { + if idx.embedder == nil { + return false, nil + } + installer, ok := durable.(graph.AtomicVectorCorpusInstaller) + if !ok { + return false, nil + } + vectorSearcher, ok := durable.(graph.VectorSearcher) + if !ok { + return false, nil + } + if ctx == nil { + ctx = context.Background() + } + + restored := false + sw := idx.swappable() + err := sw.SerializeVectorUpdate(func() error { + queryDims := idx.embedder.Dimensions() + stats, err := installer.VectorCorpusStatsForRepo(ctx, idx.repoPrefix, queryDims) + if err != nil { + return err + } + if stats.RepositoryVectorCount == 0 || stats.Dims <= 0 { + return nil + } + delegated := search.NewDelegatedVector( + stats.Dims, + &vectorSearcherDelegate{s: vectorSearcher}, + stats.VectorCount, + stats.ChunkCount, + ) + sw.ReplaceHybridVector(delegated, idx.embedder) + idx.lastVectorBuildErr = nil + restored = true + idx.logger.Info("restored durable vector corpus without embedding", + zap.String("repo", idx.repoPrefix), + zap.Int("vectors", stats.VectorCount), + zap.Int("chunk_vectors", stats.ChunkCount), + zap.Int("dimensions", stats.Dims)) + return nil + }) + return restored, err +} + +func (idx *Indexer) buildSearchIndexCtx(ctx context.Context) error { + plan, err := idx.prepareSearchIndexForPublication(ctx) + if err != nil || plan == nil { + return err + } + return idx.installVectorPlan(ctx, idx.graph, plan) +} + +// publishVectorCorpusAfterRepoRemoval runs inside the shared Swappable vector +// update lane after graph/sidecar purge. The empty atomic replacement is both +// a compatibility cleanup for pre-v10 chunk residue and the authoritative +// aggregate-statistics snapshot used for publication. +func (mi *MultiIndexer) publishVectorCorpusAfterRepoRemoval( + ctx context.Context, + repoPrefix string, + sw *search.Swappable, +) error { + installer, ok := mi.graph.(graph.AtomicVectorCorpusInstaller) + if !ok { + return nil + } + + dims := 0 + if mi.embedder != nil { + dims = mi.embedder.Dimensions() + } + if dims <= 0 { + if existing, err := installer.VectorCorpusStats(ctx, 0); err == nil && existing.Dims > 0 { + dims = existing.Dims + } + } + if dims <= 0 { + dims = embeddingDimsOrDefault(mi.embedder) + } + if dims <= 0 { + // Empty replacement still requires a positive validation width. With no + // embedder there is no vector publication, so this sentinel is not used + // to interpret any stored vectors. + dims = 1 + } + + stats, err := installer.ReplaceVectorCorpus(ctx, repoPrefix, dims, nil) + if err != nil { + return fmt.Errorf("remove vector corpus for %s: %w", repoPrefix, err) + } + if mi.embedder == nil || sw == nil { + return nil + } + vectorSearcher, ok := mi.graph.(graph.VectorSearcher) + if !ok { + return fmt.Errorf("publish vector corpus after untrack: store lacks VectorSearcher") + } + publishedDims := stats.Dims + if publishedDims <= 0 { + publishedDims = dims + } + delegated := search.NewDelegatedVector( + publishedDims, + &vectorSearcherDelegate{s: vectorSearcher}, + stats.VectorCount, + stats.ChunkCount, + ) + sw.ReplaceHybridVector(delegated, mi.embedder) + mi.logger.Info("vector corpus refreshed after repository removal", + zap.String("repo", repoPrefix), + zap.Int("vectors", stats.VectorCount), + zap.Int("chunk_vectors", stats.ChunkCount), + zap.Int("dimensions", publishedDims)) + return nil +} diff --git a/internal/indexer/vector_plan_test.go b/internal/indexer/vector_plan_test.go new file mode 100644 index 000000000..c272bbb49 --- /dev/null +++ b/internal/indexer/vector_plan_test.go @@ -0,0 +1,270 @@ +package indexer + +import ( + "context" + "errors" + "math" + "sync" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "go.uber.org/zap" + + "github.com/zzet/gortex/internal/config" + "github.com/zzet/gortex/internal/graph" + "github.com/zzet/gortex/internal/parser" + "github.com/zzet/gortex/internal/search" +) + +type vectorPlanTestStore struct { + graph.Store + mu sync.Mutex + replaceErr error + replaceStats graph.VectorCorpusStats + replaceCalls int + hits []graph.VectorHit +} + +func newVectorPlanTestStore() *vectorPlanTestStore { + return &vectorPlanTestStore{Store: graph.New()} +} + +func (s *vectorPlanTestStore) ReplaceVectorCorpus( + _ context.Context, + _ string, + dims int, + _ []graph.VectorCorpusItem, +) (graph.VectorCorpusStats, error) { + s.mu.Lock() + defer s.mu.Unlock() + s.replaceCalls++ + if s.replaceErr != nil { + return graph.VectorCorpusStats{}, s.replaceErr + } + stats := s.replaceStats + if stats.Dims <= 0 { + stats.Dims = dims + } + return stats, nil +} + +func (s *vectorPlanTestStore) VectorCorpusStats(context.Context, int) (graph.VectorCorpusStats, error) { + return s.replaceStats, nil +} + +func (s *vectorPlanTestStore) VectorCorpusStatsForRepo(context.Context, string, int) (graph.VectorCorpusStats, error) { + return s.replaceStats, nil +} + +func (s *vectorPlanTestStore) UpsertEmbedding(string, []float32) error { return nil } +func (s *vectorPlanTestStore) BulkUpsertEmbeddings([]graph.VectorItem) error { + return nil +} +func (s *vectorPlanTestStore) BuildVectorIndex(int) error { return nil } +func (s *vectorPlanTestStore) SimilarTo(_ []float32, limit int) ([]graph.VectorHit, error) { + s.mu.Lock() + defer s.mu.Unlock() + if limit < len(s.hits) { + return append([]graph.VectorHit(nil), s.hits[:limit]...), nil + } + return append([]graph.VectorHit(nil), s.hits...), nil +} +func (s *vectorPlanTestStore) GetEmbeddings([]string) map[string][]float32 { return nil } + +func vectorPlanTestIndexer(store graph.Store) *Indexer { + idx := New(store, parser.NewRegistry(), config.Default().Index, zap.NewNop()) + idx.SetEmbedder(&poolEmbedder{}) + return idx +} + +func assertVectorPlanReleased(t *testing.T, plan *preparedVectorPlan) { + t.Helper() + assert.Empty(t, plan.repoPrefix) + assert.Zero(t, plan.dims) + assert.Nil(t, plan.items) + assert.Nil(t, plan.chunkMap) + assert.Nil(t, plan.droppedIDs) + assert.Zero(t, plan.dropped) + assert.Zero(t, plan.embeddedText) +} + +func TestPreparedVectorPlanReleaseClearsOwnedState(t *testing.T) { + plan := &preparedVectorPlan{ + repoPrefix: "repo", + dims: 3, + items: []graph.VectorCorpusItem{ + {NodeID: "repo/a", Vec: []float32{1, 0, 0}}, + }, + chunkMap: map[string]string{"repo/a#chunk0": "repo/a"}, + dropped: 1, + droppedIDs: []string{"repo/b"}, + embeddedText: 2, + } + + plan.Release() + assertVectorPlanReleased(t, plan) + plan.Release() // idempotent +} + +func TestInstallVectorPlanPublishesDelegatedBackendAndReleasesPlan(t *testing.T) { + store := newVectorPlanTestStore() + store.replaceStats = graph.VectorCorpusStats{ + VectorCount: 2, ChunkCount: 1, + RepositoryVectorCount: 2, RepositoryChunkCount: 1, + Dims: 3, + } + idx := vectorPlanTestIndexer(store) + plan := &preparedVectorPlan{ + repoPrefix: "repo", + dims: 3, + items: []graph.VectorCorpusItem{ + {NodeID: "repo/a", Vec: []float32{1, 0, 0}}, + {NodeID: "repo/b#chunk0", ParentID: "repo/b", Vec: []float32{0, 1, 0}}, + }, + chunkMap: map[string]string{"repo/b#chunk0": "repo/b"}, + } + + require.NoError(t, idx.installVectorPlan(context.Background(), store, plan)) + assertVectorPlanReleased(t, plan) + require.Equal(t, 1, store.replaceCalls) + + backend, release := idx.swappable().AcquireBackend() + defer release() + hybrid, ok := backend.(*search.HybridBackend) + require.True(t, ok) + require.Equal(t, 2, hybrid.VectorIndex().Count()) + require.True(t, hybrid.VectorIndex().HasChunks()) + require.Zero(t, hybrid.VectorSizeBytes()) +} + +func TestInstallVectorPlanCancellationKeepsPublicationAndReleasesPlan(t *testing.T) { + store := newVectorPlanTestStore() + idx := vectorPlanTestIndexer(store) + before, release := idx.swappable().AcquireBackend() + release() + plan := &preparedVectorPlan{ + repoPrefix: "repo", + dims: 3, + items: []graph.VectorCorpusItem{ + {NodeID: "repo/a", Vec: []float32{1, 0, 0}}, + }, + } + ctx, cancel := context.WithCancel(context.Background()) + cancel() + + err := idx.installVectorPlan(ctx, store, plan) + require.ErrorIs(t, err, context.Canceled) + assertVectorPlanReleased(t, plan) + require.Zero(t, store.replaceCalls) + after, release := idx.swappable().AcquireBackend() + release() + if after != before { + t.Fatal("cancelled install changed the active search backend") + } +} + +func TestInstallVectorPlanNativeFailureUsesPreparedHeapWithoutReembedding(t *testing.T) { + store := newVectorPlanTestStore() + store.replaceErr = errors.New("native install failed") + emb := &poolEmbedder{} + idx := New(store, parser.NewRegistry(), config.Default().Index, zap.NewNop()) + idx.SetEmbedder(emb) + plan := &preparedVectorPlan{ + repoPrefix: "repo", + dims: 3, + items: []graph.VectorCorpusItem{ + {NodeID: "repo/a", Vec: []float32{1, 0, 0}}, + }, + } + + require.NoError(t, idx.installVectorPlan(context.Background(), store, plan)) + assertVectorPlanReleased(t, plan) + require.Zero(t, emb.calls, "heap fallback must reuse the plan without embedding") + require.ErrorIs(t, idx.LastVectorBuildError(), store.replaceErr) + + backend, release := idx.swappable().AcquireBackend() + defer release() + hybrid, ok := backend.(*search.HybridBackend) + require.True(t, ok) + require.Equal(t, 1, hybrid.VectorIndex().Count()) + require.Greater(t, hybrid.VectorSizeBytes(), uint64(0)) +} + +func TestInstallVectorPlanMultiRepoFailurePreservesGlobalPublication(t *testing.T) { + store := newVectorPlanTestStore() + store.replaceStats = graph.VectorCorpusStats{VectorCount: 2, ChunkCount: 1, Dims: 3} + idx := vectorPlanTestIndexer(store) + initial := &preparedVectorPlan{ + repoPrefix: "A", + dims: 3, + items: []graph.VectorCorpusItem{ + {NodeID: "A/a", Vec: []float32{1, 0, 0}}, + }, + } + require.NoError(t, idx.installVectorPlan(context.Background(), store, initial)) + before, release := idx.swappable().AcquireBackend() + release() + + idx.repositoryMutationOwner = &MultiIndexer{} + store.replaceErr = errors.New("repo B install failed") + plan := &preparedVectorPlan{ + repoPrefix: "B", + dims: 3, + items: []graph.VectorCorpusItem{ + {NodeID: "B/b", Vec: []float32{0, 1, 0}}, + }, + } + require.NoError(t, idx.installVectorPlan(context.Background(), store, plan)) + assertVectorPlanReleased(t, plan) + require.ErrorIs(t, idx.LastVectorBuildError(), store.replaceErr) + after, release := idx.swappable().AcquireBackend() + release() + if after != before { + t.Fatal("a repository-local fallback replaced the shared global vector publication") + } +} + +type vectorPlanResultEmbedder struct { + vectors [][]float32 +} + +func (e *vectorPlanResultEmbedder) Embed(context.Context, string) ([]float32, error) { + if len(e.vectors) == 0 { + return nil, nil + } + return e.vectors[0], nil +} +func (e *vectorPlanResultEmbedder) EmbedBatch(context.Context, []string) ([][]float32, error) { + return e.vectors, nil +} +func (e *vectorPlanResultEmbedder) Dimensions() int { return 3 } +func (e *vectorPlanResultEmbedder) Close() error { return nil } + +func TestPrepareVectorPlanRejectsCardinalityAndNonFiniteResults(t *testing.T) { + nodes := []*graph.Node{{ + ID: "repo/a.go::Alpha", Kind: graph.KindFunction, Name: "Alpha", + FilePath: "repo/a.go", RepoPrefix: "repo", Language: "go", + }} + tests := []struct { + name string + vectors [][]float32 + want string + }{ + {name: "missing", vectors: nil, want: "returned 0 vectors for 1 texts"}, + {name: "extra", vectors: [][]float32{{1, 0, 0}, {0, 1, 0}}, want: "returned 2 vectors for 1 texts"}, + {name: "nan", vectors: [][]float32{{float32(math.NaN()), 0, 0}}, want: "all 1 embedding vectors were invalid"}, + {name: "infinity", vectors: [][]float32{{float32(math.Inf(1)), 0, 0}}, want: "all 1 embedding vectors were invalid"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + idx := New(graph.New(), parser.NewRegistry(), config.Default().Index, zap.NewNop()) + idx.SetRepoPrefix("repo") + idx.SetEmbedder(&vectorPlanResultEmbedder{vectors: tt.vectors}) + plan, err := idx.prepareVectorPlan(context.Background(), nodes) + require.Error(t, err) + require.Contains(t, err.Error(), tt.want) + require.Nil(t, plan) + }) + } +} diff --git a/internal/indexer/watcher_storm_batch_test.go b/internal/indexer/watcher_storm_batch_test.go index 8773e4509..8df54a7ed 100644 --- a/internal/indexer/watcher_storm_batch_test.go +++ b/internal/indexer/watcher_storm_batch_test.go @@ -533,10 +533,6 @@ func TestMultiWatcherThreeChunksRunEveryCatchupTailOnce(t *testing.T) { require.Len(t, provider.batches, 1) require.Len(t, provider.batches[0], len(paths)) - passes, resolved, dropped := idx.AffectedByCounts() - require.Equal(t, int64(1), passes) - require.Equal(t, int64(1), resolved) - require.Zero(t, dropped) require.Equal(t, fnNodeID(t, g, "repo/target.go", "Target"), callTargetFrom(t, g, callerID), "the one affected-by catch-up must preserve the caller binding") } diff --git a/internal/indexer/zzbench_backends_test.go b/internal/indexer/zzbench_backends_test.go index d9ef9e71e..7cac5125b 100644 --- a/internal/indexer/zzbench_backends_test.go +++ b/internal/indexer/zzbench_backends_test.go @@ -24,10 +24,11 @@ import ( ) // TestBackendBench cold-indexes GORTEX_BENCH_ROOT through the full indexer -// pipeline into the backend named by GORTEX_BENCH_BACKEND (memory | sqlite), -// then runs a fixed query workload. Reports cold-index time, graph size, -// process RSS, and query throughput so the sqlite backend can be compared -// head-to-head with the in-memory baseline on real repositories. +// pipeline into the sink named by GORTEX_BENCH_BACKEND, then runs a fixed +// query workload. Reports cold-index time, graph size, process RSS, and query +// throughput. "sqlite" is the shipping store; "memory" is not a backend any +// more — that arm drains into the indexer's cold-index staging shadow and +// stands in as the ceiling the sqlite path is measured against. // // GORTEX_BENCH_ROOT=/path/to/gortex \ // GORTEX_BENCH_BACKEND=sqlite \ diff --git a/internal/llm/agent/compact.go b/internal/llm/agent/compact.go index e985285a4..7b05a539d 100644 --- a/internal/llm/agent/compact.go +++ b/internal/llm/agent/compact.go @@ -51,14 +51,6 @@ type RollingCompactor struct { summarizer llm.Provider - // sync forces synchronous-but-throttled compaction. When the agent's - // Run deadline is too short to safely outlive a background summarizer - // round (so the derived ctx would routinely cancel summaries mid-flight), - // the compactor summarizes inline on the compaction turn instead of - // spawning a goroutine. The agent loop already has a bounded step count, - // so a synchronous summarize is bounded. - sync bool - // pending guards the in-flight / completed async summary state. pendingMu sync.Mutex inflight bool // a background summarizer goroutine is running @@ -104,18 +96,6 @@ func WithCompactor(c *RollingCompactor) AgentOption { return func(a *Agent) { a.compactor = c } } -// WithSyncCompaction is a test/throttling seam that forces synchronous -// (inline) summarization. Used when the run deadline is too short for the -// async path. It is exposed as an option so callers that already know their -// per-call deadline is tight can opt in without reflection. -func WithSyncCompaction() AgentOption { - return func(a *Agent) { - if a.compactor != nil { - a.compactor.sync = true - } - } -} - // enabled reports whether the compactor can actually compact (it has a // summarizer). A nil receiver or nil summarizer disables compaction. func (c *RollingCompactor) enabled() bool { @@ -209,9 +189,8 @@ func estimateConvTokens(conv []llm.Message, lastUsage llm.TokenUsage) int { // site checks summCtx.Err()==nil before applying. // - The summarizer's own token usage is attributed: it is folded into the // passed *usage under mu, so background spend is never unaccounted. -// - When sync==true (deadline too short for the async path) the summarize -// runs inline; otherwise a single background goroutine is spawned and the -// result is spliced on the next turn that finds it ready. +// - A single background goroutine is spawned; its result is spliced on the +// next turn that finds it ready. func (c *RollingCompactor) maybeCompact(runCtx context.Context, conv []llm.Message, sizeTokens int, usage *llm.TokenUsage, mu *sync.Mutex) (compacted []llm.Message, didCompact bool) { if !c.enabled() { return conv, false @@ -227,28 +206,13 @@ func (c *RollingCompactor) maybeCompact(runCtx context.Context, conv []llm.Messa return conv, false } - frozen, compress, active := partitionZones(conv, c.ActiveRounds) + _, compress, _ := partitionZones(conv, c.ActiveRounds) if countRounds(compress) < 2 { return conv, false // nothing eligible to fold yet } - // Synchronous path: summarize inline and splice now. - if c.sync { - summCtx, cancel := context.WithCancel(runCtx) - defer cancel() - summary, u, err := c.summarizeZone(summCtx, compress) - // Attribute the summarizer's spend regardless of splice outcome. - mu.Lock() - usage.Add(u) - mu.Unlock() - if err != nil || summCtx.Err() != nil || strings.TrimSpace(summary) == "" { - return conv, false - } - return c.spliceSummary(conv, frozen, active, summary), true - } - - // Async path: spawn a single background summarizer. The next turn that - // finds the result ready splices it (above). + // Spawn a single background summarizer. The next turn that finds the result + // ready splices it (above). c.spawnSummarize(runCtx, compress, usage, mu) return conv, false } diff --git a/internal/llm/agent/compact_test.go b/internal/llm/agent/compact_test.go index 250ec7999..d2f5344a7 100644 --- a/internal/llm/agent/compact_test.go +++ b/internal/llm/agent/compact_test.go @@ -129,16 +129,16 @@ func TestPartitionZones_ShortConversation(t *testing.T) { } } -func TestMaybeCompact_SyncReducesBelowLowWater(t *testing.T) { +func TestMaybeCompact_AsyncReducesBelowLowWater(t *testing.T) { // A big conversation: 12 rounds of long filler so it crosses the trigger. filler := strings.Repeat("lorem ipsum dolor sit amet ", 40) conv := buildConv(12, filler) - c := NewRollingCompactor(&stubSummarizer{ + stub := &stubSummarizer{ summary: "compact summary of earlier work", usage: llm.TokenUsage{InputTokens: 500, OutputTokens: 20}, - }, 4, 3000, 100) - c.sync = true + } + c := NewRollingCompactor(stub, 4, 3000, 100) pre := estimateConvTokens(conv, llm.TokenUsage{}) if pre < c.CompressTriggerTokens { @@ -147,9 +147,18 @@ func TestMaybeCompact_SyncReducesBelowLowWater(t *testing.T) { var usage llm.TokenUsage var mu sync.Mutex + first, did := c.maybeCompact(context.Background(), conv, pre, &usage, &mu) + if did { + t.Fatal("first async call compacted synchronously") + } + if len(first) != len(conv) { + t.Fatal("first async call mutated the conversation") + } + c.wait() + out, did := c.maybeCompact(context.Background(), conv, pre, &usage, &mu) if !did { - t.Fatal("maybeCompact did not compact a conversation over the high-water mark") + t.Fatal("ready asynchronous summary was not spliced") } post := estimateConvTokens(out, llm.TokenUsage{}) if post >= pre { @@ -228,7 +237,7 @@ func TestMaybeCompact_LateSummaryDroppedOnCancel(t *testing.T) { block: block, started: started, } - c := NewRollingCompactor(stub, 4, 3000, 100) // async (sync=false) + c := NewRollingCompactor(stub, 4, 3000, 100) // async path ctx, cancel := context.WithCancel(context.Background()) var usage llm.TokenUsage @@ -341,7 +350,6 @@ func TestRun_CompactionWiredIntoRun(t *testing.T) { usage: llm.TokenUsage{InputTokens: 7, OutputTokens: 1}, } comp := NewRollingCompactor(summ, 2, 200, 50) - comp.sync = true // deterministic in-test compaction ag, err := New(tp, []Tool{{ Name: "noop", diff --git a/internal/localizationauth/receipt.go b/internal/localizationauth/receipt.go index 5e6c980c3..cb954e470 100644 --- a/internal/localizationauth/receipt.go +++ b/internal/localizationauth/receipt.go @@ -158,14 +158,6 @@ func Consume(token string) (Receipt, bool) { return envelope.Receipt, true } -// Discard removes a pending receipt without exposing its contents. -func Discard(token string) { - path, _, ok := receiptPath(token) - if ok { - _ = os.Remove(path) - } -} - func validReceipt(receipt Receipt) bool { return receipt.FinalResponse != "" && len(receipt.FinalResponse) <= maxFinalResponseSize && receipt.ContractVersion >= contractVersionV2 diff --git a/internal/mcp/edge_audit_integrity.go b/internal/mcp/edge_audit_integrity.go new file mode 100644 index 000000000..2c335a19b --- /dev/null +++ b/internal/mcp/edge_audit_integrity.go @@ -0,0 +1,74 @@ +package mcp + +import ( + "context" + + "github.com/zzet/gortex/internal/graph" +) + +type edgeAuditIntegritySinceOpen struct { + Status string `json:"status"` + Totals graph.StructuralIntegrityTotals `json:"totals"` + Attribution []graph.StructuralIntegrityAttribution `json:"attribution,omitempty"` + Samples []graph.StructuralIntegritySample `json:"samples,omitempty"` + AttributionOverflow uint64 `json:"attribution_overflow,omitempty"` + SampleOverflow uint64 `json:"sample_overflow,omitempty"` + RepoTotalsOverflow uint64 `json:"repo_totals_overflow,omitempty"` + TotalsLowerBound bool `json:"totals_lower_bound,omitempty"` + SamplesTruncated bool `json:"samples_truncated,omitempty"` +} + +type edgeAuditGraphIntegrity struct { + SinceOpen edgeAuditIntegritySinceOpen `json:"since_open"` + Persisted graph.StructuralIntegrityExactAudit `json:"persisted"` +} + +func (s *Server) edgeAuditGraphIntegrity(ctx context.Context, sampleLimit int, scopeActive bool, repoPrefixes []string) (edgeAuditGraphIntegrity, error) { + out := edgeAuditGraphIntegrity{ + SinceOpen: edgeAuditIntegritySinceOpen{Status: graph.StructuralAuditUnsupported}, + Persisted: graph.StructuralIntegrityExactAudit{Status: graph.StructuralAuditUnsupported}, + } + if err := ctx.Err(); err != nil { + return out, err + } + if sampleLimit <= 0 { + sampleLimit = 10 + } + if sampleLimit > 100 { + sampleLimit = 100 + } + if snapshotter, ok := s.graph.(graph.StructuralIntegritySnapshotter); ok { + snapshot := snapshotter.StructuralIntegritySnapshot(graph.StructuralIntegritySnapshotOptions{ + RepoPrefixes: repoPrefixes, + RepoScopeActive: scopeActive, + IncludeAttribution: true, + IncludeSamples: true, + }) + out.SinceOpen = edgeAuditIntegritySinceOpen{ + Status: graph.StructuralAuditSupported, + Totals: snapshot.Totals, + Attribution: snapshot.Attribution, + Samples: snapshot.Samples, + AttributionOverflow: snapshot.AttributionOverflow, + SampleOverflow: snapshot.SampleOverflow, + RepoTotalsOverflow: snapshot.RepoTotalsOverflow, + TotalsLowerBound: snapshot.TotalsLowerBound, + } + if len(out.SinceOpen.Samples) > sampleLimit { + out.SinceOpen.Samples = out.SinceOpen.Samples[:sampleLimit] + out.SinceOpen.SamplesTruncated = true + } + } + if auditor, ok := s.graph.(graph.StructuralIntegrityAuditor); ok { + audit, err := auditor.AuditStructuralIntegrity(ctx, graph.StructuralIntegrityAuditOptions{ + RepoPrefixes: repoPrefixes, + RepoScopeActive: scopeActive, + SampleLimit: sampleLimit, + }) + if err != nil { + return out, err + } + out.Persisted = audit + } + return out, nil +} diff --git a/internal/mcp/errors.go b/internal/mcp/errors.go index df1b8b119..3216931c8 100644 --- a/internal/mcp/errors.go +++ b/internal/mcp/errors.go @@ -2,7 +2,6 @@ package mcp import ( "encoding/json" - "errors" "fmt" "github.com/mark3labs/mcp-go/mcp" @@ -41,12 +40,6 @@ const ( // project slug doesn't (e.g. monorepo missing the named project). ErrCodeProjectUnknown ErrorCode = "project_unknown" - // ErrCodeCrossWorkspaceDenied — the source workspace's - // `cross_workspace_deps` doesn't declare the target workspace - // (or the import path doesn't match a declared module). The - // query is refused at the matcher / resolver boundary. - ErrCodeCrossWorkspaceDenied ErrorCode = "cross_workspace_denied" - // ErrCodeRepoNotTracked — the cwd wasn't found in any tracked // repo's root tree. Used by the daemon's MCP front-door. ErrCodeRepoNotTracked ErrorCode = "repo_not_tracked" @@ -164,69 +157,3 @@ func newStructuredErrorResult(err StructuredError, explicitRetriable bool) *mcp. res.IsError = true return res } - -// Common constructors for the codes above so handlers can call -// `mcp.WorkspaceUnknownError(slug)` instead of building structs by -// hand. - -func WorkspaceUnknownError(workspace string) *mcp.CallToolResult { - return NewStructuredErrorResult(StructuredError{ - ErrorCode: ErrCodeWorkspaceUnknown, - Message: fmt.Sprintf("workspace %q is not known to this server", workspace), - Retriable: false, - Data: map[string]any{"workspace": workspace}, - }) -} - -func ProjectUnknownError(workspace, project string) *mcp.CallToolResult { - return NewStructuredErrorResult(StructuredError{ - ErrorCode: ErrCodeProjectUnknown, - Message: fmt.Sprintf("project %q does not exist in workspace %q", project, workspace), - Retriable: false, - Data: map[string]any{"workspace": workspace, "project": project}, - }) -} - -func CrossWorkspaceDeniedError(source, target, importPath string) *mcp.CallToolResult { - msg := fmt.Sprintf("cross-workspace access from %q to %q is not declared in cross_workspace_deps", source, target) - if importPath != "" { - msg = fmt.Sprintf("%s (import path %q)", msg, importPath) - } - return NewStructuredErrorResult(StructuredError{ - ErrorCode: ErrCodeCrossWorkspaceDenied, - Message: msg, - Retriable: false, - Data: map[string]any{ - "source_workspace": source, - "target_workspace": target, - "import_path": importPath, - }, - }) -} - -// AsStructuredError unwraps a typed error from the package's known -// sentinel set, returning a CallToolResult on hit. Returns (nil, -// false) when err isn't one of the recognised sentinels — caller -// falls back to its own error handling. -func AsStructuredError(err error) (*mcp.CallToolResult, bool) { - if err == nil { - return nil, false - } - // Future: extend with errors.Is checks for daemon.Err* sentinels - // once the daemon's errors are imported here. For now we only - // match generic shapes used by handlers. - switch { - case errors.Is(err, errInvalidArgument): - return NewStructuredErrorResult(StructuredError{ - ErrorCode: ErrCodeInvalidArgument, - Message: err.Error(), - }), true - } - return nil, false -} - -// errInvalidArgument is the canonical sentinel a tool can return when -// its args fail validation; AsStructuredError converts it to the -// structured form. Wrapping (`fmt.Errorf("%w: ...", errInvalidArgument)`) -// is supported via errors.Is. -var errInvalidArgument = errors.New("invalid argument") diff --git a/internal/mcp/facade_registry.go b/internal/mcp/facade_registry.go index 857d17a85..4924a0cfe 100644 --- a/internal/mcp/facade_registry.go +++ b/internal/mcp/facade_registry.go @@ -135,10 +135,6 @@ func (r *facadeRegistry) availableOperations(facade string) []facadeOperationSpe return out } -func (r *facadeRegistry) mapsLegacy(name string) bool { - return r != nil && len(r.byLegacy[name]) > 0 -} - func facadeToolNames() []string { return []string{ "analyze", "ask", "capabilities", "change", "edit", "explore", @@ -148,22 +144,10 @@ func facadeToolNames() []string { } } -// FacadeToolNames returns the complete stable facade-v1 tool roster. The -// returned slice is a fresh copy so CLI/help callers cannot mutate the server's -// canonical surface. -func FacadeToolNames() []string { - return append([]string(nil), facadeToolNames()...) -} - // IsFacadeToolName reports whether name belongs to the public facade-v1 // surface. func IsFacadeToolName(name string) bool { return isFacadeToolName(name) } -// IsDedicatedFacadeToolName reports whether name exists only on facade-v1. -// Shared names such as analyze/explore/review/ask retain legacy meanings and -// are deliberately excluded. -func IsDedicatedFacadeToolName(name string) bool { return isDedicatedFacadeTool(name) } - // PublicOperationForLegacy resolves an implementation-era tool name to the // compact public domain and operation used in user-facing migration guidance. // When one legacy handler has deliberate effect splits, the first safe public diff --git a/internal/mcp/facade_tools.go b/internal/mcp/facade_tools.go index bc4f6403d..3b8d3e640 100644 --- a/internal/mcp/facade_tools.go +++ b/internal/mcp/facade_tools.go @@ -1586,10 +1586,6 @@ const ( facadeOutcomeEmptyResult = "empty_result" ) -func facadeTelemetryDimension(spec facadeOperationSpec) string { - return boundedFacadeTelemetryDimension(spec.Facade, spec.Operation) -} - // boundedFacadeTelemetryDimension joins fixed, low-cardinality tokens and // deterministically folds long combinations under telemetry's 32-byte guard. // Callers must pass registry values or fixed sentinels, never request values. diff --git a/internal/mcp/facade_tools_test.go b/internal/mcp/facade_tools_test.go index 00f4d2b83..bcb2d3436 100644 --- a/internal/mcp/facade_tools_test.go +++ b/internal/mcp/facade_tools_test.go @@ -29,7 +29,7 @@ func TestFacadeRegistryCoversRegisteredLegacyCatalog(t *testing.T) { if isFacadeToolName(descriptor.Name) { continue } - if !srv.facades.mapsLegacy(descriptor.Name) { + if len(srv.facades.byLegacy[descriptor.Name]) == 0 { missing = append(missing, descriptor.Name) } } @@ -1449,7 +1449,7 @@ func TestFacadeDispatchRecordsOperationTelemetry(t *testing.T) { require.Equal(t, 1, latencyCounts["analyze.coverage"]) require.Equal(t, 1, latencyCounts["analyze.unknown"]) - long := facadeTelemetryDimension(facadeOperationSpec{Facade: "session", Operation: "unsubscribe_workspace_readiness"}) + long := boundedFacadeTelemetryDimension("session", "unsubscribe_workspace_readiness") require.LessOrEqual(t, len(long), 32) } diff --git a/internal/mcp/guard_rules.go b/internal/mcp/guard_rules.go index 5df5258bc..253474144 100644 --- a/internal/mcp/guard_rules.go +++ b/internal/mcp/guard_rules.go @@ -163,8 +163,8 @@ func (s *Server) anyGuardRulesConfigured() bool { // guardsFamily adapts the per-repo guard resolution to // analysis.RuleFamily, so the change-contract pipeline scopes rules to -// each changed symbol's repo exactly like check_guards does. It replaces -// analysis.GuardsFamily, which can only carry one flat rule list. +// each changed symbol's repo exactly like check_guards does. A family +// carrying one flat rule list cannot express that per-repo scoping. type guardsFamily struct{ srv *Server } func (f guardsFamily) Name() string { return "guards" } diff --git a/internal/mcp/overlay_temporal_options_test.go b/internal/mcp/overlay_temporal_options_test.go new file mode 100644 index 000000000..1366188d9 --- /dev/null +++ b/internal/mcp/overlay_temporal_options_test.go @@ -0,0 +1,73 @@ +package mcp + +import ( + "os" + "path/filepath" + "testing" + + "go.uber.org/zap" + + "github.com/zzet/gortex/internal/config" + "github.com/zzet/gortex/internal/daemon" + "github.com/zzet/gortex/internal/graph" + "github.com/zzet/gortex/internal/indexer" + "github.com/zzet/gortex/internal/parser" + "github.com/zzet/gortex/internal/parser/languages" +) + +func TestConstructOverlayLayerUsesRepositoryTemporalOptions(t *testing.T) { + t.Setenv(config.LocalTemporalOptInEnv, "true") + root := t.TempDir() + allowlistDir := filepath.Join(root, ".gortex") + if err := os.MkdirAll(allowlistDir, 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(allowlistDir, "temporal-allowlist.yaml"), []byte("env_helpers:\n - OverlayEnvHelper\n"), 0o600); err != nil { + t.Fatal(err) + } + + reg := parser.NewRegistry() + languages.RegisterAll(reg) + g := graph.New() + idx := indexer.New(g, reg, config.IndexConfig{}, zap.NewNop()) + idx.SetRootPath(root) + defer idx.Close() + srv := &Server{graph: g, indexer: idx, logger: zap.NewNop()} + + source := `package sample +import "go.temporal.io/sdk/workflow" +func Run(ctx workflow.Context) { + name := OverlayEnvHelper("ACTIVITY", "MyActivity") + workflow.ExecuteActivity(ctx, name) +} +` + layer, paths, err := srv.constructOverlayLayer([]daemon.OverlayFile{{Path: "overlay.go", Content: source}}) + if err != nil { + t.Fatal(err) + } + if layer == nil || len(paths) != 1 || paths[0] != "overlay.go" { + t.Fatalf("overlay layer/paths = %#v, %#v", layer, paths) + } + view := graph.NewOverlaidView(g, layer) + var found bool + for _, node := range view.GetFileNodes("overlay.go") { + if node == nil || node.Kind != graph.KindFunction || node.Name != "Run" { + continue + } + for _, edge := range view.GetOutEdges(node.ID) { + if edge == nil || edge.Meta == nil || edge.Meta["via"] != "temporal.stub" { + continue + } + found = true + if got := edge.Meta["temporal_env_source"]; got != "allowlist" { + t.Fatalf("overlay Temporal source = %#v", got) + } + if got := edge.Meta["temporal_name"]; got != "MyActivity" { + t.Fatalf("overlay Temporal target = %#v", got) + } + } + } + if !found { + t.Fatal("overlay Temporal dispatch edge not found") + } +} diff --git a/internal/mcp/overlay_view.go b/internal/mcp/overlay_view.go index 0f3df32ec..831f616d4 100644 --- a/internal/mcp/overlay_view.go +++ b/internal/mcp/overlay_view.go @@ -315,10 +315,6 @@ func (s *Server) constructOverlayLayer(files []daemon.OverlayFile) (*graph.Overl if !ok { continue } - ext, _ := reg.GetByLanguage(lang) - if ext == nil { - continue - } root := idx.RootPath() relPath := graphPath if idx.RepoPrefix() != "" { @@ -328,7 +324,7 @@ func (s *Server) constructOverlayLayer(files []daemon.OverlayFile) (*graph.Overl relPath = filepath.ToSlash(r) } } - result, err := ext.Extract(relPath, []byte(ov.Content)) + result, err := idx.ExtractBuffer(lang, relPath, []byte(ov.Content)) if err != nil { return nil, nil, fmt.Errorf("overlay parse %s: %w", ov.Path, err) } diff --git a/internal/mcp/tools_analysis.go b/internal/mcp/tools_analysis.go index bdb93af13..8e8351cfb 100644 --- a/internal/mcp/tools_analysis.go +++ b/internal/mcp/tools_analysis.go @@ -691,20 +691,13 @@ func impactMetaString(m map[string]any, key string) string { return v } -// CommunityCoupling describes the coupling between two communities. -type CommunityCoupling struct { - CommunityA string `json:"community_a"` - CommunityB string `json:"community_b"` - LabelA string `json:"label_a"` - LabelB string `json:"label_b"` - CouplingScore float64 `json:"coupling_score"` - TightlyCoupled bool `json:"tightly_coupled"` -} - // CrossCommunityWarning describes cross-community impact. +// +// It carries the affected community names and nothing else on purpose: the +// mandatory impact path must never perform a graph-wide coupling scan, so +// there is deliberately no field here for per-pair coupling scores. type CrossCommunityWarning struct { - AffectedCommunities []string `json:"affected_communities"` - Couplings []CommunityCoupling `json:"couplings,omitempty"` + AffectedCommunities []string `json:"affected_communities"` } // computeCrossCommunityWarning names the communities a change reaches. diff --git a/internal/mcp/tools_analysis_warning_test.go b/internal/mcp/tools_analysis_warning_test.go index 33255006d..cca7ea2fd 100644 --- a/internal/mcp/tools_analysis_warning_test.go +++ b/internal/mcp/tools_analysis_warning_test.go @@ -18,9 +18,6 @@ func TestComputeCrossCommunityWarningDoesNotScanGraph(t *testing.T) { if !reflect.DeepEqual(warning.AffectedCommunities, affected) { t.Fatalf("affected communities = %v, want %v", warning.AffectedCommunities, affected) } - if len(warning.Couplings) != 0 { - t.Fatalf("mandatory impact path computed %d coupling(s); want no graph-wide coupling scan", len(warning.Couplings)) - } } func TestImpactCompleteRejectsDispatchLowerBound(t *testing.T) { diff --git a/internal/mcp/tools_analyze_edge_audit.go b/internal/mcp/tools_analyze_edge_audit.go index 81e3b8b1f..1cf9c2436 100644 --- a/internal/mcp/tools_analyze_edge_audit.go +++ b/internal/mcp/tools_analyze_edge_audit.go @@ -48,9 +48,10 @@ func (s *Server) handleAnalyzeEdgeAudit(ctx context.Context, req mcp.CallToolReq edgeTiers := map[string]int{} callTiers := map[string]int{} - inCalls := map[string][]string{} // target → caller IDs - implemented := map[string]bool{} // interface ID → has an implementor - var weakCalls []string // text-matched "from -> to" + inCalls := map[string][]string{} // target → caller IDs + implemented := map[string]bool{} // interface ID → has an implementor + visibleRepos := map[string]struct{}{} + var weakCalls []string // text-matched "from -> to" // When the request narrows scope (workspace-bound session or repo // allow-set), drop edges/nodes outside it so every count map and @@ -85,6 +86,9 @@ func (s *Server) handleAnalyzeEdgeAudit(ctx context.Context, req mcp.CallToolReq if scoped && !s.analyzeNodeVisible(ctx, n) { continue } + if scoped && n.RepoPrefix != "" { + visibleRepos[n.RepoPrefix] = struct{}{} + } switch n.Kind { case graph.KindInterface: if !implemented[n.ID] { @@ -126,6 +130,19 @@ func (s *Server) handleAnalyzeEdgeAudit(ctx context.Context, req mcp.CallToolReq highPct = float64(highConf) * 100 / float64(totalCalls) } + var scopedRepos []string + if scoped { + scopedRepos = make([]string, 0, len(visibleRepos)) + for repo := range visibleRepos { + scopedRepos = append(scopedRepos, repo) + } + sort.Strings(scopedRepos) + } + integrity, err := s.edgeAuditGraphIntegrity(ctx, sample, scoped, scopedRepos) + if err != nil { + return nil, err + } + payload := map[string]any{ "edge_tiers": edgeTiers, "call_tiers": callTiers, @@ -137,6 +154,7 @@ func (s *Server) handleAnalyzeEdgeAudit(ctx context.Context, req mcp.CallToolReq "unimplemented_interfaces": auditBucket(unimplemented, sample), "test_only_targets": auditBucket(testOnly, sample), "weak_call_edges": auditBucket(weakCalls, sample), + "graph_integrity": integrity, } if s.isGCX(ctx, req) { @@ -147,6 +165,9 @@ func (s *Server) handleAnalyzeEdgeAudit(ctx context.Context, req mcp.CallToolReq fmt.Fprintf(&b, "edges=%d calls=%d high_conf=%.1f%%\n", totalEdges, totalCalls, highPct) fmt.Fprintf(&b, "unimplemented_interfaces=%d test_only_targets=%d weak_call_edges=%d\n", len(unimplemented), len(testOnly), len(weakCalls)) + fmt.Fprintf(&b, "graph_integrity since_open_status=%s writes=%d reads=%d persisted_status=%s persisted_rows=%d\n", + integrity.SinceOpen.Status, integrity.SinceOpen.Totals.WriteRejected, + integrity.SinceOpen.Totals.ReadSuppressed, integrity.Persisted.Status, integrity.Persisted.TotalRows) return mcp.NewToolResultText(b.String()), nil } return s.respondJSONOrTOON(ctx, req, payload) diff --git a/internal/mcp/tools_analyze_edge_audit_integrity_test.go b/internal/mcp/tools_analyze_edge_audit_integrity_test.go new file mode 100644 index 000000000..f1aad26c8 --- /dev/null +++ b/internal/mcp/tools_analyze_edge_audit_integrity_test.go @@ -0,0 +1,165 @@ +package mcp + +import ( + "context" + "encoding/json" + "errors" + "strings" + "testing" + + mcplib "github.com/mark3labs/mcp-go/mcp" + "github.com/zzet/gortex/internal/graph" +) + +func edgeAuditIntegrityRequest(args map[string]any) mcplib.CallToolRequest { + req := mcplib.CallToolRequest{} + req.Params.Name = "analyze" + req.Params.Arguments = args + return req +} + +func edgeAuditIntegrityText(t *testing.T, srv *Server, ctx context.Context, args map[string]any) string { + t.Helper() + res, err := srv.handleAnalyzeEdgeAudit(ctx, edgeAuditIntegrityRequest(args)) + if err != nil { + t.Fatal(err) + } + if res.IsError { + t.Fatalf("edge_audit returned tool error: %+v", res.Content) + } + return res.Content[0].(mcplib.TextContent).Text +} + +func addRejectedIntegrityAttempt(store graph.Store, repo, suffix string) { + source := repo + "/source-" + suffix + store.AddNode(&graph.Node{ID: source, Name: source, Kind: graph.KindFunction, RepoPrefix: repo, FilePath: repo + "/source.go"}) + store.AddEdge(&graph.Edge{ + From: source, To: repo + "/target#param:" + suffix, + Kind: graph.EdgeImplements, Origin: "LSP_DISPATCH", FilePath: repo + "/source.go", Line: 5, + }) +} + +func TestAnalyzeEdgeAuditGraphIntegrityJSONCompactAndGCX(t *testing.T) { + srv, _ := setupTestServer(t) + addRejectedIntegrityAttempt(srv.graph, "repo-a", "x") + + text := edgeAuditIntegrityText(t, srv, context.Background(), map[string]any{"limit": 10.0}) + var payload map[string]any + if err := json.Unmarshal([]byte(text), &payload); err != nil { + t.Fatalf("decode edge_audit JSON: %v\n%s", err, text) + } + integrity := payload["graph_integrity"].(map[string]any) + sinceOpen := integrity["since_open"].(map[string]any) + if sinceOpen["status"] != graph.StructuralAuditSupported { + t.Fatalf("since-open capability status: %+v", sinceOpen) + } + totals := sinceOpen["totals"].(map[string]any) + if totals["write_rejected"] != float64(1) { + t.Fatalf("missing rejection total: %+v", totals) + } + samples := sinceOpen["samples"].([]any) + if len(samples) != 1 || samples[0].(map[string]any)["from"] != "repo-a/source-x" { + t.Fatalf("explicit audit must expose its attributed diagnostic sample: %+v", samples) + } + persisted := integrity["persisted"].(map[string]any) + if persisted["status"] != graph.StructuralAuditUnsupported { + t.Fatalf("memory graph must distinguish unsupported persisted audit: %+v", persisted) + } + + compact := edgeAuditIntegrityText(t, srv, context.Background(), map[string]any{"compact": true}) + for _, want := range []string{"graph_integrity", "since_open_status=supported", "writes=1", "persisted_status=unsupported"} { + if !strings.Contains(compact, want) { + t.Fatalf("compact output missing %q: %s", want, compact) + } + } + + gcx := edgeAuditIntegrityText(t, srv, context.Background(), map[string]any{"format": "gcx"}) + for _, want := range []string{"GCX1", "analyze.edge_audit", "graph_integrity", "write_rejected", "persisted"} { + if !strings.Contains(gcx, want) { + t.Fatalf("GCX output missing %q: %s", want, gcx) + } + } +} + +type scopedIntegrityAuditStore struct { + *graph.Graph + exactSamples []graph.StructuralIntegrityExactSample +} + +func (s *scopedIntegrityAuditStore) AuditStructuralIntegrity(ctx context.Context, opts graph.StructuralIntegrityAuditOptions) (graph.StructuralIntegrityExactAudit, error) { + out := graph.StructuralIntegrityExactAudit{Status: graph.StructuralAuditSupported} + if err := ctx.Err(); err != nil { + return out, err + } + allowed := make(map[string]struct{}, len(opts.RepoPrefixes)) + for _, repo := range opts.RepoPrefixes { + allowed[repo] = struct{}{} + } + for _, sample := range s.exactSamples { + if opts.RepoScopeActive { + if _, ok := allowed[sample.Repo]; !ok { + continue + } + } + out.TotalRows++ + out.Groups = append(out.Groups, graph.StructuralIntegrityExactGroup{ + Repo: sample.Repo, Kind: sample.Kind, Reason: sample.Reason, Origin: sample.Origin, Count: 1, + }) + out.Samples = append(out.Samples, sample) + } + limit := opts.SampleLimit + if limit > 0 && len(out.Samples) > limit { + out.Samples = out.Samples[:limit] + out.Truncated = true + } + return out, nil +} + +func TestAnalyzeEdgeAuditGraphIntegrityRepoScopeIsolation(t *testing.T) { + store := &scopedIntegrityAuditStore{Graph: graph.New()} + addRejectedIntegrityAttempt(store, "repo-a", "a") + addRejectedIntegrityAttempt(store, "repo-b", "b") + store.exactSamples = []graph.StructuralIntegrityExactSample{ + {Repo: "repo-a", Kind: graph.EdgeImplements, Reason: graph.StructuralReasonParameterTarget, Origin: "lsp", From: "repo-a/source-a", To: "repo-a/target#param:a"}, + {Repo: "repo-b", Kind: graph.EdgeImplements, Reason: graph.StructuralReasonParameterTarget, Origin: "lsp", From: "repo-b/source-b", To: "repo-b/target#param:b"}, + } + srv := &Server{graph: store} + ctx := withRepoAllow(context.Background(), map[string]bool{"repo-a": true}) + text := edgeAuditIntegrityText(t, srv, ctx, map[string]any{"limit": 10.0}) + if strings.Contains(text, "repo-b") { + t.Fatalf("scoped audit leaked another repository: %s", text) + } + if !strings.Contains(text, "repo-a") { + t.Fatalf("scoped audit lost allowed repository: %s", text) + } + var payload map[string]any + if err := json.Unmarshal([]byte(text), &payload); err != nil { + t.Fatal(err) + } + integrity := payload["graph_integrity"].(map[string]any) + sinceTotals := integrity["since_open"].(map[string]any)["totals"].(map[string]any) + if sinceTotals["write_rejected"] != float64(1) { + t.Fatalf("scoped since-open total mismatch: %+v", sinceTotals) + } + persisted := integrity["persisted"].(map[string]any) + if persisted["total_rows"] != float64(1) { + t.Fatalf("scoped persisted total mismatch: %+v", persisted) + } + + noneCtx := withRepoAllow(context.Background(), map[string]bool{"repo-none": true}) + noneText := edgeAuditIntegrityText(t, srv, noneCtx, map[string]any{}) + if strings.Contains(noneText, "repo-a") || strings.Contains(noneText, "repo-b") { + t.Fatalf("active empty-result scope leaked diagnostic samples: %s", noneText) + } +} + +func TestEdgeAuditGraphIntegrityHonorsCancellation(t *testing.T) { + store := &scopedIntegrityAuditStore{Graph: graph.New()} + srv := &Server{graph: store} + ctx, cancel := context.WithCancel(context.Background()) + cancel() + _, err := srv.edgeAuditGraphIntegrity(ctx, 10, false, nil) + if !errors.Is(err, context.Canceled) { + t.Fatalf("expected context cancellation, got %v", err) + } +} diff --git a/internal/mcp/tools_analyze_temporal_verify.go b/internal/mcp/tools_analyze_temporal_verify.go index d152932ab..95cab15f7 100644 --- a/internal/mcp/tools_analyze_temporal_verify.go +++ b/internal/mcp/tools_analyze_temporal_verify.go @@ -34,9 +34,9 @@ import ( ) // maxTemporalNodeSourceBytes caps the per-node source handed to the LLM so a -// giant function body can't blow the prompt budget. Mirrors the cap baked into -// analyzer.NewFileSourceProvider; restated here because this handler resolves -// paths through the server (resolveNodePath) rather than a bare root join. +// giant function body can't blow the prompt budget. Source reading lives here +// because this handler resolves paths through the server's multi-repository and +// worktree-aware resolveNodePath logic. const maxTemporalNodeSourceBytes = 6000 // serverSourceProvider implements resolver.TemporalSourceProvider by reading a diff --git a/internal/mcp/tools_enhancements_test.go b/internal/mcp/tools_enhancements_test.go index 7ffbf86c2..04c1cc800 100644 --- a/internal/mcp/tools_enhancements_test.go +++ b/internal/mcp/tools_enhancements_test.go @@ -539,11 +539,6 @@ func TestPropertyCrossCommunityWarningCorrectness(t *testing.T) { if len(warning.AffectedCommunities) != 2 { rt.Errorf("expected 2 affected communities, got %d", len(warning.AffectedCommunities)) } - - if len(warning.Couplings) != 0 { - rt.Fatalf("impact scored %d coupling pair(s); the safety gate must not read the edge set", - len(warning.Couplings)) - } }) } diff --git a/internal/parser/crashpool/crashpool.go b/internal/parser/crashpool/crashpool.go index ee15c159e..c7248ced2 100644 --- a/internal/parser/crashpool/crashpool.go +++ b/internal/parser/crashpool/crashpool.go @@ -35,10 +35,11 @@ func init() { // extractRequest is one unit of parse work sent parent → worker. type extractRequest struct { - Seq uint64 - RelPath string - Language string - Content []byte + Seq uint64 + RelPath string + Language string + Content []byte + TemporalEnvHelpers []string } // extractResponse is the worker → parent reply for one request. diff --git a/internal/parser/crashpool/options_test.go b/internal/parser/crashpool/options_test.go new file mode 100644 index 000000000..4b6c09974 --- /dev/null +++ b/internal/parser/crashpool/options_test.go @@ -0,0 +1,108 @@ +package crashpool + +import ( + "bytes" + "encoding/gob" + "fmt" + "testing" + + "github.com/zzet/gortex/internal/parser" + "github.com/zzet/gortex/internal/parser/languages" +) + +func crashpoolTemporalSource(helper string) []byte { + return []byte(fmt.Sprintf(`package sample +import "go.temporal.io/sdk/workflow" +func Run(ctx workflow.Context) { + name := %s("ACTIVITY", "MyActivity") + workflow.ExecuteActivity(ctx, name) +} +`, helper)) +} + +func responseTemporalMeta(resp extractResponse) map[string]any { + for _, edge := range resp.Edges { + if edge != nil && edge.Meta != nil && edge.Meta["via"] == "temporal.stub" { + return edge.Meta + } + } + return nil +} + +func TestWorkerRequestOptionsAreInterleavedWithoutContamination(t *testing.T) { + reg := parser.NewRegistry() + languages.RegisterAll(reg) + + requests := []extractRequest{ + {Seq: 1, RelPath: "a.go", Language: "go", Content: crashpoolTemporalSource("RepoAHelper"), TemporalEnvHelpers: []string{"RepoAHelper"}}, + {Seq: 2, RelPath: "b.go", Language: "go", Content: crashpoolTemporalSource("RepoBHelper"), TemporalEnvHelpers: []string{"RepoBHelper"}}, + {Seq: 3, RelPath: "a-again.go", Language: "go", Content: crashpoolTemporalSource("RepoAHelper"), TemporalEnvHelpers: []string{"RepoBHelper"}}, + } + var in bytes.Buffer + enc := gob.NewEncoder(&in) + for _, req := range requests { + if err := enc.Encode(&req); err != nil { + t.Fatal(err) + } + } + var out bytes.Buffer + if err := serveWorker(reg, &in, &out); err != nil { + t.Fatal(err) + } + + dec := gob.NewDecoder(&out) + for i, wantSource := range []any{"allowlist", "allowlist", nil} { + var resp extractResponse + if err := dec.Decode(&resp); err != nil { + t.Fatal(err) + } + if resp.Err != "" { + t.Fatalf("response %d: %s", i, resp.Err) + } + meta := responseTemporalMeta(resp) + if meta == nil { + t.Fatalf("response %d has no Temporal edge", i) + } + if got := meta["temporal_env_source"]; got != wantSource { + t.Fatalf("response %d source = %#v, want %#v", i, got, wantSource) + } + } +} + +func TestWorkerAndInProcessOptionsParity(t *testing.T) { + reg := parser.NewRegistry() + languages.RegisterAll(reg) + src := crashpoolTemporalSource("CorporateHelper") + opts := parser.NewExtractionOptions([]string{"CorporateHelper"}) + + worker := serveOne(reg, extractRequest{ + Seq: 1, RelPath: "worker.go", Language: "go", Content: src, + TemporalEnvHelpers: opts.TemporalEnvHelpers(), + }) + workerMeta := responseTemporalMeta(worker) + if workerMeta == nil { + t.Fatal("worker Temporal edge missing") + } + + ext, _ := reg.GetByLanguage("go") + result, err := parser.Extract(ext, "worker.go", src, opts) + if err != nil { + t.Fatal(err) + } + defer result.ReleaseTree() + var inProcessMeta map[string]any + for _, edge := range result.Edges { + if edge != nil && edge.Meta != nil && edge.Meta["via"] == "temporal.stub" { + inProcessMeta = edge.Meta + break + } + } + if inProcessMeta == nil { + t.Fatal("in-process Temporal edge missing") + } + for _, key := range []string{"temporal_name", "temporal_name_origin", "temporal_env_source", "temporal_kind"} { + if workerMeta[key] != inProcessMeta[key] { + t.Fatalf("%s mismatch: worker=%#v in-process=%#v", key, workerMeta[key], inProcessMeta[key]) + } + } +} diff --git a/internal/parser/crashpool/pool.go b/internal/parser/crashpool/pool.go index f20b40f43..6d293ad5d 100644 --- a/internal/parser/crashpool/pool.go +++ b/internal/parser/crashpool/pool.go @@ -13,6 +13,7 @@ import ( "go.uber.org/zap" + "github.com/zzet/gortex/internal/parser" "github.com/zzet/gortex/internal/platform" "github.com/zzet/gortex/internal/procio" ) @@ -165,12 +166,18 @@ func (w *procWorker) roundTrip(req *extractRequest, resp *extractResponse) error return nil } -// Submit extracts one file in a worker subprocess. It blocks until a -// worker is free, then runs the round-trip under requestTimeout. A -// crashed or hung worker is killed, replaced, and reported via -// Result.Crashed; the pool stays at full strength so the caller can -// keep submitting. +// Submit extracts one file with empty request options. It preserves the base +// pool API for ordinary callers and tests. func (p *Pool) Submit(relPath, language string, content []byte) Result { + return p.SubmitWithOptions(relPath, language, content, nil) +} + +// SubmitWithOptions extracts one file in a worker subprocess. It blocks until +// a worker is free, then runs the round-trip under requestTimeout. Temporal +// helper names are normalized into the request and never mutate worker state. +// A crashed or hung worker is killed, replaced, and reported via Result.Crashed; +// the pool stays at full strength so the caller can keep submitting. +func (p *Pool) SubmitWithOptions(relPath, language string, content []byte, temporalEnvHelpers []string) Result { p.mu.Lock() closed := p.closed p.mu.Unlock() @@ -184,10 +191,11 @@ func (p *Pool) Submit(relPath, language string, content []byte) Result { } req := extractRequest{ - Seq: p.seq.Add(1), - RelPath: relPath, - Language: language, - Content: content, + Seq: p.seq.Add(1), + RelPath: relPath, + Language: language, + Content: content, + TemporalEnvHelpers: parser.NewExtractionOptions(temporalEnvHelpers).TemporalEnvHelpers(), } var resp extractResponse done := make(chan error, 1) diff --git a/internal/parser/crashpool/worker.go b/internal/parser/crashpool/worker.go index e3f5548a8..58bf4c819 100644 --- a/internal/parser/crashpool/worker.go +++ b/internal/parser/crashpool/worker.go @@ -123,7 +123,8 @@ func serveOne(reg *parser.Registry, req extractRequest) (resp extractResponse) { resp.Err = "crashpool: no extractor for language " + req.Language return resp } - result, err := ext.Extract(req.RelPath, req.Content) + opts := parser.NewExtractionOptions(req.TemporalEnvHelpers) + result, err := parser.Extract(ext, req.RelPath, req.Content, opts) if err != nil { resp.Err = err.Error() return resp diff --git a/internal/parser/extractor.go b/internal/parser/extractor.go index 714f9a7e5..67d816c84 100644 --- a/internal/parser/extractor.go +++ b/internal/parser/extractor.go @@ -2,6 +2,8 @@ package parser import ( "io" + "sort" + "strings" "github.com/zzet/gortex/internal/graph" ) @@ -13,6 +15,62 @@ type Extractor interface { Extract(filePath string, src []byte) (*ExtractionResult, error) } +// ExtractionOptions is immutable request-scoped parser configuration. Its +// fields remain private so an extractor cannot mutate repository-owned state. +// Use NewExtractionOptions to construct a normalized value and accessors to +// obtain defensive copies for serialization. +type ExtractionOptions struct { + temporalEnvHelpers []string +} + +// NewExtractionOptions normalizes Temporal helper names by trimming whitespace, +// dropping empty entries, exact case-sensitive deduplication, and sorting. The +// stable order makes crash-worker serialization deterministic without changing +// Go identifier semantics. +func NewExtractionOptions(temporalEnvHelpers []string) ExtractionOptions { + if len(temporalEnvHelpers) == 0 { + return ExtractionOptions{} + } + seen := make(map[string]struct{}, len(temporalEnvHelpers)) + normalized := make([]string, 0, len(temporalEnvHelpers)) + for _, name := range temporalEnvHelpers { + name = strings.TrimSpace(name) + if name == "" { + continue + } + if _, exists := seen[name]; exists { + continue + } + seen[name] = struct{}{} + normalized = append(normalized, name) + } + if len(normalized) == 0 { + return ExtractionOptions{} + } + sort.Strings(normalized) + return ExtractionOptions{temporalEnvHelpers: normalized} +} + +// TemporalEnvHelpers returns a defensive copy of the configured names. +func (o ExtractionOptions) TemporalEnvHelpers() []string { + return append([]string(nil), o.temporalEnvHelpers...) +} + +// OptionsExtractor is an optional capability. Extractor remains unchanged so +// existing languages and external plugins keep their source compatibility. +type OptionsExtractor interface { + ExtractWithOptions(filePath string, src []byte, opts ExtractionOptions) (*ExtractionResult, error) +} + +// Extract dispatches one request through the optional options-aware capability, +// falling back to the base Extractor contract for ordinary languages/plugins. +func Extract(e Extractor, filePath string, src []byte, opts ExtractionOptions) (*ExtractionResult, error) { + if oe, ok := e.(OptionsExtractor); ok { + return oe.ExtractWithOptions(filePath, src, opts) + } + return e.Extract(filePath, src) +} + // PreParser is an optional Extractor capability: a source-rewriting hook run // before tree-sitter parsing. It lets a language neutralise constructs that // confuse the grammar (e.g. C-family conditional-compilation directives that diff --git a/internal/parser/extractor_options_test.go b/internal/parser/extractor_options_test.go new file mode 100644 index 000000000..c5de5df33 --- /dev/null +++ b/internal/parser/extractor_options_test.go @@ -0,0 +1,68 @@ +package parser + +import ( + "reflect" + "testing" +) + +type optionsTestExtractor struct { + seen []string +} + +func (e *optionsTestExtractor) Language() string { return "test" } +func (e *optionsTestExtractor) Extensions() []string { return []string{".test"} } +func (e *optionsTestExtractor) Extract(string, []byte) (*ExtractionResult, error) { + return &ExtractionResult{}, nil +} +func (e *optionsTestExtractor) ExtractWithOptions(_ string, _ []byte, opts ExtractionOptions) (*ExtractionResult, error) { + e.seen = opts.TemporalEnvHelpers() + return &ExtractionResult{}, nil +} + +type baseTestExtractor struct { + called bool +} + +func (e *baseTestExtractor) Language() string { return "base" } +func (e *baseTestExtractor) Extensions() []string { return []string{".base"} } +func (e *baseTestExtractor) Extract(string, []byte) (*ExtractionResult, error) { + e.called = true + return &ExtractionResult{}, nil +} + +func TestExtractionOptionsNormalizeAndDefend(t *testing.T) { + input := []string{" HelperB ", "HelperA", "HelperB", "", "helpera"} + opts := NewExtractionOptions(input) + input[0] = "mutated" + + want := []string{"HelperA", "HelperB", "helpera"} + got := opts.TemporalEnvHelpers() + if !reflect.DeepEqual(got, want) { + t.Fatalf("TemporalEnvHelpers() = %#v, want %#v", got, want) + } + got[0] = "mutated" + if again := opts.TemporalEnvHelpers(); !reflect.DeepEqual(again, want) { + t.Fatalf("accessor exposed mutable state: %#v", again) + } +} + +func TestExtractDispatchesOptionalCapability(t *testing.T) { + ext := &optionsTestExtractor{} + _, err := Extract(ext, "x.test", nil, NewExtractionOptions([]string{"CustomEnv"})) + if err != nil { + t.Fatal(err) + } + if !reflect.DeepEqual(ext.seen, []string{"CustomEnv"}) { + t.Fatalf("options seen = %#v", ext.seen) + } +} + +func TestExtractFallsBackToBaseExtractor(t *testing.T) { + ext := &baseTestExtractor{} + if _, err := Extract(ext, "x.base", nil, NewExtractionOptions([]string{"ignored"})); err != nil { + t.Fatal(err) + } + if !ext.called { + t.Fatal("base Extract was not called") + } +} diff --git a/internal/parser/languages/go_function_shape_test.go b/internal/parser/languages/go_function_shape_test.go index b96c20322..cd1fc7122 100644 --- a/internal/parser/languages/go_function_shape_test.go +++ b/internal/parser/languages/go_function_shape_test.go @@ -4,6 +4,7 @@ import ( "testing" "github.com/zzet/gortex/internal/graph" + "github.com/zzet/gortex/internal/parser" ) // runGoExtract is a small harness used by the function-shape tests @@ -39,8 +40,8 @@ func runGoExtract(t *testing.T, src string) *extractedFixture { func runGoExtractWithEnvHelpers(t *testing.T, src string, envHelpers []string) *extractedFixture { t.Helper() ext := NewGoExtractor() - ext.SetEnvHelperNames(envHelpers) - result, err := ext.Extract("pkg/foo.go", []byte(src)) + opts := parser.NewExtractionOptions(envHelpers) + result, err := ext.ExtractWithOptions("pkg/foo.go", []byte(src), opts) if err != nil { t.Fatalf("extract: %v", err) } diff --git a/internal/parser/languages/golang.go b/internal/parser/languages/golang.go index 7412096b5..545d1d2d4 100644 --- a/internal/parser/languages/golang.go +++ b/internal/parser/languages/golang.go @@ -197,31 +197,6 @@ const qGoAll = ` type GoExtractor struct { lang *sitter.Language qAll *parser.PreparedQuery - // envHelperExtra is the per-repo corporate env-helper allow-list (lower- - // cased names) loaded from the git-ignored `.gortex/temporal-allowlist.yaml`. - // Names here are merged with the built-in goEnvHelperNames when recognising - // a Temporal env-or-default dispatch helper, promoting the resolved edge to - // the inferred (visible) tier. Empty/nil when no local allow-list is loaded. - envHelperExtra map[string]bool -} - -// SetEnvHelperNames installs the per-repo corporate env-helper allow-list on -// the extractor. Names are stored lower-cased for case-insensitive matching. -// Called once during extractor registration (config is not available at parse -// time); a nil / empty slice clears it. Safe to call before indexing begins; -// must not be called concurrently with Extract. -func (e *GoExtractor) SetEnvHelperNames(names []string) { - if len(names) == 0 { - e.envHelperExtra = nil - return - } - m := make(map[string]bool, len(names)) - for _, n := range names { - if n = strings.TrimSpace(n); n != "" { - m[strings.ToLower(n)] = true - } - } - e.envHelperExtra = m } func NewGoExtractor() *GoExtractor { @@ -375,6 +350,17 @@ type goDeferredValueIdent struct { } func (e *GoExtractor) Extract(filePath string, src []byte) (*parser.ExtractionResult, error) { + return e.ExtractWithOptions(filePath, src, parser.ExtractionOptions{}) +} + +// ExtractWithOptions applies request-scoped repository configuration without +// mutating this shared extractor. Configured Temporal helpers extend the +// built-in helper set for this extraction only. +func (e *GoExtractor) ExtractWithOptions(filePath string, src []byte, opts parser.ExtractionOptions) (*parser.ExtractionResult, error) { + envHelperExtra := make(map[string]bool) + for _, name := range opts.TemporalEnvHelpers() { + envHelperExtra[name] = true + } tree, err := parser.ParseFile(src, e.lang) if err != nil { return nil, err @@ -547,7 +533,7 @@ func (e *GoExtractor) Extract(filePath string, src []byte) (*parser.ExtractionRe // variable, try to resolve it to an env-var-with-literal // -default so the dispatch lands on the default activity. if argNode != nil && argNode.Type() == "identifier" { - if litDef, constName, source, ok := goTemporalEnvDefaultName(expr.Node, name, src, e.envHelperExtra); ok { + if litDef, constName, source, ok := goTemporalEnvDefaultName(expr.Node, name, src, envHelperExtra); ok { dc.tempEnvDefault = true dc.tempEnvSource = source if constName != "" { diff --git a/internal/parser/languages/golang_temporal.go b/internal/parser/languages/golang_temporal.go index 7ca37d609..6f7ea64fb 100644 --- a/internal/parser/languages/golang_temporal.go +++ b/internal/parser/languages/golang_temporal.go @@ -1254,7 +1254,7 @@ func goEnvHelperDefaultLiteral(call *sitter.Node, src []byte, extra map[string]b break } } - if !matched && extra[strings.ToLower(callee)] { + if !matched && extra[callee] { matched = true } if !matched { diff --git a/internal/parser/languages/golang_temporal_options_test.go b/internal/parser/languages/golang_temporal_options_test.go new file mode 100644 index 000000000..1ca321e3a --- /dev/null +++ b/internal/parser/languages/golang_temporal_options_test.go @@ -0,0 +1,194 @@ +package languages + +import ( + "fmt" + "sync" + "testing" + + "github.com/zzet/gortex/internal/graph" + "github.com/zzet/gortex/internal/parser" +) + +func temporalOptionSource(helper, reassignment string) []byte { + return []byte(fmt.Sprintf(`package sample +import "go.temporal.io/sdk/workflow" +func Run(ctx workflow.Context) { + name := %s("ACTIVITY", "MyActivity") + %s + workflow.ExecuteActivity(ctx, name) +} +`, helper, reassignment)) +} + +func findTemporalDispatchMeta(result *parser.ExtractionResult) map[string]any { + if result == nil { + return nil + } + defer result.ReleaseTree() + for _, edge := range result.Edges { + if edge == nil || edge.Kind != graph.EdgeCalls || edge.Meta == nil { + continue + } + if edge.Meta["via"] == "temporal.stub" && edge.Meta["temporal_kind"] == "activity" { + return edge.Meta + } + } + return nil +} + +func temporalDispatchMeta(t *testing.T, result *parser.ExtractionResult) map[string]any { + t.Helper() + meta := findTemporalDispatchMeta(result) + if meta == nil { + t.Fatal("Temporal activity dispatch edge not found") + } + return meta +} + +func TestGoExtractorTemporalHelperOptionsExtendBuiltins(t *testing.T) { + ext := NewGoExtractor() + + builtIn, err := ext.Extract("builtin.go", temporalOptionSource("GetEnvOrDefault", "")) + if err != nil { + t.Fatal(err) + } + if got := temporalDispatchMeta(t, builtIn)["temporal_env_source"]; got != "allowlist" { + t.Fatalf("built-in source = %#v", got) + } + + ordinary, err := ext.Extract("ordinary.go", temporalOptionSource("CorporateHelper", "")) + if err != nil { + t.Fatal(err) + } + ordinaryMeta := temporalDispatchMeta(t, ordinary) + if got := ordinaryMeta["temporal_env_source"]; got != nil { + t.Fatalf("ordinary Extract unexpectedly used local options: %#v", got) + } + + opts := parser.NewExtractionOptions([]string{"CorporateHelper"}) + configured, err := ext.ExtractWithOptions("configured.go", temporalOptionSource("CorporateHelper", ""), opts) + if err != nil { + t.Fatal(err) + } + configuredMeta := temporalDispatchMeta(t, configured) + if got := configuredMeta["temporal_env_source"]; got != "allowlist" { + t.Fatalf("configured source = %#v", got) + } + if got := configuredMeta["temporal_name"]; got != "MyActivity" { + t.Fatalf("configured target = %#v", got) + } + + constSource := []byte(`package sample +import "go.temporal.io/sdk/workflow" +const DefaultActivity = "MyActivity" +func Run(ctx workflow.Context) { + name := CorporateHelper("ACTIVITY", DefaultActivity) + workflow.ExecuteActivity(ctx, name) +} +`) + withConst, err := ext.ExtractWithOptions("const.go", constSource, opts) + if err != nil { + t.Fatal(err) + } + constMeta := temporalDispatchMeta(t, withConst) + if got := constMeta["temporal_env_source"]; got != "const_ref" { + t.Fatalf("constant source = %#v", got) + } + if got := constMeta["temporal_default_const"]; got != "DefaultActivity" { + t.Fatalf("constant reference = %#v", got) + } + + wrongCase, err := ext.ExtractWithOptions("case.go", temporalOptionSource("CorporateHelper", ""), parser.NewExtractionOptions([]string{"corporatehelper"})) + if err != nil { + t.Fatal(err) + } + if got := temporalDispatchMeta(t, wrongCase)["temporal_env_source"]; got != nil { + t.Fatalf("case-insensitive local match = %#v", got) + } +} + +func TestGoExtractorTemporalGenericEnvPromotionAndReassignment(t *testing.T) { + ext := NewGoExtractor() + src := temporalOptionSource("CorporateEnv", "") + + heuristic, err := ext.Extract("heuristic.go", src) + if err != nil { + t.Fatal(err) + } + if got := temporalDispatchMeta(t, heuristic)["temporal_env_source"]; got != "heuristic" { + t.Fatalf("heuristic source = %#v", got) + } + + opts := parser.NewExtractionOptions([]string{"CorporateEnv"}) + promoted, err := ext.ExtractWithOptions("promoted.go", src, opts) + if err != nil { + t.Fatal(err) + } + if got := temporalDispatchMeta(t, promoted)["temporal_env_source"]; got != "allowlist" { + t.Fatalf("promoted source = %#v", got) + } + + invalidated, err := ext.ExtractWithOptions("invalidated.go", temporalOptionSource("CorporateEnv", "name = choose()"), opts) + if err != nil { + t.Fatal(err) + } + if got := temporalDispatchMeta(t, invalidated)["temporal_env_source"]; got != nil { + t.Fatalf("later reassignment retained env source = %#v", got) + } +} + +func TestGoExtractorTemporalOptionsConcurrentIsolation(t *testing.T) { + ext := NewGoExtractor() + type tc struct { + helper string + other string + } + cases := []tc{{helper: "RepoAHelper", other: "RepoBHelper"}, {helper: "RepoBHelper", other: "RepoAHelper"}} + + var wg sync.WaitGroup + errs := make(chan error, len(cases)*10) + for _, testCase := range cases { + testCase := testCase + for i := 0; i < 10; i++ { + wg.Add(1) + go func() { + defer wg.Done() + opts := parser.NewExtractionOptions([]string{testCase.helper}) + own, err := ext.ExtractWithOptions(testCase.helper+".go", temporalOptionSource(testCase.helper, ""), opts) + if err != nil { + errs <- err + return + } + ownMeta := findTemporalDispatchMeta(own) + if ownMeta == nil { + errs <- fmt.Errorf("%s dispatch edge missing", testCase.helper) + return + } + if got := ownMeta["temporal_env_source"]; got != "allowlist" { + errs <- fmt.Errorf("%s source = %#v", testCase.helper, got) + return + } + foreign, err := ext.ExtractWithOptions(testCase.other+".go", temporalOptionSource(testCase.other, ""), opts) + if err != nil { + errs <- err + return + } + foreignMeta := findTemporalDispatchMeta(foreign) + if foreignMeta == nil { + errs <- fmt.Errorf("%s dispatch edge missing", testCase.other) + return + } + if got := foreignMeta["temporal_env_source"]; got != nil { + errs <- fmt.Errorf("%s contaminated %s: %#v", testCase.helper, testCase.other, got) + } + }() + } + } + wg.Wait() + close(errs) + for err := range errs { + if err != nil { + t.Error(err) + } + } +} diff --git a/internal/parser/languages/helpers_complexity.go b/internal/parser/languages/helpers_complexity.go index a2ba9cd65..d8925cb8f 100644 --- a/internal/parser/languages/helpers_complexity.go +++ b/internal/parser/languages/helpers_complexity.go @@ -151,30 +151,6 @@ var javaComplexitySkip = map[string]bool{ "class_declaration": true, } -// GoComplexity / TSComplexity / PyComplexity / RustComplexity / -// JavaComplexity — convenience wrappers picking the right table. -// Pass the function/method's body block (not the whole declaration) -// so the count excludes any header-side noise. -func GoComplexity(body *sitter.Node) int { - return CyclomaticComplexity(body, goComplexityNodes, goComplexitySkip) -} - -func TSComplexity(body *sitter.Node) int { - return CyclomaticComplexity(body, tsComplexityNodes, tsComplexitySkip) -} - -func PyComplexity(body *sitter.Node) int { - return CyclomaticComplexity(body, pyComplexityNodes, pyComplexitySkip) -} - -func RustComplexity(body *sitter.Node) int { - return CyclomaticComplexity(body, rustComplexityNodes, rustComplexitySkip) -} - -func JavaComplexity(body *sitter.Node) int { - return CyclomaticComplexity(body, javaComplexityNodes, javaComplexitySkip) -} - // --- Cognitive complexity & loop depth (NEW-CBM-1) ------------------ // // Cyclomatic complexity counts decision points flatly; cognitive diff --git a/internal/parser/languages/register.go b/internal/parser/languages/register.go index b76241686..0ffb5b673 100644 --- a/internal/parser/languages/register.go +++ b/internal/parser/languages/register.go @@ -230,24 +230,3 @@ func ConfigureTemporalJavaInvokers(reg *parser.Registry, invokers, methods []str } } } - -// EnvHelperConfigurable is implemented by extractors that accept a per-repo -// Temporal env-helper allow-list. Only the Go extractor implements it today. -type EnvHelperConfigurable interface { - SetEnvHelperNames(names []string) -} - -// ConfigureTemporalEnvHelpers installs the per-repo corporate env-helper -// allow-list (loaded from a git-ignored `.gortex/temporal-allowlist.yaml`) onto -// every registered extractor that supports it. No-op when names is empty, so -// callers can pass the loader result unconditionally. -func ConfigureTemporalEnvHelpers(reg *parser.Registry, names []string) { - if len(names) == 0 { - return - } - if ext, ok := reg.GetByLanguage("go"); ok { - if c, ok := ext.(EnvHelperConfigurable); ok { - c.SetEnvHelperNames(names) - } - } -} diff --git a/internal/parser/treesitter.go b/internal/parser/treesitter.go index 111ca74e7..42272a6ee 100644 --- a/internal/parser/treesitter.go +++ b/internal/parser/treesitter.go @@ -343,8 +343,3 @@ func putMatchScratch(s *matchScratch) { // set), and preserve-across-calls made it worse as the growing map // slowed every lookup. The text copy in Utf8Text is fine as-is — keep // pooling, drop interning. - -// NodeText extracts the text content of a tree-sitter node from source bytes. -func NodeText(node *sitter.Node, src []byte) string { - return node.Content(src) -} diff --git a/internal/parser/tsitter/tsitter.go b/internal/parser/tsitter/tsitter.go index 1c0456e63..ee9cd866f 100644 --- a/internal/parser/tsitter/tsitter.go +++ b/internal/parser/tsitter/tsitter.go @@ -232,19 +232,6 @@ func putArena(a *nodeArena) { arenaPool.Put(a) } -// WrapNode wraps a value Node from the new API into our shim. It derives -// the language key eagerly so navigation from the result stays alloc-free, -// and seeds a fresh arena so the subtree walk below it allocates in chunks. -func WrapNode(n ts.Node) *Node { - a := newNodeArena() - nn := a.alloc() - nn.inner = n - nn.valid = true - nn.langKey = unsafe.Pointer(n.Language().Inner) - nn.arena = a - return nn -} - // WrapVal wraps a ts.Node reached from n (e.g. a query capture), // carrying n's language key so Type() on the result and its descendants // needs neither CGO nor allocation. @@ -541,9 +528,6 @@ type Tree struct { arena *nodeArena // pooled; taken lazily on first RootNode, returned on Close } -// WrapTree wraps a *ts.Tree for internal use by the parser package. -func WrapTree(t *ts.Tree) *Tree { return &Tree{inner: t} } - // Inner exposes the underlying *ts.Tree for internal use. func (t *Tree) Inner() *ts.Tree { return t.inner } diff --git a/internal/profiles/profiles.go b/internal/profiles/profiles.go index bc6e5b8d5..42faa6b20 100644 --- a/internal/profiles/profiles.go +++ b/internal/profiles/profiles.go @@ -20,7 +20,6 @@ import ( "fmt" "os" "path/filepath" - "sort" "strings" "time" @@ -324,10 +323,3 @@ func Remove(dir string) error { } return nil } - -// SortedEagerTools is a display helper for `gortex instructions list`. -func (p Profile) SortedEagerTools() []string { - out := append([]string(nil), p.EagerTools...) - sort.Strings(out) - return out -} diff --git a/internal/progress/logo.go b/internal/progress/logo.go index 9626f4259..428c0c264 100644 --- a/internal/progress/logo.go +++ b/internal/progress/logo.go @@ -135,25 +135,6 @@ func MeshLogo(tick int) string { return strings.Join(lines[:], "\n") } -// MeshFrame returns the gortex mark with label (bold) and sub (dim) beside -// it. Used by watch loops or custom views that want the brand block without -// owning a live tracker. -func MeshFrame(tick int, label, sub string) string { - mesh := MeshLogo(tick) - if label == "" && sub == "" { - return mesh + "\n" - } - right := lipgloss.JoinVertical( - lipgloss.Left, - "", - styleLabel.Render(label), - "", - styleSub.Render(sub), - "", - ) - return lipgloss.JoinHorizontal(lipgloss.Top, mesh, " ", right) + "\n" -} - // MeshLogoLines returns the number of vertical rows the mark occupies. // Exported so wizard / dashboard layouts can reserve space without re-counting // the constant. diff --git a/internal/progress/spinner.go b/internal/progress/spinner.go index c01246287..d4c5c4d06 100644 --- a/internal/progress/spinner.go +++ b/internal/progress/spinner.go @@ -1,9 +1,6 @@ package progress -import ( - "context" - "io" -) +import "io" // Spinner is the single-operation face of the Tracker, kept for the many // call sites that want "animate this one label, then ✓ or ✗". It is a thin @@ -24,9 +21,6 @@ func NewSpinner(w io.Writer) *Spinner { // Disable forces the spinner into plain-text mode. Effective before Start. func (s *Spinner) Disable() { s.t.disable() } -// Enabled reports whether the spinner is animating. -func (s *Spinner) Enabled() bool { return s.t.Animated() } - // Start begins animating with the given label. func (s *Spinner) Start(label string) { s.t.Start(label) } @@ -49,62 +43,3 @@ func (s *Spinner) Done() { s.t.Done("", "") } // Fail stops the spinner and replaces the frame with a red ✗ summary. func (s *Spinner) Fail(err error) { s.t.Fail(err) } - -// Tracker exposes the underlying tracker for call sites that outgrow the -// single-label surface (explicit steps, log lines above the animation). -func (s *Spinner) Tracker() *Tracker { return s.t } - -// Multi fans out reporter ticks to all of rs. Nil entries are skipped. -func Multi(rs ...Reporter) Reporter { - out := make([]Reporter, 0, len(rs)) - for _, r := range rs { - if r == nil { - continue - } - out = append(out, r) - } - switch len(out) { - case 0: - return Nop{} - case 1: - return out[0] - default: - return multiReporter(out) - } -} - -type multiReporter []Reporter - -func (m multiReporter) Report(stage string, current, total int) { - for _, r := range m { - r.Report(stage, current, total) - } -} - -// Run animates a spinner around fn. The context passed to fn carries the -// spinner as a Reporter, so any progress.FromContext(ctx).Report(…) inside fn -// drives the live step rows. The spinner is finished (✓ or ✗) before Run -// returns. -func Run(ctx context.Context, w io.Writer, label string, fn func(context.Context) error) error { - sp := NewSpinner(w) - return runWith(ctx, sp, label, fn) -} - -// RunDisabled is Run with the spinner forced into plain-text mode. -func RunDisabled(ctx context.Context, w io.Writer, label string, fn func(context.Context) error) error { - sp := NewSpinner(w) - sp.Disable() - return runWith(ctx, sp, label, fn) -} - -func runWith(ctx context.Context, sp *Spinner, label string, fn func(context.Context) error) error { - sp.Start(label) - ctx = WithReporter(ctx, sp) - err := fn(ctx) - if err != nil { - sp.Fail(err) - } else { - sp.Done() - } - return err -} diff --git a/internal/progress/spinner_test.go b/internal/progress/spinner_test.go index 21a57d478..0a862fb16 100644 --- a/internal/progress/spinner_test.go +++ b/internal/progress/spinner_test.go @@ -77,38 +77,6 @@ func TestSpinnerDoneIsIdempotent(t *testing.T) { } } -func TestMultiFansOutAndSkipsNil(t *testing.T) { - a := &countingReporter{} - b := &countingReporter{} - r := Multi(a, nil, b) - - r.Report("walk", 1, 10) - r.Report("parse", 0, 0) - - if a.calls != 2 || b.calls != 2 { - t.Errorf("expected 2 calls each, got a=%d b=%d", a.calls, b.calls) - } -} - -func TestMultiCollapsesToSingle(t *testing.T) { - a := &countingReporter{} - r := Multi(nil, a, nil) - if r != a { - t.Errorf("expected Multi with one non-nil to return that reporter directly") - } -} - -func TestMultiAllNilReturnsNop(t *testing.T) { - r := Multi(nil, nil) - if _, ok := r.(Nop); !ok { - t.Errorf("expected Nop when all inputs nil, got %T", r) - } -} - -type countingReporter struct{ calls int } - -func (c *countingReporter) Report(string, int, int) { c.calls++ } - // TestASCIIGlyphFallbackOnOEMCodepage proves the F8 contract: a terminal that // cannot render UTF-8 (a legacy OEM codepage, a linux virtual console, or an // explicit GORTEX_ASCII) gets an ASCII glyph set for the spinner finish diff --git a/internal/progress/theme.go b/internal/progress/theme.go index 07499fd89..e0662372c 100644 --- a/internal/progress/theme.go +++ b/internal/progress/theme.go @@ -62,10 +62,8 @@ var ( StyleBox = styleBox ) -// PaletteFg / PaletteAccent / PaletteErr expose the resolved lipgloss colors -// for callers that need to apply them to a freshly-built style (rather than -// re-using one of the pre-composed styles above). Returned values are -// lipgloss.Color, ready to feed into any lipgloss.NewStyle().Foreground call. -func PaletteFg() lipgloss.Color { return colFg } -func PaletteAccent() lipgloss.Color { return colAccent } -func PaletteErr() lipgloss.Color { return colErr } +// PaletteFg exposes the resolved lipgloss foreground color for callers that +// need to apply it to a freshly-built style (rather than re-using one of the +// pre-composed styles above). The returned value is a lipgloss.Color, ready to +// feed into any lipgloss.NewStyle().Foreground call. +func PaletteFg() lipgloss.Color { return colFg } diff --git a/internal/progress/timing.go b/internal/progress/timing.go deleted file mode 100644 index 937369aa0..000000000 --- a/internal/progress/timing.go +++ /dev/null @@ -1,102 +0,0 @@ -package progress - -import ( - "fmt" - "io" - "sync" - "time" -) - -// TimingReporter records the first-seen timestamp of each stage and, -// when printed, emits a per-stage duration breakdown. Shows where -// wall-clock time is being spent during a full index pass. -// -// Stage transitions are detected by the reporter seeing a *new* stage -// label arrive — subsequent ticks for the same stage (progress updates -// like "parsing 3000/5000") don't create a new entry. This matches the -// indexer's usage where it calls Report once at stage entry and then -// again with counter updates. -type TimingReporter struct { - mu sync.Mutex - start time.Time - stages []stageEntry - seen map[string]int // stage → index in stages (also suppresses duplicates) -} - -type stageEntry struct { - name string - seen time.Time - ticks int -} - -// NewTimingReporter returns a reporter with its clock anchored at now. -func NewTimingReporter() *TimingReporter { - return &TimingReporter{ - start: time.Now(), - seen: make(map[string]int), - } -} - -// Report records a stage tick. The first tick for a given stage name -// is treated as the stage's start timestamp. -func (r *TimingReporter) Report(stage string, _, _ int) { - if stage == "" { - return - } - r.mu.Lock() - defer r.mu.Unlock() - if idx, ok := r.seen[stage]; ok { - r.stages[idx].ticks++ - return - } - r.seen[stage] = len(r.stages) - r.stages = append(r.stages, stageEntry{name: stage, seen: time.Now(), ticks: 1}) -} - -// WriteReport prints a two-column breakdown: per-stage duration and -// cumulative time since the reporter was created. end defaults to -// time.Now() when zero; pass an explicit value when the caller -// finishes before a final "indexing complete" stage tick is emitted. -func (r *TimingReporter) WriteReport(w io.Writer, end time.Time) { - r.mu.Lock() - defer r.mu.Unlock() - - if end.IsZero() { - end = time.Now() - } - if len(r.stages) == 0 { - fmt.Fprintln(w, "no stages recorded") - return - } - - fmt.Fprintf(w, "%-32s %12s %12s\n", "stage", "duration", "cumulative") - fmt.Fprintf(w, "%-32s %12s %12s\n", - "--------------------------------", "------------", "------------") - for i, s := range r.stages { - var nextStart time.Time - if i+1 < len(r.stages) { - nextStart = r.stages[i+1].seen - } else { - nextStart = end - } - duration := nextStart.Sub(s.seen) - cumulative := nextStart.Sub(r.start) - fmt.Fprintf(w, "%-32s %12s %12s\n", - s.name, formatMs(duration), formatMs(cumulative)) - } -} - -// formatMs formats a duration as "123ms" / "4.56s" / "1m23s" depending -// on magnitude. Small-enough to be readable in a CLI dump. -func formatMs(d time.Duration) string { - switch { - case d < time.Second: - return fmt.Sprintf("%dms", d.Milliseconds()) - case d < time.Minute: - return fmt.Sprintf("%.2fs", d.Seconds()) - default: - m := int(d / time.Minute) - s := int((d % time.Minute) / time.Second) - return fmt.Sprintf("%dm%02ds", m, s) - } -} diff --git a/internal/progress/ui.go b/internal/progress/ui.go index fe8591c68..e1f76e21f 100644 --- a/internal/progress/ui.go +++ b/internal/progress/ui.go @@ -137,12 +137,6 @@ func Card(title, body string) string { return styleBox.Border(activeGlyphs().Border).Render(body) + "\n" } -// Indent prefixes every line of s with the given number of spaces. -func Indent(s string, n int) string { - pad := strings.Repeat(" ", n) - return pad + strings.ReplaceAll(s, "\n", "\n"+pad) -} - // SortStrings is a small convenience used by callers preparing chip lists. func SortStrings(s []string) []string { out := append([]string(nil), s...) diff --git a/internal/progress/zaplog.go b/internal/progress/zaplog.go deleted file mode 100644 index 65342e4d9..000000000 --- a/internal/progress/zaplog.go +++ /dev/null @@ -1,114 +0,0 @@ -package progress - -import ( - "context" - "sync" - "time" - - "go.uber.org/zap" -) - -// ZapReporter logs every Report call as a zap INFO line. Used in -// non-TTY environments (the daemon, CI) where the Spinner is -// silent so progress is invisible. Stage transitions get logged -// immediately; intra-stage progress (current/total) gets logged on -// transition AND every progressInterval seconds so a slow stage -// emits a heartbeat instead of going quiet. -type ZapReporter struct { - logger *zap.Logger - prefix string - interval time.Duration - - mu sync.Mutex - lastStage string - stageStart time.Time - lastEmitted time.Time - lastCur int - lastTotal int -} - -// NewZapReporter creates a reporter that logs to the given logger. -// prefix is added to every log line ("indexer", "multi-repo", …). -// interval is the heartbeat cadence for intra-stage progress -// (0 disables heartbeats — only stage transitions log). -func NewZapReporter(logger *zap.Logger, prefix string, interval time.Duration) *ZapReporter { - if logger == nil { - logger = zap.NewNop() - } - return &ZapReporter{ - logger: logger, - prefix: prefix, - interval: interval, - } -} - -// Report records a stage advancement. Always logs on a stage -// transition; logs intra-stage updates at most once per interval. -func (r *ZapReporter) Report(stage string, cur, total int) { - r.mu.Lock() - defer r.mu.Unlock() - now := time.Now() - if stage != r.lastStage { - if r.lastStage != "" { - r.logger.Info(r.prefix+": stage end", - zap.String("stage", r.lastStage), - zap.Duration("elapsed", now.Sub(r.stageStart)), - ) - } - r.lastStage = stage - r.stageStart = now - r.lastEmitted = now - r.lastCur = cur - r.lastTotal = total - r.logger.Info(r.prefix+": stage start", - zap.String("stage", stage), - zap.Int("current", cur), - zap.Int("total", total), - ) - return - } - // Same stage — heartbeat at most once per interval. - if r.interval > 0 && now.Sub(r.lastEmitted) < r.interval { - return - } - r.lastEmitted = now - r.lastCur = cur - r.lastTotal = total - r.logger.Info(r.prefix+": stage progress", - zap.String("stage", stage), - zap.Int("current", cur), - zap.Int("total", total), - zap.Duration("elapsed", now.Sub(r.stageStart)), - ) -} - -// StartHeartbeat runs a goroutine that logs an "alive" line every -// interval until the context is done. Useful when the indexer is -// inside a long-running phase that doesn't call Report itself -// (e.g. the disk backend's bulk writes during a slow drain). -func StartHeartbeat(ctx context.Context, logger *zap.Logger, prefix string, interval time.Duration, snapshot func() map[string]any) { - if logger == nil || interval <= 0 { - return - } - go func() { - t := time.NewTicker(interval) - defer t.Stop() - start := time.Now() - for { - select { - case <-ctx.Done(): - return - case <-t.C: - fields := []zap.Field{ - zap.Duration("elapsed", time.Since(start)), - } - if snapshot != nil { - for k, v := range snapshot() { - fields = append(fields, zap.Any(k, v)) - } - } - logger.Info(prefix+": heartbeat", fields...) - } - } - }() -} diff --git a/internal/query/cosine_refine.go b/internal/query/cosine_refine.go index a00442229..6a58414fc 100644 --- a/internal/query/cosine_refine.go +++ b/internal/query/cosine_refine.go @@ -27,9 +27,9 @@ type embedderProvider interface { } // backendEmbedder extracts the query embedder from a search backend, -// unwrapping one level of Swappable. Returns nil when no embedder is -// reachable — the caller treats that as "vector channel inactive" and -// skips the refinement entirely. +// through the backend's lock-scoped capability surface. Returns nil when no +// embedder is reachable — the caller treats that as "vector channel inactive" +// and skips the refinement entirely. func backendEmbedder(b search.Backend) embedding.Provider { if b == nil { return nil @@ -39,11 +39,6 @@ func backendEmbedder(b search.Backend) embedding.Provider { return e } } - if sw, ok := b.(*search.Swappable); ok { - if ep, ok := sw.Inner().(embedderProvider); ok { - return ep.Embedder() - } - } return nil } diff --git a/internal/query/engine.go b/internal/query/engine.go index 58809d8cb..70572fdd7 100644 --- a/internal/query/engine.go +++ b/internal/query/engine.go @@ -491,8 +491,10 @@ func (e *Engine) GetCluster(nodeID string, opts QueryOptions) *SubGraph { } // SearchSymbols performs full-text search across all nodes. -// When a search backend is configured, uses BM25/Bleve ranking with -// camelCase-aware tokenization. Falls back to substring matching otherwise. +// When a search backend is configured, uses that backend's ranking — +// the in-process BM25 index in tests and evals, the store-native FTS +// index in production — with camelCase-aware tokenization. Falls back +// to substring matching otherwise. func (e *Engine) SearchSymbols(query string, limit int) []*graph.Node { return e.SearchSymbolsScoped(query, limit, QueryOptions{}) } diff --git a/internal/resolver/temporal_calls_test.go b/internal/resolver/temporal_calls_test.go index 687eca820..cacdafebb 100644 --- a/internal/resolver/temporal_calls_test.go +++ b/internal/resolver/temporal_calls_test.go @@ -257,6 +257,23 @@ func TestResolveTemporalCalls_EnvDefaultResolvesSpeculative(t *testing.T) { assert.Equal(t, true, call.Meta[graph.MetaSpeculative], "env-default edge must be hidden-by-default") } +func TestResolveTemporalCalls_AllowlistedEnvDefaultResolvesInferred(t *testing.T) { + b := newTemporalTestGraph() + b.addGoFunc("wf/workflow.go::OrderWorkflow", "OrderWorkflow", "wf/workflow.go", "svc") + call := b.addStubCallEnvDefault("wf/workflow.go::OrderWorkflow", "activity", "ChargeCard", "wf/workflow.go") + call.Meta["temporal_env_source"] = "allowlist" + activity := b.addGoFunc("wf/activity.go::ChargeCard", "ChargeCard", "wf/activity.go", "svc") + b.addGoFunc("wf/main.go::setupWorker", "setupWorker", "wf/main.go", "svc") + b.addGoRegister("wf/main.go::setupWorker", "activity", "ChargeCard", "wf/main.go") + + resolved := ResolveTemporalCalls(b.g) + assert.Equal(t, 1, resolved) + assert.Equal(t, activity.ID, call.To) + assert.Equal(t, graph.OriginASTInferred, call.Origin) + assert.GreaterOrEqual(t, call.Confidence, 0.6) + assert.NotEqual(t, true, call.Meta[graph.MetaSpeculative], "allow-listed default must be visible") +} + func TestResolveTemporalCalls_EnvDefaultUnresolvedStaysPlaceholder(t *testing.T) { b := newTemporalTestGraph() b.addGoFunc("wf/workflow.go::WF", "WF", "wf/workflow.go", "svc") diff --git a/internal/review/critique.go b/internal/review/critique.go index 82cf0d780..4a8a280a3 100644 --- a/internal/review/critique.go +++ b/internal/review/critique.go @@ -4,7 +4,6 @@ import ( "context" "encoding/json" "fmt" - "sort" "strings" ) @@ -278,18 +277,3 @@ func valueOr(v, fallback string) string { } return v } - -// SortCritiquedBySeverity orders critiqued findings worst-severity-first for a -// deterministic dropped-list rendering. -func SortCritiquedBySeverity(rows []CritiquedFinding) { - sort.SliceStable(rows, func(i, j int) bool { - si, sj := severityRank(rows[i].Finding.Severity), severityRank(rows[j].Finding.Severity) - if si != sj { - return si > sj - } - if rows[i].Finding.File != rows[j].Finding.File { - return rows[i].Finding.File < rows[j].Finding.File - } - return rows[i].Finding.Line < rows[j].Finding.Line - }) -} diff --git a/internal/review/ground.go b/internal/review/ground.go index 241682ac4..d32fe4030 100644 --- a/internal/review/ground.go +++ b/internal/review/ground.go @@ -15,8 +15,6 @@ package review import ( - "strings" - "github.com/zzet/gortex/internal/astquery" "github.com/zzet/gortex/internal/graph" ) @@ -139,15 +137,3 @@ func loopDepth(n *graph.Node) int { } return 0 } - -// IsReviewDetector reports whether a detector name belongs to the -// review rulepack's undecidable set — exposed so the review flow can -// decide which matches still need grounding when it reuses -// pre-computed rulepack results. -func IsReviewDetector(name string) bool { - switch strings.TrimSpace(name) { - case detectorLoopQueryGo, detectorLoopQueryPy, detectorCheckActMapGo, detectorCheckActDictPy: - return true - } - return false -} diff --git a/internal/runtimeactivity/activity.go b/internal/runtimeactivity/activity.go index 22393ded5..57cefaf19 100644 --- a/internal/runtimeactivity/activity.go +++ b/internal/runtimeactivity/activity.go @@ -40,9 +40,6 @@ type Tracker struct { kindsMu sync.Mutex byKind map[string]int64 - - hooksMu sync.RWMutex - hooks []func(string) } // NewTracker returns an independent tracker. Most production code uses the @@ -82,10 +79,8 @@ func (t *Tracker) End(kind string) { } kind = normalizeKind(kind) t.gate.RLock() - remaining := t.active.Add(-1) - if remaining < 0 { + if t.active.Add(-1) < 0 { t.active.Store(0) - remaining = 0 } t.epoch.Add(1) t.lastNano.Store(time.Now().UnixNano()) @@ -97,10 +92,6 @@ func (t *Tracker) End(kind string) { } t.kindsMu.Unlock() t.gate.RUnlock() - - if remaining == 0 { - t.notifyIdle(kind) - } } // Snapshot returns process activity without blocking a running maintenance @@ -154,41 +145,6 @@ func (t *Tracker) RunIfQuiet(quiet time.Duration, fn func()) (ran bool, retryAft return true, 0 } -// RegisterIdleHook registers a process-lifetime callback invoked after tracked -// activity transitions to zero. Hooks must return quickly; schedulers should -// launch their expensive work asynchronously. The returned function unregisters -// the hook and is primarily useful to tests. -func (t *Tracker) RegisterIdleHook(hook func(string)) func() { - if t == nil || hook == nil { - return func() {} - } - t.hooksMu.Lock() - t.hooks = append(t.hooks, hook) - idx := len(t.hooks) - 1 - t.hooksMu.Unlock() - var once sync.Once - return func() { - once.Do(func() { - t.hooksMu.Lock() - if idx < len(t.hooks) { - t.hooks[idx] = nil - } - t.hooksMu.Unlock() - }) - } -} - -func (t *Tracker) notifyIdle(kind string) { - t.hooksMu.RLock() - hooks := append([]func(string){}, t.hooks...) - t.hooksMu.RUnlock() - for _, hook := range hooks { - if hook != nil { - hook(kind) - } - } -} - func normalizeKind(kind string) string { if kind == "" { return "unspecified" @@ -211,6 +167,3 @@ func Current() Snapshot { return process.Snapshot() } func RunIfQuiet(quiet time.Duration, fn func()) (bool, time.Duration) { return process.RunIfQuiet(quiet, fn) } - -// RegisterIdleHook registers a process-wide idle-transition callback. -func RegisterIdleHook(hook func(string)) func() { return process.RegisterIdleHook(hook) } diff --git a/internal/runtimeactivity/activity_test.go b/internal/runtimeactivity/activity_test.go index 683cb0881..e89d90f5d 100644 --- a/internal/runtimeactivity/activity_test.go +++ b/internal/runtimeactivity/activity_test.go @@ -2,7 +2,6 @@ package runtimeactivity import ( "sync" - "sync/atomic" "testing" "time" ) @@ -71,38 +70,6 @@ func TestTrackerExclusiveMaintenanceBlocksNewWork(t *testing.T) { } } -func TestTrackerIdleHookCoversLostWakeupTransition(t *testing.T) { - tracker := NewTracker() - var calls atomic.Int64 - called := make(chan struct{}, 2) - unregister := tracker.RegisterIdleHook(func(kind string) { - if kind != "analysis" { - t.Errorf("idle kind = %q, want analysis", kind) - } - calls.Add(1) - called <- struct{}{} - }) - defer unregister() - - tracker.Begin("analysis") - tracker.End("analysis") - select { - case <-called: - case <-time.After(time.Second): - t.Fatal("idle hook was lost") - } - tracker.Begin("analysis") - tracker.End("analysis") - select { - case <-called: - case <-time.After(time.Second): - t.Fatal("second idle hook was lost") - } - if got := calls.Load(); got != 2 { - t.Fatalf("idle hook calls = %d, want 2", got) - } -} - func TestTrackerConcurrentKindsBalance(t *testing.T) { tracker := NewTracker() const workers = 32 diff --git a/internal/search/bleve.go b/internal/search/bleve.go deleted file mode 100644 index 555c806bb..000000000 --- a/internal/search/bleve.go +++ /dev/null @@ -1,227 +0,0 @@ -package search - -import ( - "fmt" - "os" - "path/filepath" - "strings" - "sync/atomic" - - "github.com/blevesearch/bleve/v2" - "github.com/blevesearch/bleve/v2/analysis/analyzer/custom" - "github.com/blevesearch/bleve/v2/analysis/token/lowercase" - "github.com/blevesearch/bleve/v2/analysis/tokenizer/unicode" - "github.com/blevesearch/bleve/v2/mapping" - - // Register default KV store. - _ "github.com/blevesearch/bleve/v2/index/upsidedown/store/gtreap" -) - -// BleveBackend wraps Bleve for full-text search over code symbols. -// Better for large repos (50k+ symbols) and multi-repo mode. -type BleveBackend struct { - index bleve.Index - count atomic.Int64 - diskPath string // non-empty when the index is disk-backed (scorch) -} - -// DiskPath returns the directory containing the on-disk index, or "" -// when the backend is running fully in memory. -func (b *BleveBackend) DiskPath() string { return b.diskPath } - -// DiskBytes walks the index directory and sums file sizes. Zero when -// in-memory. Called at most once per `gortex daemon status` invocation. -func (b *BleveBackend) DiskBytes() uint64 { - if b.diskPath == "" { - return 0 - } - var total uint64 - _ = filepath.Walk(b.diskPath, func(_ string, info os.FileInfo, err error) error { - if err != nil || info == nil || info.IsDir() { - return nil - } - total += uint64(info.Size()) - return nil - }) - return total -} - -// symbolDoc is the document structure indexed by Bleve. -type symbolDoc struct { - Name string `json:"name"` - Path string `json:"path"` - Signature string `json:"signature"` - // Combined field for broader matching. - All string `json:"all"` -} - -// NewBleve creates a Bleve-backed search index (in-memory via -// upsidedown + gtreap). Heavy but self-contained — no disk writes. -func NewBleve() (*BleveBackend, error) { - indexMapping := buildMapping() - - idx, err := bleve.NewMemOnly(indexMapping) - if err != nil { - return nil, err - } - - return &BleveBackend{index: idx}, nil -} - -// NewBleveDisk creates a disk-backed Bleve index under dir using the -// scorch storage engine. The dir is created if missing; any existing -// index at the same path is removed first because we always rebuild -// from scratch (we don't have incremental-write semantics yet). -// Scorch is ~10-20× more memory-efficient than the in-memory -// upsidedown+gtreap store at the cost of disk IO on write, which only -// matters during initial index construction. -func NewBleveDisk(dir string) (*BleveBackend, error) { - if err := os.MkdirAll(dir, 0o755); err != nil { - return nil, fmt.Errorf("bleve disk dir: %w", err) - } - // Use a fixed child name so a caller passing a shared parent dir - // doesn't overwrite neighbouring state. - indexPath := filepath.Join(dir, "bleve.scorch") - if _, err := os.Stat(indexPath); err == nil { - if rmErr := os.RemoveAll(indexPath); rmErr != nil { - return nil, fmt.Errorf("clearing old bleve index at %s: %w", indexPath, rmErr) - } - } - - indexMapping := buildMapping() - idx, err := bleve.New(indexPath, indexMapping) - if err != nil { - return nil, fmt.Errorf("opening bleve disk index at %s: %w", indexPath, err) - } - - return &BleveBackend{index: idx, diskPath: indexPath}, nil -} - -func buildMapping() *mapping.IndexMappingImpl { - indexMapping := bleve.NewIndexMapping() - - // Custom analyzer: unicode tokenizer + lowercase. - // We pre-tokenize camelCase in Add(), so the analyzer just needs - // to handle the space-separated tokens we give it. - err := indexMapping.AddCustomAnalyzer("code", map[string]any{ - "type": custom.Name, - "tokenizer": unicode.Name, - "token_filters": []string{ - lowercase.Name, - }, - }) - if err != nil { - // Fallback to default analyzer. - return indexMapping - } - - // Document mapping. - docMapping := bleve.NewDocumentMapping() - - nameField := bleve.NewTextFieldMapping() - nameField.Analyzer = "code" - nameField.Store = false - docMapping.AddFieldMappingsAt("name", nameField) - - pathField := bleve.NewTextFieldMapping() - pathField.Analyzer = "code" - pathField.Store = false - docMapping.AddFieldMappingsAt("path", pathField) - - sigField := bleve.NewTextFieldMapping() - sigField.Analyzer = "code" - sigField.Store = false - docMapping.AddFieldMappingsAt("signature", sigField) - - allField := bleve.NewTextFieldMapping() - allField.Analyzer = "code" - allField.Store = false - docMapping.AddFieldMappingsAt("all", allField) - - indexMapping.DefaultMapping = docMapping - indexMapping.DefaultAnalyzer = "code" - - return indexMapping -} - -func (b *BleveBackend) Add(id string, fields ...string) { - // Pre-tokenize camelCase and rejoin with spaces so Bleve's - // unicode tokenizer can split them. - var parts []string - for _, f := range fields { - tokens := NormalizeFTSTokens(Tokenize(f)) - parts = append(parts, strings.Join(tokens, " ")) - } - - doc := symbolDoc{ - All: strings.Join(parts, " "), - } - if len(parts) > 0 { - doc.Name = parts[0] - } - if len(parts) > 1 { - doc.Path = parts[1] - } - if len(parts) > 2 { - doc.Signature = parts[2] - } - - if err := b.index.Index(id, doc); err == nil { - b.count.Add(1) - } -} - -func (b *BleveBackend) Remove(id string) { - if err := b.index.Delete(id); err == nil { - b.count.Add(-1) - } -} - -func (b *BleveBackend) Search(query string, limit int) []SearchResult { - // Pre-tokenize the query for camelCase splitting. - tokens := NormalizeFTSTokens(TokenizeQuery(query)) - if len(tokens) == 0 { - return nil - } - q := strings.Join(tokens, " ") - - searchReq := bleve.NewSearchRequest(bleve.NewQueryStringQuery(q)) - searchReq.Size = limit - - res, err := b.index.Search(searchReq) - if err != nil || res.Total == 0 { - return nil - } - - out := make([]SearchResult, 0, len(res.Hits)) - for _, hit := range res.Hits { - out = append(out, SearchResult{ - ID: hit.ID, - Score: hit.Score, - }) - } - return out -} - -func (b *BleveBackend) Count() int { - return int(b.count.Load()) -} - -// SizeBytes approximates Bleve's in-memory footprint. Bleve (with the -// default upsidedown + gtreap KV store) is much hungrier than the -// fields-and-postings accounting would suggest — gtreap's immutable -// persistent trees retain copy-on-write versions, and the upsidedown -// row layout expands every symbol into many keys. Calibrated against -// heap profiles on a ~65k-symbol index: live Bleve heap was ~2.0 GiB, -// i.e. ~32 KiB per document. Earlier estimates of ~2 KiB/doc were off -// by 16× and were the root cause of the large "other" bucket users -// saw in `gortex daemon status`. -func (b *BleveBackend) SizeBytes() uint64 { - return uint64(b.count.Load()) * 32768 -} - -func (b *BleveBackend) Close() { - if b.index != nil { - _ = b.index.Close() - } -} diff --git a/internal/search/chunk_dechunk_test.go b/internal/search/chunk_dechunk_test.go index 8e926b7c8..82c234f6c 100644 --- a/internal/search/chunk_dechunk_test.go +++ b/internal/search/chunk_dechunk_test.go @@ -2,7 +2,6 @@ package search import ( "context" - "strings" "testing" "github.com/stretchr/testify/assert" @@ -88,9 +87,9 @@ func TestHybridSearch_DeChunkPreservesOrder(t *testing.T) { "b.go::B#chunk0": "b.go::B", }) - // Empty text backend so only the vector channel decides ordering. + // Keep construction realistic; this assertion exercises the vector + // de-chunk order directly, before channel fusion. h := NewHybrid(NewBM25(), vec, fixedEmbedder{dims: dims}) - h.SetAutoAlpha(false) // plain RRF — vector ranks drive the order got := h.dechunkVectorIDs(vec.Search([]float32{1, 0, 0}, 8), 8) require.Len(t, got, 2) @@ -129,59 +128,3 @@ func TestVectorBackend_ResolveChunk(t *testing.T) { assert.False(t, isChunk) assert.Equal(t, "g.go::G", plain, "an unmapped ID must pass through unchanged") } - -// TestVectorBackend_ChunkMapSurvivesSaveLoad asserts the chunk map is -// persisted by Save and restored by LoadFrom — the daemon snapshot and -// the per-repo cache both rely on this so de-chunking still works after -// a restart. -func TestVectorBackend_ChunkMapSurvivesSaveLoad(t *testing.T) { - src := NewVector(3) - src.Add("big.go::Big#chunk0", []float32{1, 0, 0}) - src.Add("big.go::Big#chunk1", []float32{0, 1, 0}) - src.SetChunkMap(map[string]string{ - "big.go::Big#chunk0": "big.go::Big", - "big.go::Big#chunk1": "big.go::Big", - }) - - var buf strings.Builder - require.NoError(t, src.Save(&stringWriter{&buf})) - - dst := NewVector(3) - require.NoError(t, dst.LoadFrom(strings.NewReader(buf.String()))) - require.True(t, dst.HasChunks(), "chunk map must survive a Save/Load round-trip") - - parent, isChunk := dst.ResolveChunk("big.go::Big#chunk1") - assert.True(t, isChunk) - assert.Equal(t, "big.go::Big", parent) -} - -// TestVectorBackend_LegacyBlobLoadsWithoutChunkMap asserts a legacy raw -// HNSW export (written before the framed format) still loads, with an -// empty chunk map — the back-compat path. -func TestVectorBackend_LegacyBlobLoadsWithoutChunkMap(t *testing.T) { - // A VectorBackend with no chunk map, saved, then a fresh backend - // loaded from a stream that has had the frame magic stripped to - // simulate a pre-framing blob. - src := NewVector(3) - src.Add("a.go::A", []float32{1, 0, 0}) - var framed strings.Builder - require.NoError(t, src.Save(&stringWriter{&framed})) - - raw := framed.String() - // Frame layout: 4-byte magic + 4-byte map length + map JSON + HNSW. - // Strip magic+length+"{}" (an empty map JSON) to get the bare HNSW. - require.Greater(t, len(raw), 10) - bare := raw[4+4+2:] // 4 magic, 4 length, 2 = len("{}") - - dst := NewVector(3) - require.NoError(t, dst.LoadFrom(strings.NewReader(bare)), - "a legacy un-framed HNSW blob must still load") - assert.False(t, dst.HasChunks(), "a legacy blob has no chunk map") -} - -// stringWriter adapts a strings.Builder to io.Writer for the tests -// above (strings.Builder already satisfies io.Writer, but the wrapper -// keeps the intent explicit). -type stringWriter struct{ b *strings.Builder } - -func (w *stringWriter) Write(p []byte) (int, error) { return w.b.Write(p) } diff --git a/internal/search/fts_normalize.go b/internal/search/fts_normalize.go index 0f2d5226c..89959d2a0 100644 --- a/internal/search/fts_normalize.go +++ b/internal/search/fts_normalize.go @@ -49,8 +49,8 @@ var ftsStopWords = map[string]struct{}{ // NormalizeFTSTokens applies the FR63 stopword filter and Porter stemmer // to a token list produced by Tokenize / TokenizeQuery. The index path -// (BM25Backend.Add, BleveBackend.Add) and the query path -// (BM25Backend.Search, BleveBackend.Search) both call it, so a stemmed +// (BM25Backend.Add) and the query path (BM25Backend.Search) both +// call it, so a stemmed // posting list is always probed with stemmed query terms. // // Stopwords are dropped before stemming so a stemmed form can never diff --git a/internal/search/hybrid.go b/internal/search/hybrid.go index 7088878c1..ce5eccc1c 100644 --- a/internal/search/hybrid.go +++ b/internal/search/hybrid.go @@ -9,42 +9,28 @@ import ( "github.com/zzet/gortex/internal/search/rerank" ) -// HybridBackend combines text search (BM25/Bleve) with vector search (HNSW) -// using Reciprocal Rank Fusion (RRF) for result ranking. -// -// When autoAlpha is true (the default), Search() classifies the query as -// identifier-shaped or natural-language and applies an α-weighted fusion -// instead of even-weight RRF: identifier queries lean toward BM25 (small -// α) where exact-token matches are most reliable, NL queries balance both -// channels (larger α) so semantic similarity catches synonymous wording. -// Set autoAlpha=false via SetAutoAlpha to fall back to the original -// equal-weight RRF — useful for tests pinning the legacy ranking. +// HybridBackend combines text search (BM25 or the store-native FTS +// adapter) with vector search (HNSW) using query-adaptive, α-weighted +// Reciprocal Rank Fusion (RRF). Identifier-shaped queries lean toward BM25, +// where exact-token matches are most reliable; natural-language queries give +// semantic similarity more weight so synonymous wording can surface. type HybridBackend struct { - text Backend - vector *VectorBackend - embedder embedding.Provider - k int // RRF constant (default 60) - autoAlpha bool + text Backend + vector *VectorBackend + embedder embedding.Provider + k int // RRF constant (default 60) } -// NewHybrid creates a hybrid search backend with auto-α enabled. +// NewHybrid creates a hybrid search backend with adaptive α fusion. func NewHybrid(text Backend, vector *VectorBackend, embedder embedding.Provider) *HybridBackend { return &HybridBackend{ - text: text, - vector: vector, - embedder: embedder, - k: 60, - autoAlpha: true, + text: text, + vector: vector, + embedder: embedder, + k: 60, } } -// SetAutoAlpha toggles auto-α fusion. When false, Search() reverts to -// the original equal-weight RRF. -func (h *HybridBackend) SetAutoAlpha(on bool) { h.autoAlpha = on } - -// AutoAlpha reports whether auto-α fusion is active. -func (h *HybridBackend) AutoAlpha() bool { return h.autoAlpha } - // Add indexes a symbol in both text and vector backends. func (h *HybridBackend) Add(id string, fields ...string) { h.text.Add(id, fields...) @@ -63,12 +49,9 @@ func (h *HybridBackend) Remove(id string) { // they won't match graph nodes and will be filtered out. } -// Search runs both text and vector search, fuses results with RRF -// (equal weight) when autoAlpha is off, or α-weighted RRF when on. -// Auto-α leans toward BM25 for identifier queries (where exact-token -// matches are the most reliable signal) and balances both channels -// for natural-language queries (where semantic similarity catches -// synonymous wording). +// Search runs both text and vector search and fuses them with adaptive +// α-weighted RRF. Identifier queries lean toward BM25; natural-language +// queries give semantic similarity more weight. func (h *HybridBackend) Search(query string, limit int) []SearchResult { textResults, vecIDs, _ := h.searchChannels(query, limit) if len(vecIDs) == 0 { @@ -77,10 +60,7 @@ func (h *HybridBackend) Search(query string, limit int) []SearchResult { } return textResults } - if h.autoAlpha { - return alphaFuse(textResults, vecIDs, rerank.AlphaFor(query), h.k, limit) - } - return rrfFuse(textResults, vecIDs, h.k, limit) + return alphaFuse(textResults, vecIDs, rerank.AlphaFor(query), h.k, limit) } // SearchChannels returns the raw per-channel results — BM25 ranks @@ -243,9 +223,36 @@ func (h *HybridBackend) dechunkVectorIDs(rawIDs []string, want int) []string { // Count returns the text backend document count. func (h *HybridBackend) Count() int { return h.text.Count() } -// Close releases resources. +// Close releases resources owned by the hybrid. The embedding provider and a +// delegated vector searcher are externally owned; VectorBackend.Close only +// releases process-local vector state. func (h *HybridBackend) Close() { - h.text.Close() + if h == nil { + return + } + text := h.text + vector := h.vector + h.text = nil + h.vector = nil + h.embedder = nil + if vector != nil { + vector.Close() + } + if text != nil { + text.Close() + } +} + +// detachTextBackend transfers text-backend ownership to a replacement hybrid. +// It is intentionally private and may only be called after all users of h have +// drained (Swappable.ReplaceHybridVector holds the write lock when calling it). +func (h *HybridBackend) detachTextBackend() Backend { + if h == nil { + return nil + } + text := h.text + h.text = nil + return text } // TextBackend returns the underlying text search backend. @@ -262,10 +269,6 @@ func (h *HybridBackend) SizeBytes() uint64 { return BackendSize(h.text) + h.vector.SizeBytes() } -// TextSizeBytes returns just the text backend's size — used by the -// daemon status report to split "search" from "vectors" visually. -func (h *HybridBackend) TextSizeBytes() uint64 { return BackendSize(h.text) } - // VectorSizeBytes returns just the vector backend's size. func (h *HybridBackend) VectorSizeBytes() uint64 { return h.vector.SizeBytes() } @@ -281,8 +284,8 @@ func (h *HybridBackend) VectorSizeBytes() uint64 { return h.vector.SizeBytes() } // score(doc) = (1-α) × 1/(k+rank_text+1) + α × 1/(k+rank_vector+1) // // α=0 reduces to text-only; α=1 reduces to vector-only; α=0.5 is -// equivalent to rrfFuse with each channel halved (so absolute scores -// differ from rrfFuse but the relative ordering is the same). +// equal-weight RRF with each channel halved, so absolute scores differ +// from the unscaled formula but relative ordering is unchanged. func alphaFuse(textResults []SearchResult, vecIDs []string, alpha float64, k, limit int) []SearchResult { if alpha < 0 { alpha = 0 @@ -327,50 +330,3 @@ func alphaFuse(textResults []SearchResult, vecIDs []string, alpha float64, k, li } return out } - -// rrfFuse combines text and vector results using Reciprocal Rank Fusion. -// score(doc) = 1/(k+rank_text) + 1/(k+rank_vector) -func rrfFuse(textResults []SearchResult, vecIDs []string, k, limit int) []SearchResult { - scores := make(map[string]float64) - - // Text ranks. - for rank, r := range textResults { - scores[r.ID] += 1.0 / float64(k+rank+1) - } - - // Vector ranks. - for rank, id := range vecIDs { - scores[id] += 1.0 / float64(k+rank+1) - } - - // Sort by combined RRF score. - type scored struct { - id string - score float64 - } - var results []scored - for id, score := range scores { - results = append(results, scored{id: id, score: score}) - } - sort.Slice(results, func(i, j int) bool { - if results[i].score != results[j].score { - return results[i].score > results[j].score - } - // Stable secondary key: equal-score runs ship in a fixed order. - return results[i].id < results[j].id - }) - - if len(results) > limit { - results = results[:limit] - } - - // Convert back to SearchResult (use RRF score). - out := make([]SearchResult, len(results)) - for i, r := range results { - out[i] = SearchResult{ - ID: r.id, - Score: r.score, - } - } - return out -} diff --git a/internal/search/hybrid_test.go b/internal/search/hybrid_test.go index 6ae77bddb..54ad10718 100644 --- a/internal/search/hybrid_test.go +++ b/internal/search/hybrid_test.go @@ -7,7 +7,7 @@ import ( "github.com/stretchr/testify/require" ) -func TestRRFFuse(t *testing.T) { +func TestAlphaFuse_EqualWeights(t *testing.T) { textResults := []SearchResult{ {ID: "a", Score: 10}, {ID: "b", Score: 8}, @@ -15,7 +15,7 @@ func TestRRFFuse(t *testing.T) { } vecIDs := []string{"b", "d", "a"} - results := rrfFuse(textResults, vecIDs, 60, 10) + results := alphaFuse(textResults, vecIDs, 0.5, 60, 10) require.GreaterOrEqual(t, len(results), 3) // "a" and "b" appear in both lists → highest RRF scores. @@ -33,28 +33,28 @@ func TestRRFFuse(t *testing.T) { } } -func TestRRFFuse_EmptyVec(t *testing.T) { +func TestAlphaFuse_EmptyVec(t *testing.T) { textResults := []SearchResult{ {ID: "a", Score: 10}, {ID: "b", Score: 8}, } - results := rrfFuse(textResults, nil, 60, 10) + results := alphaFuse(textResults, nil, 0.5, 60, 10) // With no vec results, only text results contribute. assert.Len(t, results, 2) assert.Equal(t, "a", results[0].ID) } -func TestRRFFuse_Limit(t *testing.T) { +func TestAlphaFuse_Limit(t *testing.T) { textResults := []SearchResult{ {ID: "a"}, {ID: "b"}, {ID: "c"}, {ID: "d"}, {ID: "e"}, } vecIDs := []string{"f", "g", "h", "i", "j"} - results := rrfFuse(textResults, vecIDs, 60, 3) + results := alphaFuse(textResults, vecIDs, 0.5, 60, 3) assert.Len(t, results, 3) } -// --- alphaFuse + auto-α coverage ------------------------------------- +// --- adaptive alphaFuse coverage ------------------------------------- func TestAlphaFuse_SmallAlphaFavorsText(t *testing.T) { // Text-only candidate "t" and vector-only candidate "v" at the @@ -111,18 +111,3 @@ func TestAlphaFuse_DeterministicTieBreak(t *testing.T) { } } } - -func TestNewHybrid_AutoAlphaDefaultOn(t *testing.T) { - h := NewHybrid(nil, nil, nil) - if !h.AutoAlpha() { - t.Errorf("NewHybrid().AutoAlpha() = false, want true (auto-α default)") - } - h.SetAutoAlpha(false) - if h.AutoAlpha() { - t.Errorf("SetAutoAlpha(false) did not take effect") - } - h.SetAutoAlpha(true) - if !h.AutoAlpha() { - t.Errorf("SetAutoAlpha(true) did not take effect") - } -} diff --git a/internal/search/ngram_weights.go b/internal/search/ngram_weights.go index e7dea5394..d1e25b650 100644 --- a/internal/search/ngram_weights.go +++ b/internal/search/ngram_weights.go @@ -135,7 +135,7 @@ func BuildNgramBoundaries(g graph.Reader) *NgramTable { // tokenizer's split decisions become data-driven. The production // backend is a Swappable wrapping either a HybridBackend (text+vector) // or a bare BM25Backend; this unwraps both. Backends with no BM25 layer -// (Bleve, SymbolSearcher) do not run the sparse-ngram stage, so there +// (SymbolSearcher) do not run the sparse-ngram stage, so there // is nothing to install and the call is a harmless no-op returning // false. // @@ -147,6 +147,11 @@ func BuildNgramBoundaries(g graph.Reader) *NgramTable { // them while the gate is on — callers re-install only as part of a // fresh (re)index, never against a live, already-populated index. func InstallNgramBoundaries(backend Backend, table NgramBoundaries) bool { + if swappable, ok := backend.(*Swappable); ok { + inner, release := swappable.AcquireBackend() + defer release() + return InstallNgramBoundaries(inner, table) + } bm := bm25Of(backend) if bm == nil { return false @@ -157,9 +162,14 @@ func InstallNgramBoundaries(backend Backend, table NgramBoundaries) bool { // BuildAndInstallNgramBoundaries mines and installs a boundary table only // when backend actually contains a BM25 layer. Capability detection must -// precede BuildNgramBoundaries: native SQLite FTS and Bleve never consume the +// precede BuildNgramBoundaries: the native SQLite FTS never consumes the // table, and walking the whole graph for them is pure allocation and I/O. func BuildAndInstallNgramBoundaries(backend Backend, g graph.Reader) bool { + if swappable, ok := backend.(*Swappable); ok { + inner, release := swappable.AcquireBackend() + defer release() + return BuildAndInstallNgramBoundaries(inner, g) + } bm := bm25Of(backend) if bm == nil { return false @@ -168,16 +178,14 @@ func BuildAndInstallNgramBoundaries(backend Backend, g graph.Reader) bool { return true } -// bm25Of unwraps a backend down to its *BM25Backend, or returns nil -// when the backend has no BM25 layer. Mirrors the unwrap chain the -// engine uses for the bundle fast path: Swappable → HybridBackend → -// BM25Backend. +// bm25Of unwraps a non-swappable backend down to its *BM25Backend, or +// returns nil when the backend has no BM25 layer. Swappable callers are +// handled by the public installation functions so the returned pointer is +// consumed while its AcquireBackend pin remains held. func bm25Of(backend Backend) *BM25Backend { switch b := backend.(type) { case *BM25Backend: return b - case *Swappable: - return bm25Of(b.Inner()) case *HybridBackend: return bm25Of(b.TextBackend()) default: diff --git a/internal/search/ngram_weights_test.go b/internal/search/ngram_weights_test.go index 16477fd3f..e29e1b707 100644 --- a/internal/search/ngram_weights_test.go +++ b/internal/search/ngram_weights_test.go @@ -162,24 +162,27 @@ func TestInstallNgramBoundaries_BM25Chain(t *testing.T) { defer sw.Close() assert.True(t, InstallNgramBoundaries(sw, tbl)) - // A backend with no BM25 layer: no-op, returns false. Bleve has no - // BM25 inner. - blv, err := NewBleve() - require.NoError(t, err) - defer blv.Close() - assert.False(t, InstallNgramBoundaries(blv, tbl)) + // A backend with no BM25 layer: no-op, returns false. + assert.False(t, InstallNgramBoundaries(nonBM25Backend{}, tbl)) } +// nonBM25Backend is an inert Backend with no BM25 anywhere in its +// unwrap chain — the shape the ngram installers must refuse. +type nonBM25Backend struct{} + +func (nonBM25Backend) Add(string, ...string) {} +func (nonBM25Backend) Remove(string) {} +func (nonBM25Backend) Search(string, int) []SearchResult { return nil } +func (nonBM25Backend) Count() int { return 0 } +func (nonBM25Backend) Close() {} + func TestBuildAndInstallNgramBoundaries_ChecksCapabilityBeforeGraphScan(t *testing.T) { g := &countingNgramReader{Reader: boundaryFixtureGraph([]string{ "tokenAlpha", "tokenBeta", "tokenGamma", "tokenDelta", "alphaToken", "betaToken", "tokenize", "tokenizer", })} - blv, err := NewBleve() - require.NoError(t, err) - defer blv.Close() - assert.False(t, BuildAndInstallNgramBoundaries(blv, g)) + assert.False(t, BuildAndInstallNgramBoundaries(nonBM25Backend{}, g)) assert.Zero(t, g.allNodesCalls, "a backend without BM25 must not enumerate the graph") bm := NewBM25() diff --git a/internal/search/rerank/pipeline.go b/internal/search/rerank/pipeline.go index 032dae9fe..8ade05dfa 100644 --- a/internal/search/rerank/pipeline.go +++ b/internal/search/rerank/pipeline.go @@ -65,10 +65,6 @@ func New(signals []Signal, weights map[string]float64) *Pipeline { // NewDefault is shorthand for New(DefaultSignals(), DefaultWeights()). func NewDefault() *Pipeline { return New(DefaultSignals(), DefaultWeights()) } -// Signals returns the signal list. Order is stable but not -// load-bearing — scores are computed independently per signal. -func (p *Pipeline) Signals() []Signal { return p.signals } - // Weights returns a copy of the current weight map. func (p *Pipeline) Weights() map[string]float64 { out := make(map[string]float64, len(p.weights)) @@ -250,16 +246,6 @@ func sameSliceHeader(a, b []*Candidate) bool { return &a[0] == &b[0] } -// Nodes is a convenience that unwraps a result slice into the -// underlying graph nodes in score order. -func Nodes(cands []*Candidate) []*graph.Node { - out := make([]*graph.Node, 0, len(cands)) - for _, c := range cands { - out = append(out, c.Node) - } - return out -} - // DefaultSignals returns the canonical signal lineup in stable order. // Callers wanting a subset should construct New() directly. func DefaultSignals() []Signal { diff --git a/internal/search/search.go b/internal/search/search.go index 965f47204..9418056f5 100644 --- a/internal/search/search.go +++ b/internal/search/search.go @@ -1,11 +1,10 @@ // Package search provides full-text search over code symbols with // camelCase/snake_case-aware tokenization and BM25 ranking. // -// Two backends are available: -// - BM25Backend: custom in-memory inverted index (fast, zero deps) -// - BleveBackend: bleve-based index (better for large repos, multi-repo) -// -// Use AutoBackend to pick the right one based on symbol count. +// Production search runs on SymbolSearcherBackend, a thin adapter over +// the graph store's own FTS index — no parallel in-process corpus. +// BM25Backend, a self-contained in-memory inverted index, is the +// fallback for stores that expose no native symbol search. package search // SearchResult is a single search hit. @@ -36,8 +35,9 @@ type Backend interface { // expose its per-channel raw retrieval output. The rerank pipeline // queries it so BM25 and semantic (vector) ranks can contribute as // separate signals instead of being collapsed via RRF before scoring. -// Backends that only do text search (BM25 / Bleve) don't satisfy this -// interface; callers fall through to plain Search(). +// Backends that only do text search (BM25, the store-native FTS +// adapter) don't satisfy this interface; callers fall through to plain +// Search(). type ChannelSearcher interface { SearchChannels(query string, limit int) (textResults []SearchResult, vectorIDs []string) } @@ -62,26 +62,9 @@ func BackendSize(b Backend) uint64 { return 0 } -// AutoThreshold is the symbol count above which BleveBackend is used. -// Calibrated against real daemon runs: Bleve (upsidedown + gtreap) costs -// ~32 KiB per document live, so a 500k-doc in-memory Bleve would cost -// ~16 GiB of heap — painful but not catastrophic on a dev machine with -// a real code monorepo that has earned it. BM25 stays plenty fast at -// that size (roughly 450 MiB at ~900 B/doc), so the threshold is set -// to match the point where BM25 query quality starts to trail Bleve's -// richer tokenization and phrase support, not the point where BM25 -// runs out of speed. Users who cross the line and can't afford the -// in-memory cost should set GORTEX_BLEVE_DISK_DIR (disk-backed scorch, -// 10-20× smaller heap at the cost of file I/O). -// -// Declared as a var rather than a const so tests that exercise the -// auto-upgrade path (idempotency, single-fire gating) can drop it to -// a small value without having to seed a huge corpus. Production -// code never writes to it. -var AutoThreshold = 500000 - -// NewAuto creates a BM25Backend initially. Call Upgrade() after indexing -// if the count exceeds AutoThreshold and multi-repo mode is desired. +// NewAuto returns the default in-process text backend. Reached only +// when the graph store exposes no native symbol search; otherwise the +// indexer wires up a SymbolSearcherBackend over the store's own FTS. func NewAuto() Backend { return NewBM25() } diff --git a/internal/search/search_test.go b/internal/search/search_test.go index e38278b89..11e3a053d 100644 --- a/internal/search/search_test.go +++ b/internal/search/search_test.go @@ -104,13 +104,6 @@ func TestBM25Backend(t *testing.T) { runBackendTests(t, "BM25", backend) } -func TestBleveBackend(t *testing.T) { - backend, err := NewBleve() - require.NoError(t, err) - defer backend.Close() - runBackendTests(t, "Bleve", backend) -} - func TestBM25_RankingQuality(t *testing.T) { b := NewBM25() defer b.Close() @@ -153,21 +146,3 @@ func BenchmarkBM25_Search(b *testing.B) { backend.Search("get user auth", 20) } } - -func BenchmarkBleve_Search(b *testing.B) { - backend, err := NewBleve() - if err != nil { - b.Fatal(err) - } - defer backend.Close() - for i := 0; i < 10000; i++ { - backend.Add( - "pkg/file.go::func"+string(rune('A'+i%26))+string(rune('0'+i%10)), - "getUserById", "internal/auth/service.go", "func getUserById(id string) User", - ) - } - b.ResetTimer() - for b.Loop() { - backend.Search("get user auth", 20) - } -} diff --git a/internal/search/swappable.go b/internal/search/swappable.go index fad0145dd..6951fb344 100644 --- a/internal/search/swappable.go +++ b/internal/search/swappable.go @@ -1,20 +1,29 @@ package search -import "sync" +import ( + "sync" -// Swappable wraps a Backend and lets a single in-place swap be performed -// concurrently with reads. Used by the indexer to upgrade from the -// in-memory BM25 backend to Bleve once the corpus crosses AutoThreshold, + "github.com/zzet/gortex/internal/embedding" +) + +// Swappable wraps a Backend and lets an in-place swap be performed +// concurrently with reads. Used by the indexer to re-wrap the active +// text backend in a HybridBackend once the vector index is ready, // without making every call site re-thread a new Backend reference and -// without holding the indexer's lock during the (potentially seconds-long) -// re-population of Bleve. +// without holding the indexer's lock across the swap. // // Callers see a stable *Swappable; reads delegate to whichever inner -// backend is currently active. Swap atomically replaces the inner -// backend and closes the previous one. +// backend is currently active. ReplaceHybridVector atomically replaces the +// vector channel while preserving ownership of the live text backend. type Swappable struct { mu sync.RWMutex inner Backend + + // vectorUpdateMu serializes the durable corpus replacement and matching + // process-local publication as one logical operation. Multi-repository + // Indexers share this Swappable, so the lock prevents an older install from + // publishing stale aggregate statistics after a newer store commit. + vectorUpdateMu sync.Mutex } // NewSwappable wraps b. Panics if b is nil — every Indexer must start @@ -26,29 +35,104 @@ func NewSwappable(b Backend) *Swappable { return &Swappable{inner: b} } -// Swap installs the new backend and closes the old one. Safe to call -// concurrently with reads; the swap itself is brief (one pointer write -// under the write lock) so reads queued during the swap return promptly -// against the new backend. -func (s *Swappable) Swap(b Backend) { +// ReplaceHybridVector atomically publishes a new vector channel while +// retaining the active text backend. Acquiring the write lock first drains all +// readers of the old backend. Existing hybrid layers are then peeled under the +// lock, transferring their text ownership into exactly one replacement hybrid. +// After publication, the retired hybrids release only their vector state. +// +// This operation is the safe alternative to composing an unpinned backend +// snapshot with NewHybrid: that sequence can race a reader, nest hybrids, and close the text +// backend that the replacement just reused. ReplaceHybridVector panics when +// vector is nil or the Swappable has already been closed. +func (s *Swappable) ReplaceHybridVector(vector *VectorBackend, embedder embedding.Provider) { + if vector == nil { + panic("search.Swappable.ReplaceHybridVector: nil vector backend") + } + s.mu.Lock() - old := s.inner - s.inner = b + if s.inner == nil { + s.mu.Unlock() + panic("search.Swappable.ReplaceHybridVector: closed swappable") + } + + text := s.inner + retired := make([]*HybridBackend, 0, 1) + for { + hybrid, ok := text.(*HybridBackend) + if !ok { + break + } + retired = append(retired, hybrid) + text = hybrid.TextBackend() + } + if text == nil { + s.mu.Unlock() + panic("search.Swappable.ReplaceHybridVector: hybrid has no text backend") + } + for _, hybrid := range retired { + hybrid.detachTextBackend() + } + s.inner = NewHybrid(text, vector, embedder) s.mu.Unlock() - if old != nil && old != b { - old.Close() + + for _, hybrid := range retired { + hybrid.Close() + } +} + +// SerializeVectorUpdate runs fn while holding the publication lane shared by +// every Indexer that uses this Swappable. The callback should prepare expensive +// embeddings before entering this lane, then perform only the durable atomic +// corpus replacement and ReplaceHybridVector publication while it is held. +func (s *Swappable) SerializeVectorUpdate(fn func() error) error { + if fn == nil { + return nil } + s.vectorUpdateMu.Lock() + defer s.vectorUpdateMu.Unlock() + return fn() } -// Inner returns the currently-active backend. Used internally to test -// upgrade outcomes; production code should always go through the -// Backend interface methods on Swappable itself. +// AcquireBackend pins the currently active backend until release is called. +// Callers that need a capability not forwarded by Swappable must defer the +// returned release immediately and must not retain the backend beyond that +// scope. Holding the pin prevents ReplaceHybridVector and Close from retiring +// the backend underneath the caller. release is idempotent. +func (s *Swappable) AcquireBackend() (backend Backend, release func()) { + s.mu.RLock() + var once sync.Once + return s.inner, func() { + once.Do(s.mu.RUnlock) + } +} + +// Inner returns an unpinned snapshot of the currently-active backend. It is +// retained for tests and diagnostics only: production callers must not keep or +// dereference the result because replacement may retire it immediately after +// this method returns. Use AcquireBackend or a forwarded capability instead. func (s *Swappable) Inner() Backend { s.mu.RLock() defer s.mu.RUnlock() return s.inner } +// Embedder returns the active hybrid's externally-owned embedding provider. +// The lookup is protected by the Swappable read lock; replacement never closes +// the provider, so the returned provider remains under its original owner's +// lifecycle rather than the retired HybridBackend's lifecycle. +func (s *Swappable) Embedder() embedding.Provider { + s.mu.RLock() + defer s.mu.RUnlock() + type embedderProvider interface { + Embedder() embedding.Provider + } + if provider, ok := s.inner.(embedderProvider); ok { + return provider.Embedder() + } + return nil +} + // --- Backend interface ------------------------------------------------ func (s *Swappable) Add(id string, fields ...string) { @@ -122,14 +206,12 @@ func (s *Swappable) SearchSymbolBundles(query string, limit int) []SymbolBundle // warm-started daemon over a populated disk FTS isn't mistaken for an // empty backend. func (s *Swappable) DocCount() (int, bool) { - // Snapshot the inner backend under the lock but run the count - // outside it — DocCount can be a real store query and holding - // the RLock across it would stall a pending Swap (and, behind - // it, every new reader). + // Hold the read lock for the complete call. Replacement owns and retires + // the old backend after acquiring the write lock, so merely snapshotting + // the pointer here would let it close underneath a live store query. s.mu.RLock() - inner := s.inner - s.mu.RUnlock() - if dc, ok := inner.(DocCounter); ok { + defer s.mu.RUnlock() + if dc, ok := s.inner.(DocCounter); ok { return dc.DocCount() } return 0, false @@ -170,6 +252,11 @@ func (s *Swappable) Count() int { } func (s *Swappable) Close() { + // Match the vector-update lock order (publication lane, then backend lock) + // so shutdown cannot retire the Swappable midway through a durable corpus + // commit/publication callback. + s.vectorUpdateMu.Lock() + defer s.vectorUpdateMu.Unlock() s.mu.Lock() defer s.mu.Unlock() if s.inner != nil { diff --git a/internal/search/symbolsearcher_backend.go b/internal/search/symbolsearcher_backend.go index e862b9b48..8941eeba9 100644 --- a/internal/search/symbolsearcher_backend.go +++ b/internal/search/symbolsearcher_backend.go @@ -11,14 +11,14 @@ import ( // SymbolSearcherBackend adapts a graph.SymbolSearcher into the // search.Backend the daemon's search-symbols path consumes. // Engine.gatherBackendCandidates and the rerank pipeline don't need -// to know whether the backend is BM25 / Bleve / native FTS — they +// to know whether the backend is BM25 or native FTS — they // see a plain search.Backend and call Search on it. // // Production wiring: when the indexer detects that the backing // graph.Store also implements graph.SymbolSearcher, it constructs // this adapter as the initial // search.Backend wrapped by search.NewSwappable. The in-process -// Bleve / BM25 build path is then bypassed entirely. +// BM25 build path is then bypassed entirely. // // Add / Remove are no-ops on the adapter because the indexer // already drives the SymbolSearcher writes directly: diff --git a/internal/search/vector.go b/internal/search/vector.go index 3bc129c46..0c4c36787 100644 --- a/internal/search/vector.go +++ b/internal/search/vector.go @@ -1,11 +1,6 @@ package search import ( - "bytes" - "encoding/binary" - "encoding/json" - "fmt" - "io" "sync" "github.com/coder/hnsw" @@ -13,13 +8,6 @@ import ( "github.com/zzet/gortex/internal/graph" ) -// vectorFrameMagic prefixes the framed VectorBackend.Save format: a -// 4-byte magic, a uint32 chunk-map JSON length, the chunk-map JSON, -// then the raw HNSW export. A blob lacking the magic is a legacy -// (pre-chunking) raw HNSW export and is loaded with an empty chunk -// map — so old snapshots keep working. -var vectorFrameMagic = [4]byte{'G', 'V', 'X', '1'} - // VectorDelegate is the subset of graph.VectorSearcher the // VectorBackend shim consults when it's been told to delegate // instead of holding an in-process HNSW. Exported (with a @@ -32,33 +20,27 @@ type VectorDelegate interface { // VectorBackend stores and searches embedding vectors using HNSW index. // -// When delegate is set (via SetDelegate), the in-process HNSW is -// bypassed entirely: Add becomes a no-op (the indexer drives the -// delegate's bulk-upsert directly), Search forwards to the -// delegate's SimilarTo. The dims and chunkMap stay live so callers -// that need them (HybridBackend.dechunkVectorIDs) keep working -// against the same VectorBackend surface. +// When a delegate is installed, the in-process HNSW is absent: Add tracks +// compatibility-path writes without allocating, and Search forwards to the +// delegate's SimilarTo. Durable hits carry their parent IDs, so chunk results +// can be collapsed without retaining a process-local chunk map. type VectorBackend struct { graph *hnsw.Graph[string] count int dims int // chunkMap maps a synthetic chunk vector ID ("#chunkK") - // to its parent symbol ID. It is non-empty only when AST - // sub-chunking split one or more symbols into multiple vectors. - // Search results are de-chunked through it so a symbol is never - // returned twice and chunk IDs never leak to callers. + // to its parent symbol ID. It is non-empty only for the legacy + // in-process build path; durable delegates return ParentID with each hit. chunkMap map[string]string mu sync.RWMutex - // delegate is the optional engine-native vector searcher (today - // only graph.SymbolSearcher-implementing stores). Set means - // "don't build the in-process HNSW; route reads through here". - // The wrapped delegateCount tracks Add-call deltas so Count() - // reports a non-zero figure once the indexer has finished its - // bulk upsert — HybridBackend gates the vector channel on - // Count() > 0. - delegate VectorDelegate - delegateCount int + // delegate is the optional durable vector searcher. Delegated backends + // never retain an in-process HNSW graph. delegateCount is the complete + // durable corpus size; delegateChunkCount is the number of vectors that + // represent chunks rather than whole symbols. + delegate VectorDelegate + delegateCount int + delegateChunkCount int } // NewVector creates a vector search backend for the given embedding dimensions. @@ -71,6 +53,31 @@ func NewVector(dims int) *VectorBackend { } } +// NewDelegatedVector creates a vector backend that routes searches to a +// durable store without allocating an in-process HNSW graph. vectorCount and +// chunkCount describe the complete durable corpus, not just this process's +// writes, so warm-started hybrid search is enabled immediately. +func NewDelegatedVector(dims int, delegate VectorDelegate, vectorCount, chunkCount int) *VectorBackend { + if delegate == nil { + panic("search.NewDelegatedVector: nil delegate") + } + if dims <= 0 { + panic("search.NewDelegatedVector: non-positive dimensions") + } + if vectorCount < 0 { + panic("search.NewDelegatedVector: negative vector count") + } + if chunkCount < 0 || chunkCount > vectorCount { + panic("search.NewDelegatedVector: invalid chunk count") + } + return &VectorBackend{ + dims: dims, + delegate: delegate, + delegateCount: vectorCount, + delegateChunkCount: chunkCount, + } +} + // SetChunkMap installs the chunk-vector → parent-symbol mapping. Called // by the indexer after a chunked vector build. A nil or empty map // means no symbol was sub-chunked. @@ -97,7 +104,7 @@ func (v *VectorBackend) ResolveChunk(id string) (string, bool) { func (v *VectorBackend) HasChunks() bool { v.mu.RLock() defer v.mu.RUnlock() - return len(v.chunkMap) > 0 + return v.delegateChunkCount > 0 || len(v.chunkMap) > 0 } // Add indexes a symbol with its embedding vector. @@ -105,12 +112,10 @@ func (v *VectorBackend) Add(id string, vector []float32) { v.mu.Lock() defer v.mu.Unlock() if v.delegate != nil { - // Delegated mode: the indexer pushes vectors to the - // engine-native HNSW via the graph.VectorSearcher - // interface directly. Add here is a no-op so the - // in-process hnsw.Graph never allocates memory for what - // the delegate already owns; count tracks deltas so - // Count()'s "is the index populated" gate fires. + // Legacy compatibility for callers that install a delegate before + // pushing vectors themselves. New publication must construct the + // backend with NewDelegatedVector and complete durable statistics. + // Either way Add never allocates an in-process HNSW in delegated mode. v.delegateCount++ return } @@ -121,40 +126,80 @@ func (v *VectorBackend) Add(id string, vector []float32) { v.count++ } -// SetDelegate routes Search / Count through an engine-native vector -// searcher (the disk store's graph.VectorSearcher). After -// the call: -// - Add is a no-op (the indexer talks to the delegate directly via -// graph.VectorSearcher.BulkUpsertEmbeddings / UpsertEmbedding), -// - Search forwards to delegate.SimilarTo, -// - Count reflects the delegate-delta count (not the in-process -// graph), so HybridBackend.searchChannels's `v.Count() > 0` gate -// fires once the indexer has populated the backend. +// SetDelegate is the legacy compatibility switch for callers that select a +// delegate before writing their corpus. It releases all prior heap and chunk +// state atomically; subsequent Add calls only advance this process's accepted +// write count. New and warm-start publication must use NewDelegatedVector so +// Count and HasChunks describe the complete durable corpus immediately. func (v *VectorBackend) SetDelegate(d VectorDelegate) { v.mu.Lock() defer v.mu.Unlock() + + // Switching modes must also release the old heap index. Merely routing + // reads to d would leave the complete HNSW graph, its count, and chunk + // ownership map retained for the lifetime of the daemon. + v.graph = nil + v.count = 0 + v.chunkMap = nil v.delegate = d + v.delegateCount = 0 + v.delegateChunkCount = 0 } // Search returns the k nearest neighbors to the query vector. func (v *VectorBackend) Search(query []float32, k int) []string { + if k <= 0 { + return nil + } v.mu.RLock() d := v.delegate + delegateCount := v.delegateCount + delegateChunks := v.delegateChunkCount v.mu.RUnlock() if d != nil { - hits, err := d.SimilarTo(query, k) + // Similarity rows are vector-ranked, but several top rows may be sibling + // chunks of one parent. Overfetch by the complete chunk population (capped + // by the corpus size), then stop after k unique symbols so de-duplication + // cannot silently under-fill an otherwise sufficient result set. + fetch := k + if delegateChunks > 0 { + if delegateCount-k <= delegateChunks { + fetch = delegateCount + } else { + fetch = k + delegateChunks + } + if fetch < k { + fetch = k + } + } + hits, err := d.SimilarTo(query, fetch) if err != nil || len(hits) == 0 { return nil } - ids := make([]string, len(hits)) - for i, h := range hits { - ids[i] = h.NodeID + ids := make([]string, 0, min(k, len(hits))) + seen := make(map[string]struct{}, len(hits)) + for _, h := range hits { + id := h.NodeID + if h.ParentID != "" { + id = h.ParentID + } + if id == "" { + continue + } + if _, duplicate := seen[id]; duplicate { + continue + } + seen[id] = struct{}{} + ids = append(ids, id) + if len(ids) == k { + break + } } return ids } v.mu.RLock() defer v.mu.RUnlock() - if v.count == 0 { + if v.count == 0 || v.graph == nil { return nil } results := v.graph.Search(hnsw.Vector(query), k) @@ -189,7 +234,7 @@ func (v *VectorBackend) Dims() int { return v.dims } func (v *VectorBackend) SizeBytes() uint64 { v.mu.RLock() defer v.mu.RUnlock() - if v.count == 0 { + if v.delegate != nil || v.count == 0 || v.graph == nil { return 0 } const hnswOverhead = 5900 // neighbor lists + map headers + priority-queue slack @@ -197,80 +242,19 @@ func (v *VectorBackend) SizeBytes() uint64 { return uint64(v.count) * perVector } -// Save writes the vector index to a writer in the framed format: -// magic + chunk-map JSON + raw HNSW export. The chunk map is persisted -// so query-time de-chunking still works after a daemon restart. -func (v *VectorBackend) Save(w io.Writer) error { - v.mu.RLock() - defer v.mu.RUnlock() - - mapJSON := []byte("{}") - if len(v.chunkMap) > 0 { - b, err := json.Marshal(v.chunkMap) - if err != nil { - return fmt.Errorf("marshal chunk map: %w", err) - } - mapJSON = b - } - if _, err := w.Write(vectorFrameMagic[:]); err != nil { - return fmt.Errorf("write vector frame magic: %w", err) - } - var lenBuf [4]byte - binary.LittleEndian.PutUint32(lenBuf[:], uint32(len(mapJSON))) - if _, err := w.Write(lenBuf[:]); err != nil { - return fmt.Errorf("write chunk map length: %w", err) - } - if _, err := w.Write(mapJSON); err != nil { - return fmt.Errorf("write chunk map: %w", err) - } - if err := v.graph.Export(w); err != nil { - return fmt.Errorf("export vector index: %w", err) +// Close releases all process-local vector state. A delegate is externally +// owned (normally by the durable graph store), so it is detached but never +// closed here. +func (v *VectorBackend) Close() { + if v == nil { + return } - return nil -} - -// LoadFrom restores the vector index from a reader. It accepts both -// the framed format (magic + chunk map + HNSW) and the legacy raw -// HNSW export written before AST sub-chunking shipped — a legacy blob -// has no magic and loads with an empty chunk map. -func (v *VectorBackend) LoadFrom(r io.Reader) error { v.mu.Lock() defer v.mu.Unlock() - - // Buffer the whole blob so a missing magic can be replayed into the - // HNSW importer. Vector index blobs are small relative to the graph. - all, err := io.ReadAll(r) - if err != nil { - return fmt.Errorf("read vector index: %w", err) - } - hnswBytes := all + v.graph = nil + v.count = 0 v.chunkMap = nil - if len(all) >= 8 && bytes.Equal(all[:4], vectorFrameMagic[:]) { - mapLen := binary.LittleEndian.Uint32(all[4:8]) - if int(mapLen)+8 > len(all) { - return fmt.Errorf("vector index frame: chunk map length %d exceeds blob", mapLen) - } - mapJSON := all[8 : 8+mapLen] - hnswBytes = all[8+mapLen:] - if mapLen > 0 { - m := make(map[string]string) - if err := json.Unmarshal(mapJSON, &m); err != nil { - return fmt.Errorf("unmarshal chunk map: %w", err) - } - if len(m) > 0 { - v.chunkMap = m - } - } - } - if err := v.graph.Import(bytes.NewReader(hnswBytes)); err != nil { - return fmt.Errorf("import vector index: %w", err) - } - return nil -} - -// SetCount sets the node count (used after loading from persistence). -func (v *VectorBackend) SetCount(n int) { - v.mu.Lock() - defer v.mu.Unlock() - v.count = n + v.delegate = nil + v.delegateCount = 0 + v.delegateChunkCount = 0 } diff --git a/internal/search/vector_publication_test.go b/internal/search/vector_publication_test.go new file mode 100644 index 000000000..1ed7f710d --- /dev/null +++ b/internal/search/vector_publication_test.go @@ -0,0 +1,451 @@ +package search + +import ( + "context" + "reflect" + "sync" + "testing" + "time" + + "github.com/zzet/gortex/internal/graph" +) + +type publicationTestDelegate struct { + mu sync.Mutex + hits []graph.VectorHit + limits []int +} + +func (d *publicationTestDelegate) SimilarTo(_ []float32, limit int) ([]graph.VectorHit, error) { + d.mu.Lock() + defer d.mu.Unlock() + d.limits = append(d.limits, limit) + if limit < len(d.hits) { + return append([]graph.VectorHit(nil), d.hits[:limit]...), nil + } + return append([]graph.VectorHit(nil), d.hits...), nil +} + +type publicationTestEmbedder struct { + started chan struct{} + release chan struct{} + once sync.Once +} + +func (e *publicationTestEmbedder) Embed(_ context.Context, _ string) ([]float32, error) { + if e.started != nil { + e.once.Do(func() { close(e.started) }) + <-e.release + } + return []float32{1, 0}, nil +} + +func (e *publicationTestEmbedder) EmbedBatch(_ context.Context, texts []string) ([][]float32, error) { + vectors := make([][]float32, len(texts)) + for i := range vectors { + vectors[i] = []float32{1, 0} + } + return vectors, nil +} + +func (*publicationTestEmbedder) Dimensions() int { return 2 } +func (*publicationTestEmbedder) Close() error { return nil } + +type publicationTestTextBackend struct { + mu sync.Mutex + closeCount int +} + +func (*publicationTestTextBackend) Add(string, ...string) {} +func (*publicationTestTextBackend) Remove(string) {} +func (*publicationTestTextBackend) Search(string, int) []SearchResult { return nil } +func (*publicationTestTextBackend) Count() int { return 1 } + +func (b *publicationTestTextBackend) Close() { + b.mu.Lock() + b.closeCount++ + b.mu.Unlock() +} + +func (b *publicationTestTextBackend) closes() int { + b.mu.Lock() + defer b.mu.Unlock() + return b.closeCount +} + +func TestNewDelegatedVectorHasNoHeapIndexAndUsesDurableStats(t *testing.T) { + delegate := &publicationTestDelegate{hits: []graph.VectorHit{ + {NodeID: "symbol-a#chunk0", ParentID: "symbol-a"}, + {NodeID: "symbol-a#chunk1", ParentID: "symbol-a"}, + {NodeID: "symbol-b"}, + }} + backend := NewDelegatedVector(384, delegate, 17, 2) + + if backend.graph != nil { + t.Fatal("delegated backend allocated an in-process HNSW graph") + } + if got := backend.Count(); got != 17 { + t.Fatalf("Count() = %d, want complete durable count 17", got) + } + if !backend.HasChunks() { + t.Fatal("HasChunks() = false, want durable chunk metadata") + } + if got := backend.SizeBytes(); got != 0 { + t.Fatalf("SizeBytes() = %d, want zero process-local vector bytes", got) + } + if got, want := backend.Search([]float32{1, 0}, 8), []string{"symbol-a", "symbol-b"}; !reflect.DeepEqual(got, want) { + t.Fatalf("Search() = %v, want parent-deduplicated %v", got, want) + } +} + +func TestDelegatedVectorSearchOverfetchesForUniqueParents(t *testing.T) { + delegate := &publicationTestDelegate{hits: []graph.VectorHit{ + {NodeID: "symbol-a#chunk0", ParentID: "symbol-a"}, + {NodeID: "symbol-a#chunk1", ParentID: "symbol-a"}, + {NodeID: "symbol-b"}, + {NodeID: "symbol-c"}, + }} + backend := NewDelegatedVector(2, delegate, 4, 2) + + if got, want := backend.Search([]float32{1, 0}, 3), []string{"symbol-a", "symbol-b", "symbol-c"}; !reflect.DeepEqual(got, want) { + t.Fatalf("Search() = %v, want k unique parent IDs %v", got, want) + } + + delegate.mu.Lock() + limits := append([]int(nil), delegate.limits...) + delegate.mu.Unlock() + if got, want := limits, []int{4}; !reflect.DeepEqual(got, want) { + t.Fatalf("delegate limits = %v, want min(delegateCount, k+chunkCount) = %v", got, want) + } +} + +func TestDelegatedVectorSearchRejectsNonPositiveLimitWithoutDelegateCall(t *testing.T) { + delegate := &publicationTestDelegate{hits: []graph.VectorHit{{NodeID: "symbol-a"}}} + backend := NewDelegatedVector(2, delegate, 1, 0) + + for _, k := range []int{0, -1} { + if got := backend.Search([]float32{1, 0}, k); got != nil { + t.Fatalf("Search(k=%d) = %v, want nil", k, got) + } + } + delegate.mu.Lock() + limits := append([]int(nil), delegate.limits...) + delegate.mu.Unlock() + if len(limits) != 0 { + t.Fatalf("delegate called with limits %v for non-positive k", limits) + } +} + +func TestSetDelegateReleasesHeapAndChunkState(t *testing.T) { + backend := NewVector(2) + backend.Add("symbol-a#chunk0", []float32{1, 0}) + backend.Add("symbol-b", []float32{0, 1}) + backend.SetChunkMap(map[string]string{"symbol-a#chunk0": "symbol-a"}) + if backend.SizeBytes() == 0 { + t.Fatal("heap backend reported no retained vector bytes before delegation") + } + + backend.SetDelegate(&publicationTestDelegate{}) + + if backend.graph != nil { + t.Fatal("SetDelegate retained the old HNSW graph") + } + if backend.chunkMap != nil { + t.Fatal("SetDelegate retained the old chunk ownership map") + } + if got := backend.Count(); got != 0 { + t.Fatalf("Count() = %d after mode switch, want fresh delegated count", got) + } + if backend.HasChunks() { + t.Fatal("SetDelegate retained old chunk statistics") + } + if got := backend.SizeBytes(); got != 0 { + t.Fatalf("SizeBytes() = %d after delegation, want zero", got) + } + + // Preserve the legacy build path: Add records successfully accepted + // delegate-side writes without rebuilding an HNSW graph. + backend.Add("symbol-c", []float32{1, 0}) + if got := backend.Count(); got != 1 { + t.Fatalf("Count() = %d after delegated Add, want 1", got) + } + if backend.graph != nil { + t.Fatal("delegated Add allocated an HNSW graph") + } +} + +func TestSerializeVectorUpdateSerializesConcurrentCallbacks(t *testing.T) { + swappable := NewSwappable(&publicationTestTextBackend{}) + firstStarted := make(chan struct{}) + releaseFirst := make(chan struct{}) + firstDone := make(chan struct{}) + go func() { + _ = swappable.SerializeVectorUpdate(func() error { + close(firstStarted) + <-releaseFirst + return nil + }) + close(firstDone) + }() + awaitPublicationSignal(t, "first callback start", firstStarted) + + secondAttempted := make(chan struct{}) + secondStarted := make(chan struct{}) + secondDone := make(chan struct{}) + go func() { + close(secondAttempted) + _ = swappable.SerializeVectorUpdate(func() error { + close(secondStarted) + return nil + }) + close(secondDone) + }() + awaitPublicationSignal(t, "second callback attempt", secondAttempted) + + select { + case <-secondStarted: + t.Fatal("second vector-update callback started while the first was active") + case <-time.After(50 * time.Millisecond): + } + + close(releaseFirst) + awaitPublicationSignal(t, "first callback completion", firstDone) + awaitPublicationSignal(t, "second callback start", secondStarted) + awaitPublicationSignal(t, "second callback completion", secondDone) + swappable.Close() +} + +func TestCloseWaitsForActiveVectorUpdate(t *testing.T) { + text := &publicationTestTextBackend{} + swappable := NewSwappable(text) + updateStarted := make(chan struct{}) + releaseUpdate := make(chan struct{}) + updateDone := make(chan struct{}) + go func() { + _ = swappable.SerializeVectorUpdate(func() error { + close(updateStarted) + <-releaseUpdate + return nil + }) + close(updateDone) + }() + awaitPublicationSignal(t, "vector-update callback start", updateStarted) + + closeAttempted := make(chan struct{}) + closeDone := make(chan struct{}) + go func() { + close(closeAttempted) + swappable.Close() + close(closeDone) + }() + awaitPublicationSignal(t, "Close attempt", closeAttempted) + + select { + case <-closeDone: + t.Fatal("Close completed while a vector-update callback was active") + case <-time.After(50 * time.Millisecond): + } + if got := text.closes(); got != 0 { + t.Fatalf("backend close count = %d while vector update active, want 0", got) + } + + close(releaseUpdate) + awaitPublicationSignal(t, "vector-update callback completion", updateDone) + awaitPublicationSignal(t, "Close completion", closeDone) + if got := text.closes(); got != 1 { + t.Fatalf("backend close count = %d after vector update drained, want 1", got) + } +} + +func TestAcquireBackendPinsReplacementUntilRelease(t *testing.T) { + text := &publicationTestTextBackend{} + embedder := &publicationTestEmbedder{} + oldVector := NewVector(2) + oldVector.Add("old-symbol", []float32{1, 0}) + swappable := NewSwappable(NewHybrid(text, oldVector, embedder)) + + pinned, release := swappable.AcquireBackend() + if _, ok := pinned.(*HybridBackend); !ok { + t.Fatalf("AcquireBackend() returned %T, want *HybridBackend", pinned) + } + + newVector := NewDelegatedVector(2, &publicationTestDelegate{}, 1, 0) + replaceStarted := make(chan struct{}) + replaceDone := make(chan struct{}) + go func() { + close(replaceStarted) + swappable.ReplaceHybridVector(newVector, embedder) + close(replaceDone) + }() + <-replaceStarted + + select { + case <-replaceDone: + t.Fatal("replacement completed while AcquireBackend pin was held") + case <-time.After(50 * time.Millisecond): + } + if oldVector.SizeBytes() == 0 { + t.Fatal("pinned backend was retired before release") + } + + release() + release() // The release contract is intentionally idempotent. + select { + case <-replaceDone: + case <-time.After(2 * time.Second): + t.Fatal("replacement did not complete after AcquireBackend release") + } + if got := oldVector.SizeBytes(); got != 0 { + t.Fatalf("retired vector retains %d bytes after release", got) + } + + swappable.Close() + if got := text.closes(); got != 1 { + t.Fatalf("text backend close count = %d, want 1", got) + } +} + +func TestReplaceHybridVectorWaitsForReadersAndTransfersTextOwnership(t *testing.T) { + text := &publicationTestTextBackend{} + oldVector := NewVector(2) + oldVector.Add("old-symbol", []float32{1, 0}) + oldEmbedder := &publicationTestEmbedder{ + started: make(chan struct{}), + release: make(chan struct{}), + } + swappable := NewSwappable(NewHybrid(text, oldVector, oldEmbedder)) + + oldResults := make(chan []string, 1) + go func() { + ids, _ := swappable.VectorChannelOnly("old", 1) + oldResults <- ids + }() + <-oldEmbedder.started + + newDelegate := &publicationTestDelegate{hits: []graph.VectorHit{{ + NodeID: "new-symbol#chunk0", + ParentID: "new-symbol", + }}} + newVector := NewDelegatedVector(2, newDelegate, 1, 1) + newEmbedder := &publicationTestEmbedder{} + replaceStarted := make(chan struct{}) + replaceDone := make(chan struct{}) + go func() { + close(replaceStarted) + swappable.ReplaceHybridVector(newVector, newEmbedder) + close(replaceDone) + }() + <-replaceStarted + + select { + case <-replaceDone: + t.Fatal("replacement completed while an old-backend reader was active") + case <-time.After(50 * time.Millisecond): + } + if oldVector.SizeBytes() == 0 { + t.Fatal("old vector was retired before its active reader completed") + } + + close(oldEmbedder.release) + if got, want := <-oldResults, []string{"old-symbol"}; !reflect.DeepEqual(got, want) { + t.Fatalf("in-flight reader saw %v, want complete old result %v", got, want) + } + select { + case <-replaceDone: + case <-time.After(2 * time.Second): + t.Fatal("replacement did not complete after the old reader drained") + } + + if got := oldVector.SizeBytes(); got != 0 { + t.Fatalf("retired vector retains %d bytes, want zero", got) + } + if got := text.closes(); got != 0 { + t.Fatalf("transferred text backend closed %d times during replacement", got) + } + if got, want := vectorIDs(swappable, "new"), []string{"new-symbol"}; !reflect.DeepEqual(got, want) { + t.Fatalf("new reader saw %v, want complete new result %v", got, want) + } + + current, ok := swappable.Inner().(*HybridBackend) + if !ok { + t.Fatalf("active backend is %T, want *HybridBackend", swappable.Inner()) + } + if current.TextBackend() != text { + t.Fatal("replacement did not retain the original text backend") + } + if _, nested := current.TextBackend().(*HybridBackend); nested { + t.Fatal("replacement nested a HybridBackend inside another HybridBackend") + } + + swappable.Close() + if got := text.closes(); got != 1 { + t.Fatalf("text backend close count = %d, want exactly one final close", got) + } + if got := newVector.Count(); got != 0 { + t.Fatalf("final vector Count() = %d after close, want 0", got) + } +} + +func TestReplaceHybridVectorFlattensAndSupportsRepeatedReplacement(t *testing.T) { + text := &publicationTestTextBackend{} + embedder := &publicationTestEmbedder{} + first := NewVector(2) + first.Add("first", []float32{1, 0}) + second := NewVector(2) + second.Add("second", []float32{1, 0}) + legacyNested := NewHybrid(NewHybrid(text, first, embedder), second, embedder) + swappable := NewSwappable(legacyNested) + + third := NewDelegatedVector(2, &publicationTestDelegate{}, 3, 0) + swappable.ReplaceHybridVector(third, embedder) + assertSingleHybrid(t, swappable, text) + if first.Count() != 0 || second.Count() != 0 { + t.Fatalf("flattening did not retire both legacy vectors: first=%d second=%d", first.Count(), second.Count()) + } + + fourth := NewDelegatedVector(2, &publicationTestDelegate{}, 4, 1) + swappable.ReplaceHybridVector(fourth, embedder) + assertSingleHybrid(t, swappable, text) + if got := third.Count(); got != 0 { + t.Fatalf("repeated replacement retained prior delegated vector count %d", got) + } + if got := text.closes(); got != 0 { + t.Fatalf("text backend closed %d times before final Swappable.Close", got) + } + + swappable.Close() + if got := text.closes(); got != 1 { + t.Fatalf("text backend close count = %d, want 1", got) + } + if got := fourth.Count(); got != 0 { + t.Fatalf("last vector Count() = %d after close, want 0", got) + } +} + +func awaitPublicationSignal(t *testing.T, name string, signal <-chan struct{}) { + t.Helper() + select { + case <-signal: + case <-time.After(2 * time.Second): + t.Fatalf("timed out waiting for %s", name) + } +} + +func vectorIDs(swappable *Swappable, query string) []string { + ids, _ := swappable.VectorChannelOnly(query, 1) + return ids +} + +func assertSingleHybrid(t *testing.T, swappable *Swappable, wantText Backend) { + t.Helper() + hybrid, ok := swappable.Inner().(*HybridBackend) + if !ok { + t.Fatalf("active backend is %T, want *HybridBackend", swappable.Inner()) + } + if hybrid.TextBackend() != wantText { + t.Fatalf("hybrid text backend is %T, want original %T", hybrid.TextBackend(), wantText) + } + if _, nested := hybrid.TextBackend().(*HybridBackend); nested { + t.Fatal("active HybridBackend contains a nested HybridBackend") + } +} diff --git a/internal/semantic/enricher.go b/internal/semantic/enricher.go index eb2ef9b98..74cb8735e 100644 --- a/internal/semantic/enricher.go +++ b/internal/semantic/enricher.go @@ -24,19 +24,6 @@ func RefuteEdge(g graph.Store, e *graph.Edge) bool { return g.RemoveEdge(e.From, e.To, e.Kind) } -// PersistEdge round-trips an in-place edge mutation (ConfirmEdge, an -// Origin promotion, a Meta stamp) through the backend's edge-attribute -// write path. On the in-memory backend edge reads hand back the live -// *Edge pointer, so the mutation is already durable and this is a -// no-op. Disk backends return detached row copies — without this -// round-trip every enrichment-pass promotion silently evaporates and -// the read path keeps serving the stale heuristic tier. -func PersistEdge(g graph.Store, e *graph.Edge) { - if w, ok := g.(graph.EdgePersister); ok { - w.PersistEdgeAttributes(e) - } -} - // AddSemanticEdge adds a new edge discovered by semantic analysis. Origin is // tagged LSP-grade (see ConfirmEdge). func AddSemanticEdge(g graph.Store, from, to string, kind graph.EdgeKind, filePath string, line int, provider string) *graph.Edge { @@ -96,17 +83,6 @@ func FindMatchingEdge(g graph.Store, from, to string, kind graph.EdgeKind) *grap return nil } -// FindEdgeByTarget searches for an edge from a node to a target with any kind. -func FindEdgeByTarget(g graph.Store, from, to string) *graph.Edge { - edges := g.GetOutEdges(from) - for _, e := range edges { - if e.To == to { - return e - } - } - return nil -} - // NodesByLanguage returns all nodes in the graph that match the given language. func NodesByLanguage(g graph.Store, language string) []*graph.Node { return g.GetNodesByLanguage(language) diff --git a/internal/semantic/goanalysis/definition_match.go b/internal/semantic/goanalysis/definition_match.go new file mode 100644 index 000000000..87a42d466 --- /dev/null +++ b/internal/semantic/goanalysis/definition_match.go @@ -0,0 +1,482 @@ +package goanalysis + +import ( + "go/ast" + "go/token" + "go/types" + "sort" + "strconv" + "strings" + + "github.com/zzet/gortex/internal/graph" +) + +type goDefinitionRole uint8 + +const ( + goDefinitionUnknown goDefinitionRole = iota + goDefinitionFunction + goDefinitionMethod + goDefinitionField + goDefinitionReceiver + goDefinitionParam + goDefinitionResult + goDefinitionLocal + goDefinitionPackageVar + goDefinitionConstant + goDefinitionType + goDefinitionInterface + goDefinitionGenericParam +) + +// goDefinitionContext records syntax facts that go/types.Object does not +// expose, notably whether a *types.Var is a receiver, parameter, result, or +// local. owner is the graph-facing owner identity (for example Widget.Run). +type goDefinitionContext struct { + role goDefinitionRole + owner string +} + +// buildGoDefinitionContexts projects just enough AST parent information to +// classify TypesInfo.Defs entries. The map is scoped to one packages.Package +// and is discarded as soon as its definitions have been matched. +func buildGoDefinitionContexts(files []*ast.File) map[*ast.Ident]goDefinitionContext { + contexts := make(map[*ast.Ident]goDefinitionContext) + for _, file := range files { + if file == nil { + continue + } + for _, decl := range file.Decls { + switch decl := decl.(type) { + case *ast.GenDecl: + markGoGenDeclDefinitions(decl, true, "", contexts) + case *ast.FuncDecl: + markGoFuncDeclDefinitions(decl, contexts) + } + } + } + return contexts +} + +func markGoFuncDeclDefinitions(decl *ast.FuncDecl, contexts map[*ast.Ident]goDefinitionContext) { + if decl == nil || decl.Name == nil || decl.Type == nil { + return + } + + receiver := goASTReceiverTypeName(decl.Recv) + callableOwner := decl.Name.Name + role := goDefinitionFunction + if receiver != "" { + role = goDefinitionMethod + callableOwner = receiver + "." + decl.Name.Name + } + contexts[decl.Name] = goDefinitionContext{role: role, owner: receiver} + markGoFieldListDefinitions(decl.Recv, goDefinitionReceiver, callableOwner, contexts) + markGoFieldListDefinitions(decl.Type.TypeParams, goDefinitionGenericParam, callableOwner, contexts) + markGoFieldListDefinitions(decl.Type.Params, goDefinitionParam, callableOwner, contexts) + markGoFieldListDefinitions(decl.Type.Results, goDefinitionResult, callableOwner, contexts) + markGoLocalDefinitions(decl.Body, callableOwner, contexts) +} + +func markGoFieldListDefinitions(list *ast.FieldList, role goDefinitionRole, owner string, contexts map[*ast.Ident]goDefinitionContext) { + if list == nil { + return + } + for _, field := range list.List { + if field == nil { + continue + } + for _, name := range field.Names { + if name != nil { + contexts[name] = goDefinitionContext{role: role, owner: owner} + } + } + } +} + +func markGoGenDeclDefinitions(decl *ast.GenDecl, packageScope bool, owner string, contexts map[*ast.Ident]goDefinitionContext) { + if decl == nil { + return + } + for _, spec := range decl.Specs { + switch spec := spec.(type) { + case *ast.ValueSpec: + role := goDefinitionLocal + switch decl.Tok { + case token.CONST: + role = goDefinitionConstant + case token.VAR: + if packageScope { + role = goDefinitionPackageVar + } + } + for _, name := range spec.Names { + if name != nil { + contexts[name] = goDefinitionContext{role: role, owner: owner} + } + } + case *ast.TypeSpec: + markGoTypeSpecDefinitions(spec, contexts) + } + } +} + +func markGoTypeSpecDefinitions(spec *ast.TypeSpec, contexts map[*ast.Ident]goDefinitionContext) { + if spec == nil || spec.Name == nil { + return + } + role := goDefinitionType + if _, ok := spec.Type.(*ast.InterfaceType); ok { + role = goDefinitionInterface + } + contexts[spec.Name] = goDefinitionContext{role: role} + markGoFieldListDefinitions(spec.TypeParams, goDefinitionGenericParam, spec.Name.Name, contexts) + + switch typ := spec.Type.(type) { + case *ast.StructType: + markGoFieldListDefinitions(typ.Fields, goDefinitionField, spec.Name.Name, contexts) + case *ast.InterfaceType: + // Named interface fields are methods. Embedded fields have no Defs + // entry and are intentionally ignored here. + markGoFieldListDefinitions(typ.Methods, goDefinitionMethod, spec.Name.Name, contexts) + } +} + +func markGoLocalDefinitions(body *ast.BlockStmt, owner string, contexts map[*ast.Ident]goDefinitionContext) { + if body == nil { + return + } + ast.Inspect(body, func(node ast.Node) bool { + switch node := node.(type) { + case *ast.FuncLit: + // A closure has its own parameters/results and lexical body. Its + // graph owner is line-based rather than a source identifier, so do + // not borrow the outer callable identity for owner matching. + markGoFieldListDefinitions(node.Type.TypeParams, goDefinitionGenericParam, "", contexts) + markGoFieldListDefinitions(node.Type.Params, goDefinitionParam, "", contexts) + markGoFieldListDefinitions(node.Type.Results, goDefinitionResult, "", contexts) + markGoLocalDefinitions(node.Body, "", contexts) + return false + case *ast.AssignStmt: + if node.Tok == token.DEFINE { + for _, expr := range node.Lhs { + if ident, ok := expr.(*ast.Ident); ok { + contexts[ident] = goDefinitionContext{role: goDefinitionLocal, owner: owner} + } + } + } + case *ast.RangeStmt: + if node.Tok == token.DEFINE { + for _, expr := range []ast.Expr{node.Key, node.Value} { + if ident, ok := expr.(*ast.Ident); ok { + contexts[ident] = goDefinitionContext{role: goDefinitionLocal, owner: owner} + } + } + } + case *ast.DeclStmt: + if decl, ok := node.Decl.(*ast.GenDecl); ok { + markGoGenDeclDefinitions(decl, false, owner, contexts) + } + return false + } + return true + }) +} + +func goASTReceiverTypeName(list *ast.FieldList) string { + if list == nil || len(list.List) == 0 || list.List[0] == nil { + return "" + } + return goASTTypeName(list.List[0].Type) +} + +func goASTTypeName(expr ast.Expr) string { + switch expr := expr.(type) { + case *ast.Ident: + return expr.Name + case *ast.StarExpr: + return goASTTypeName(expr.X) + case *ast.ParenExpr: + return goASTTypeName(expr.X) + case *ast.IndexExpr: + return goASTTypeName(expr.X) + case *ast.IndexListExpr: + return goASTTypeName(expr.X) + case *ast.SelectorExpr: + return expr.Sel.Name + default: + return "" + } +} + +type goDefinitionCandidate struct { + node *graph.Node + kindRank int + ownerRank int + rangeRank int + distance int + span int + owner string + stableKey string +} + +// matchRepoDefinitionNode matches one compiler definition to a graph node only +// after its source identity is compatible. Name and object kind are mandatory; +// receiver/owner identity is enforced whenever both sides expose it. Location +// ranks compatible candidates but can never turn a parameter/local into a +// function target. Ambiguous semantic identities fail closed. +func matchRepoDefinitionNode(nodes []*graph.Node, position token.Position, identifier string, obj types.Object, context goDefinitionContext) *graph.Node { + if obj == nil || identifier == "" || position.Line <= 0 { + return nil + } + kindRanks := goDefinitionKindRanks(obj, context) + if len(kindRanks) == 0 { + return nil + } + expectedOwner := goDefinitionExpectedOwner(obj, context) + + candidates := make([]goDefinitionCandidate, 0, 2) + for _, node := range nodes { + if node == nil || node.Kind == graph.KindFile || node.Kind == graph.KindImport || node.Name != identifier { + continue + } + kindRank, compatible := kindRanks[node.Kind] + if !compatible { + continue + } + + owner := goDefinitionNodeOwner(node) + ownerRank := 0 + if expectedOwner != "" { + switch owner { + case expectedOwner: + case "": + ownerRank = 1 + default: + continue + } + } + + endLine := node.EndLine + if endLine < node.StartLine { + endLine = node.StartLine + } + rangeRank := 0 + distance := 0 + if node.StartLine <= position.Line && position.Line <= endLine { + // Containing declarations rank before the constrained nearest + // fallback, with the innermost source span winning. + } else { + rangeRank = 1 + distance = goDefinitionAbs(node.StartLine - position.Line) + if distance > 2 { + continue + } + } + + candidates = append(candidates, goDefinitionCandidate{ + node: node, + kindRank: kindRank, + ownerRank: ownerRank, + rangeRank: rangeRank, + distance: distance, + span: endLine - node.StartLine, + owner: owner, + stableKey: goDefinitionStableNodeKey(node), + }) + } + if len(candidates) == 0 { + return nil + } + + sort.Slice(candidates, func(i, j int) bool { + left, right := candidates[i], candidates[j] + for _, pair := range [][2]int{ + {left.kindRank, right.kindRank}, + {left.ownerRank, right.ownerRank}, + {left.rangeRank, right.rangeRank}, + {left.distance, right.distance}, + {left.span, right.span}, + } { + if pair[0] != pair[1] { + return pair[0] < pair[1] + } + } + return left.stableKey < right.stableKey + }) + + // Stable ID ordering resolves duplicate projections of the same semantic + // declaration. Equal-ranked candidates with different kinds or owners are + // genuinely ambiguous and must not be guessed. + if len(candidates) > 1 && goDefinitionSameRank(candidates[0], candidates[1]) { + first, second := candidates[0], candidates[1] + if first.node.Kind != second.node.Kind || first.owner != second.owner { + return nil + } + } + return candidates[0].node +} + +func goDefinitionKindRanks(obj types.Object, context goDefinitionContext) map[graph.NodeKind]int { + ranks := make(map[graph.NodeKind]int, 2) + switch obj := obj.(type) { + case *types.Func: + isMethod := context.role == goDefinitionMethod + if signature, ok := obj.Type().(*types.Signature); ok && signature.Recv() != nil { + isMethod = true + } + if isMethod { + ranks[graph.KindMethod] = 0 + } else { + ranks[graph.KindFunction] = 0 + } + case *types.Const: + ranks[graph.KindConstant] = 0 + case *types.TypeName: + if context.role == goDefinitionGenericParam { + ranks[graph.KindGenericParam] = 0 + break + } + isInterface := context.role == goDefinitionInterface + if obj.Type() != nil { + if _, ok := obj.Type().Underlying().(*types.Interface); ok { + isInterface = true + } + } + if isInterface { + ranks[graph.KindInterface] = 0 + // Type aliases and older projections may retain the general type + // kind. It is compatible, but never preferred over an interface. + ranks[graph.KindType] = 1 + } else { + ranks[graph.KindType] = 0 + } + case *types.Var: + switch context.role { + case goDefinitionField: + ranks[graph.KindField] = 0 + case goDefinitionParam: + ranks[graph.KindParam] = 0 + case goDefinitionLocal: + ranks[graph.KindLocal] = 0 + case goDefinitionPackageVar: + ranks[graph.KindVariable] = 0 + case goDefinitionReceiver, goDefinitionResult: + // Receivers and named results currently have no dedicated graph + // definition nodes. Mapping either to a nearby param/local would + // fabricate identity, so leave them unmapped. + return ranks + default: + switch { + case obj.IsField(): + ranks[graph.KindField] = 0 + case obj.Pkg() != nil && obj.Parent() == obj.Pkg().Scope(): + ranks[graph.KindVariable] = 0 + default: + // Without AST role information a non-package variable might be + // either a parameter or a local. Permit a unique exact candidate, + // but the ambiguity guard rejects equal-ranked cross-kind matches. + ranks[graph.KindParam] = 0 + ranks[graph.KindLocal] = 0 + } + } + } + return ranks +} + +func goDefinitionExpectedOwner(obj types.Object, context goDefinitionContext) string { + switch obj := obj.(type) { + case *types.Func: + if signature, ok := obj.Type().(*types.Signature); ok && signature.Recv() != nil { + if receiver := goTypesReceiverName(signature.Recv().Type()); receiver != "" { + return receiver + } + } + if context.role == goDefinitionMethod { + return context.owner + } + case *types.Var: + switch context.role { + case goDefinitionField, goDefinitionParam, goDefinitionLocal: + return context.owner + } + case *types.TypeName: + if context.role == goDefinitionGenericParam { + return context.owner + } + } + return "" +} + +func goTypesReceiverName(typ types.Type) string { + switch typ := typ.(type) { + case *types.Pointer: + return goTypesReceiverName(typ.Elem()) + case *types.Named: + if typ.Obj() != nil { + return typ.Obj().Name() + } + } + return "" +} + +func goDefinitionNodeOwner(node *graph.Node) string { + if node == nil { + return "" + } + switch node.Kind { + case graph.KindMethod, graph.KindField: + _, owner := graph.EnclosingFromID(node.ID, node.Kind) + return owner + case graph.KindParam: + return goDefinitionOwnerBeforeMarker(node.ID, "#param:") + case graph.KindLocal: + return goDefinitionOwnerBeforeMarker(node.ID, "#local:") + case graph.KindGenericParam: + return goDefinitionOwnerBeforeMarker(node.ID, "#tparam:") + default: + return "" + } +} + +func goDefinitionOwnerBeforeMarker(id, marker string) string { + index := strings.LastIndex(id, marker) + if index < 0 { + return "" + } + owner := id[:index] + if scope := strings.LastIndex(owner, "::"); scope >= 0 { + owner = owner[scope+2:] + } + return owner +} + +func goDefinitionStableNodeKey(node *graph.Node) string { + if node.ID != "" { + return node.ID + } + return strings.Join([]string{ + string(node.Kind), + node.QualName, + node.Name, + strconv.Itoa(node.StartLine), + strconv.Itoa(node.EndLine), + strconv.Itoa(node.StartColumn), + strconv.Itoa(node.EndColumn), + }, "\x00") +} + +func goDefinitionSameRank(left, right goDefinitionCandidate) bool { + return left.kindRank == right.kindRank && + left.ownerRank == right.ownerRank && + left.rangeRank == right.rangeRank && + left.distance == right.distance && + left.span == right.span +} + +func goDefinitionAbs(value int) int { + if value < 0 { + return -value + } + return value +} diff --git a/internal/semantic/goanalysis/definition_match_test.go b/internal/semantic/goanalysis/definition_match_test.go new file mode 100644 index 000000000..13cd6aa33 --- /dev/null +++ b/internal/semantic/goanalysis/definition_match_test.go @@ -0,0 +1,270 @@ +package goanalysis + +import ( + "go/ast" + "go/importer" + goparser "go/parser" + "go/token" + "go/types" + "path/filepath" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/zzet/gortex/internal/graph" + "github.com/zzet/gortex/internal/graph/store_sqlite" +) + +type checkedGoDefinition struct { + ident *ast.Ident + object types.Object + position token.Position + context goDefinitionContext +} + +func checkedGoDefinitions(t *testing.T, source string) []checkedGoDefinition { + t.Helper() + fset := token.NewFileSet() + file, err := goparser.ParseFile(fset, "sample.go", source, 0) + require.NoError(t, err) + info := &types.Info{Defs: make(map[*ast.Ident]types.Object)} + _, err = (&types.Config{Importer: importer.Default()}).Check("example.com/sample", fset, []*ast.File{file}, info) + require.NoError(t, err) + contexts := buildGoDefinitionContexts([]*ast.File{file}) + definitions := make([]checkedGoDefinition, 0, len(info.Defs)) + for ident, object := range info.Defs { + if object == nil { + continue + } + definitions = append(definitions, checkedGoDefinition{ + ident: ident, + object: object, + position: fset.Position(ident.Pos()), + context: contexts[ident], + }) + } + return definitions +} + +func requireCheckedGoDefinition(t *testing.T, definitions []checkedGoDefinition, name string, line int) checkedGoDefinition { + t.Helper() + for _, definition := range definitions { + if definition.ident.Name == name && definition.position.Line == line { + return definition + } + } + t.Fatalf("definition %s at line %d not found", name, line) + return checkedGoDefinition{} +} + +func TestMatchRepoDefinitionNodeRejectsSameLineParameterForFunction(t *testing.T) { + definitions := checkedGoDefinitions(t, "package sample\nfunc Hello(name string) {}\n") + hello := requireCheckedGoDefinition(t, definitions, "Hello", 2) + functionID := "sample.go::Hello" + nodes := []*graph.Node{ + nil, + {ID: "sample.go", Kind: graph.KindFile, Name: "Hello", StartLine: 2}, + {ID: "sample.go::import", Kind: graph.KindImport, Name: "Hello", StartLine: 2}, + {ID: functionID + "#param:name", Kind: graph.KindParam, Name: "name", StartLine: 2}, + {ID: functionID, Kind: graph.KindFunction, Name: "Hello", StartLine: 2, EndLine: 2}, + } + + matched := matchRepoDefinitionNode(nodes, hello.position, hello.ident.Name, hello.object, hello.context) + require.NotNil(t, matched) + assert.Equal(t, functionID, matched.ID) +} + +func TestMatchRepoDefinitionNodeObjectRolesAndOwners(t *testing.T) { + const source = `package sample + +type Widget struct { + Field int +} + +type Contract interface { + Call() +} + +const Answer = 42 +var Global int + +func (w Widget) Method(param int) (result int) { + local := param + _ = local + var declared int + _ = declared + return +} +` + definitions := checkedGoDefinitions(t, source) + nodes := []*graph.Node{ + {ID: "sample.go::Widget", Kind: graph.KindType, Name: "Widget", StartLine: 3, EndLine: 5}, + {ID: "sample.go::Widget.Field", Kind: graph.KindField, Name: "Field", StartLine: 4, EndLine: 4}, + {ID: "sample.go::Contract", Kind: graph.KindInterface, Name: "Contract", StartLine: 7, EndLine: 9}, + {ID: "sample.go::Contract.Call", Kind: graph.KindMethod, Name: "Call", StartLine: 8, EndLine: 8}, + {ID: "sample.go::Answer", Kind: graph.KindConstant, Name: "Answer", StartLine: 11, EndLine: 11}, + {ID: "sample.go::Global", Kind: graph.KindVariable, Name: "Global", StartLine: 12, EndLine: 12}, + {ID: "sample.go::Other.Method", Kind: graph.KindMethod, Name: "Method", StartLine: 14, EndLine: 20}, + {ID: "sample.go::Widget.Method", Kind: graph.KindMethod, Name: "Method", StartLine: 14, EndLine: 20}, + {ID: "sample.go::Widget.Method#param:w", Kind: graph.KindParam, Name: "w", StartLine: 14}, + {ID: "sample.go::Widget.Method#param:param", Kind: graph.KindParam, Name: "param", StartLine: 14}, + {ID: "sample.go::Widget.Method#param:result", Kind: graph.KindParam, Name: "result", StartLine: 14}, + {ID: "sample.go::Widget.Method#local:local@+1", Kind: graph.KindLocal, Name: "local", StartLine: 15}, + {ID: "sample.go::Widget.Method#local:declared@+3", Kind: graph.KindLocal, Name: "declared", StartLine: 17}, + } + + for _, test := range []struct { + name string + line int + wantID string + }{ + {name: "Widget", line: 3, wantID: "sample.go::Widget"}, + {name: "Field", line: 4, wantID: "sample.go::Widget.Field"}, + {name: "Contract", line: 7, wantID: "sample.go::Contract"}, + {name: "Call", line: 8, wantID: "sample.go::Contract.Call"}, + {name: "Answer", line: 11, wantID: "sample.go::Answer"}, + {name: "Global", line: 12, wantID: "sample.go::Global"}, + {name: "Method", line: 14, wantID: "sample.go::Widget.Method"}, + {name: "param", line: 14, wantID: "sample.go::Widget.Method#param:param"}, + {name: "local", line: 15, wantID: "sample.go::Widget.Method#local:local@+1"}, + {name: "declared", line: 17, wantID: "sample.go::Widget.Method#local:declared@+3"}, + {name: "w", line: 14, wantID: ""}, + {name: "result", line: 14, wantID: ""}, + } { + t.Run(test.name, func(t *testing.T) { + definition := requireCheckedGoDefinition(t, definitions, test.name, test.line) + matched := matchRepoDefinitionNode(nodes, definition.position, definition.ident.Name, definition.object, definition.context) + if test.wantID == "" { + assert.Nil(t, matched) + return + } + require.NotNil(t, matched) + assert.Equal(t, test.wantID, matched.ID) + }) + } +} + +func TestMatchRepoDefinitionNodeRangeAndDeterminism(t *testing.T) { + pkg := types.NewPackage("example.com/sample", "sample") + object := types.NewFunc(token.NoPos, pkg, "Target", types.NewSignatureType(nil, nil, nil, nil, nil, false)) + position := token.Position{Filename: "sample.go", Line: 10} + node := func(id string, kind graph.NodeKind, name string, start, end int) *graph.Node { + return &graph.Node{ID: id, Kind: kind, Name: name, StartLine: start, EndLine: end} + } + + for _, test := range []struct { + name string + nodes []*graph.Node + want string + }{ + {name: "exact", nodes: []*graph.Node{node("exact", graph.KindFunction, "Target", 10, 10)}, want: "exact"}, + {name: "innermost containing", nodes: []*graph.Node{ + node("outer", graph.KindFunction, "Target", 8, 14), + node("inner", graph.KindFunction, "Target", 9, 11), + }, want: "inner"}, + {name: "nearest one", nodes: []*graph.Node{node("one", graph.KindFunction, "Target", 11, 11)}, want: "one"}, + {name: "nearest two", nodes: []*graph.Node{node("two", graph.KindFunction, "Target", 12, 12)}, want: "two"}, + {name: "greater than two", nodes: []*graph.Node{node("three", graph.KindFunction, "Target", 13, 13)}}, + {name: "wrong name", nodes: []*graph.Node{node("wrong-name", graph.KindFunction, "Other", 10, 10)}}, + {name: "wrong kind", nodes: []*graph.Node{node("wrong-kind", graph.KindParam, "Target", 10, 10)}}, + {name: "stable node identity", nodes: []*graph.Node{ + node("z-target", graph.KindFunction, "Target", 10, 10), + node("a-target", graph.KindFunction, "Target", 10, 10), + }, want: "a-target"}, + } { + t.Run(test.name, func(t *testing.T) { + matched := matchRepoDefinitionNode(test.nodes, position, "Target", object, goDefinitionContext{}) + if test.want == "" { + assert.Nil(t, matched) + return + } + require.NotNil(t, matched) + assert.Equal(t, test.want, matched.ID) + }) + } + + variable := types.NewVar(token.NoPos, pkg, "value", types.Typ[types.Int]) + ambiguous := matchRepoDefinitionNode([]*graph.Node{ + node("sample.go::F#param:value", graph.KindParam, "value", 10, 10), + node("sample.go::F#local:value@+0", graph.KindLocal, "value", 10, 10), + }, position, "value", variable, goDefinitionContext{}) + assert.Nil(t, ambiguous, "an unknown compiler variable role must fail closed across graph kinds") +} + +func TestGoAnalysisDefinitionTargetsMatchAcrossMemoryAndSQLite(t *testing.T) { + for _, backend := range []string{"memory", "sqlite"} { + t.Run(backend, func(t *testing.T) { + var store graph.Store + switch backend { + case "memory": + store = graph.New() + case "sqlite": + sqliteStore, err := store_sqlite.Open(filepath.Join(t.TempDir(), "semantic.sqlite")) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, sqliteStore.Close()) }) + store = sqliteStore + } + assertGoAnalysisDefinitionTargets(t, store) + }) + } +} + +func assertGoAnalysisDefinitionTargets(t *testing.T, store graph.Store) { + t.Helper() + root := resolvedTempDir(t) + writeGoMod(t, root, "example.com/definitiontarget") + writeFile(t, root, "main.go", `package sample + +func Hello(name string) {} + +func UseConfirmed() { Hello("confirmed") } + +func UseAdded() { Hello("added") } +`) + + const ( + helloID = "main.go::Hello" + parameterID = helloID + "#param:name" + confirmedID = "main.go::UseConfirmed" + addedID = "main.go::UseAdded" + ) + store.AddBatch([]*graph.Node{ + {ID: parameterID, Kind: graph.KindParam, Name: "name", FilePath: "main.go", StartLine: 3, Language: "go"}, + {ID: helloID, Kind: graph.KindFunction, Name: "Hello", FilePath: "main.go", StartLine: 3, EndLine: 3, Language: "go"}, + {ID: confirmedID, Kind: graph.KindFunction, Name: "UseConfirmed", FilePath: "main.go", StartLine: 5, EndLine: 5, Language: "go"}, + {ID: addedID, Kind: graph.KindFunction, Name: "UseAdded", FilePath: "main.go", StartLine: 7, EndLine: 7, Language: "go"}, + }, []*graph.Edge{{ + From: confirmedID, + To: helloID, + Kind: graph.EdgeCalls, + FilePath: "main.go", + Line: 5, + Confidence: 0.4, + ConfidenceLabel: "INFERRED", + Origin: graph.OriginASTInferred, + }}) + + provider := newTestProvider(t) + t.Cleanup(func() { require.NoError(t, provider.Close()) }) + result, err := provider.Enrich(store, root) + require.NoError(t, err) + require.GreaterOrEqual(t, result.EdgesConfirmed, 1) + require.GreaterOrEqual(t, result.EdgesAdded, 1) + + for _, callerID := range []string{confirmedID, addedID} { + var correct *graph.Edge + for _, edge := range store.GetOutEdges(callerID) { + if edge.Kind != graph.EdgeCalls { + continue + } + assert.NotEqual(t, parameterID, edge.To, "compiler call target must never be the same-line parameter") + if edge.To == helloID { + correct = edge + } + } + require.NotNil(t, correct, "semantic call from %s must target Hello", callerID) + assert.Equal(t, 1.0, correct.Confidence) + assert.Equal(t, graph.OriginLSPResolved, correct.Origin) + } +} diff --git a/internal/semantic/goanalysis/provider.go b/internal/semantic/goanalysis/provider.go index 50df8c9cf..746efc1d6 100644 --- a/internal/semantic/goanalysis/provider.go +++ b/internal/semantic/goanalysis/provider.go @@ -727,6 +727,7 @@ func (p *Provider) enrichRepoContext(ctx context.Context, g graph.Store, repoPre continue } + definitionContexts := buildGoDefinitionContexts(pkg.Syntax) for ident, obj := range pkg.TypesInfo.Defs { if obj == nil || ident.Pos() == token.NoPos { continue @@ -740,10 +741,7 @@ func (p *Provider) enrichRepoContext(ctx context.Context, g graph.Store, repoPre graphPath := scopedGraphPath(repoPrefix, relPath) fileNodes := nodesByFile[graphPath] - node := matchRepoNodeByFileLine(fileNodes, pos.Line) - if node == nil { - node = matchRepoNodeByName(fileNodes, ident.Name) - } + node := matchRepoDefinitionNode(fileNodes, pos, ident.Name, obj, definitionContexts[ident]) if node != nil { objToNode[obj] = node.ID result.SymbolsCovered++ @@ -2020,12 +2018,12 @@ func resolveGoUse( // enrichImplements confirms existing EdgeImplements edges using go/types. // implementsInterfaceNode reports whether nodeID can legitimately stand as -// the interface side of an implements edge. Phase-1 maps go/types objects to -// graph nodes by innermost file/line containment, and on a signature line the -// innermost node is a PARAMETER — one interface object mis-mapped that way -// fanned a single empty interface into 130,250 implements edges targeting a -// lone `#param:ctx` node (57% of the workspace's implements set). Node kind -// is the exact guard: only an interface-kind node may host an interface. +// the interface side of an implements edge. Phase 1 now rejects incompatible +// identities before mapping; this kind check remains defense in depth for +// persisted or legacy projections. The historical line-first matcher once +// mapped an interface to a same-line parameter and fanned it into 130,250 +// implements edges targeting one `#param:ctx` node. Only an interface-kind +// node may host an interface. func implementsInterfaceNode(nodesByID map[string]*graph.Node, nodeID string) bool { node := nodesByID[nodeID] return node != nil && node.Kind == graph.KindInterface @@ -2275,57 +2273,6 @@ func findContainingFuncInNodes(nodes []*graph.Node, line int) *graph.Node { return best } -// matchRepoNodeByFileLine mirrors semantic.MatchNodeByFileLine over an -// already-materialized file slice. Keeping the exact innermost-then-nearest -// policy preserves matching behavior while removing the per-object store read. -func matchRepoNodeByFileLine(nodes []*graph.Node, line int) *graph.Node { - var best *graph.Node - bestSize := int(^uint(0) >> 1) - for _, n := range nodes { - if n == nil || n.Kind == graph.KindFile || n.Kind == graph.KindImport { - continue - } - if n.StartLine <= line && line <= n.EndLine { - size := n.EndLine - n.StartLine - if size < bestSize { - best = n - bestSize = size - } - } - } - if best != nil { - return best - } - - bestDist := int(^uint(0) >> 1) - for _, n := range nodes { - if n == nil || n.Kind == graph.KindFile || n.Kind == graph.KindImport { - continue - } - dist := n.StartLine - line - if dist < 0 { - dist = -dist - } - if dist < bestDist { - best = n - bestDist = dist - } - } - if bestDist <= 2 { - return best - } - return nil -} - -func matchRepoNodeByName(nodes []*graph.Node, name string) *graph.Node { - for _, n := range nodes { - if n != nil && n.Name == name { - return n - } - } - return nil -} - // inferEdgeKindFromObj determines the edge kind from a go/types object. func inferEdgeKindFromObj(obj types.Object) graph.EdgeKind { switch obj.(type) { diff --git a/internal/semantic/lsp/graph_batch_test.go b/internal/semantic/lsp/graph_batch_test.go index 356d63b9e..56eecf09b 100644 --- a/internal/semantic/lsp/graph_batch_test.go +++ b/internal/semantic/lsp/graph_batch_test.go @@ -140,3 +140,43 @@ func TestLSPMutationBatchCollapsesWrites(t *testing.T) { counting.reindexBatchCalls, counting.addBatchCalls, counting.persistEdgeBatchCalls) } } + +func TestMatchCallableByFileLineUsesLiveGraphProjection(t *testing.T) { + const filePath = "callable.go" + function := &graph.Node{ID: "callable.go::Outer", Kind: graph.KindFunction, Name: "Outer", FilePath: filePath, StartLine: 10, EndLine: 20} + parameter := &graph.Node{ID: function.ID + "#param:value", Kind: graph.KindParam, Name: "value", FilePath: filePath, StartLine: 10} + closure := &graph.Node{ID: function.ID + "#closure@14", Kind: graph.KindClosure, Name: "closure@14", FilePath: filePath, StartLine: 14, EndLine: 16} + method := &graph.Node{ID: "callable.go::Widget.Run", Kind: graph.KindMethod, Name: "Run", FilePath: filePath, StartLine: 30, EndLine: 35} + variable := &graph.Node{ID: "callable.go::value", Kind: graph.KindVariable, Name: "value", FilePath: filePath, StartLine: 40, EndLine: 40} + view := newLSPGraphView([]*graph.Node{parameter, function, closure, method, variable}, nil) + + for _, test := range []struct { + name string + line int + want string + }{ + {name: "function wins over same-line parameter", line: 10, want: function.ID}, + {name: "function body", line: 12, want: function.ID}, + {name: "inner closure", line: 15, want: closure.ID}, + {name: "method body", line: 32, want: method.ID}, + {name: "two-line tolerance", line: 28, want: method.ID}, + {name: "unrelated variable", line: 40, want: ""}, + } { + t.Run(test.name, func(t *testing.T) { + matched := view.matchCallableByFileLine(filePath, test.line) + if test.want == "" { + if matched != nil { + t.Fatalf("expected no callable, got %s", matched.ID) + } + return + } + if matched == nil || matched.ID != test.want { + got := "" + if matched != nil { + got = matched.ID + } + t.Fatalf("expected %s, got %s", test.want, got) + } + }) + } +} diff --git a/internal/semantic/lsp/passive_test.go b/internal/semantic/lsp/passive_test.go index 4ec16ccc3..9aa7f7891 100644 --- a/internal/semantic/lsp/passive_test.go +++ b/internal/semantic/lsp/passive_test.go @@ -160,15 +160,6 @@ func (s *fakeLSPSocketServer) PushRequest(id any, method string, params any) err return err } -// SetHandler installs a custom request handler. The default handler -// answers initialize with an empty ServerCapabilities and any other -// method with an empty result object. -func (s *fakeLSPSocketServer) SetHandler(h func(method string, params json.RawMessage) (any, bool)) { - s.mu.Lock() - defer s.mu.Unlock() - s.handler = h -} - func (s *fakeLSPSocketServer) defaultHandler(method string, _ json.RawMessage) (any, bool) { switch method { case "initialize": diff --git a/internal/semantic/matcher.go b/internal/semantic/matcher.go index b68f53beb..b476f115b 100644 --- a/internal/semantic/matcher.go +++ b/internal/semantic/matcher.go @@ -1,31 +1,22 @@ package semantic -import ( - "path/filepath" - "strings" - - "github.com/zzet/gortex/internal/graph" -) - -// SymbolMap provides bidirectional mapping between external symbol identifiers -// (SCIP URIs, go/types object IDs, LSP URIs) and Gortex node IDs. +// SymbolMap resolves external symbol identifiers (SCIP URIs, go/types object +// IDs, LSP URIs) to Gortex node IDs. Enrichment providers only ever look up +// in that direction, so no reverse index is kept. type SymbolMap struct { externalToGortex map[string]string - gortexToExternal map[string]string } // NewSymbolMap creates an empty symbol map. func NewSymbolMap() *SymbolMap { return &SymbolMap{ externalToGortex: make(map[string]string), - gortexToExternal: make(map[string]string), } } // Add registers a mapping between an external symbol ID and a Gortex node ID. func (m *SymbolMap) Add(externalID, gortexID string) { m.externalToGortex[externalID] = gortexID - m.gortexToExternal[gortexID] = externalID } // GortexID looks up the Gortex node ID for an external symbol. @@ -33,150 +24,3 @@ func (m *SymbolMap) GortexID(externalID string) (string, bool) { id, ok := m.externalToGortex[externalID] return id, ok } - -// ExternalID looks up the external symbol ID for a Gortex node. -func (m *SymbolMap) ExternalID(gortexID string) (string, bool) { - id, ok := m.gortexToExternal[gortexID] - return id, ok -} - -// Size returns the number of mappings. -func (m *SymbolMap) Size() int { - return len(m.externalToGortex) -} - -// MatchNodeByFileLine finds a Gortex node by file path and line number. -// This is the primary matching strategy for SCIP and LSP results. -// It finds the innermost (smallest range) non-file node containing the line. -func MatchNodeByFileLine(g graph.Store, filePath string, line int) *graph.Node { - nodes := g.GetFileNodes(filePath) - - // First: find the innermost node containing this line (smallest range). - var best *graph.Node - bestSize := int(^uint(0) >> 1) - for _, n := range nodes { - if n.Kind == graph.KindFile || n.Kind == graph.KindImport { - continue - } - if n.StartLine <= line && line <= n.EndLine { - size := n.EndLine - n.StartLine - if size < bestSize { - best = n - bestSize = size - } - } - } - if best != nil { - return best - } - - // Fallback: find the closest node by start line (within tolerance). - bestDist := int(^uint(0) >> 1) - for _, n := range nodes { - if n.Kind == graph.KindFile || n.Kind == graph.KindImport { - continue - } - dist := abs(n.StartLine - line) - if dist < bestDist { - best = n - bestDist = dist - } - } - if bestDist <= 2 { - return best - } - return nil -} - -// MatchCallableByFileLine finds the innermost function / method / -// closure node containing the line. Used to match LSP call-hierarchy -// items (whose selectionRange points at a function's NAME line) back -// to graph nodes. The generic MatchNodeByFileLine is wrong for this: -// parameter nodes sit on the declaration line with a zero-height -// span, so they always win the innermost tie and the hop lands on a -// `#param:` endpoint instead of the function itself. -func MatchCallableByFileLine(g graph.Store, filePath string, line int) *graph.Node { - nodes := g.GetFileNodes(filePath) - - callable := func(k graph.NodeKind) bool { - return k == graph.KindFunction || k == graph.KindMethod || k == graph.KindClosure - } - - // First: the innermost callable containing this line (smallest range). - var best *graph.Node - bestSize := int(^uint(0) >> 1) - for _, n := range nodes { - if !callable(n.Kind) { - continue - } - if n.StartLine <= line && line <= n.EndLine { - size := n.EndLine - n.StartLine - if size < bestSize { - best = n - bestSize = size - } - } - } - if best != nil { - return best - } - - // Fallback: the closest callable by start line (within tolerance). - bestDist := int(^uint(0) >> 1) - for _, n := range nodes { - if !callable(n.Kind) { - continue - } - dist := abs(n.StartLine - line) - if dist < bestDist { - best = n - bestDist = dist - } - } - if bestDist <= 2 { - return best - } - return nil -} - -// MatchNodeByQualName finds a Gortex node by qualified name. -func MatchNodeByQualName(g graph.Store, qualName string) *graph.Node { - return g.GetNodeByQualName(qualName) -} - -// MatchNodeByNameInFile finds a Gortex node by name within a specific file. -func MatchNodeByNameInFile(g graph.Store, name, filePath string) *graph.Node { - nodes := g.GetFileNodes(filePath) - for _, n := range nodes { - if n.Name == name { - return n - } - } - return nil -} - -// NormalizeFilePath converts an absolute path to a repo-relative path. -func NormalizeFilePath(absPath, repoRoot string) string { - rel, err := filepath.Rel(repoRoot, absPath) - if err != nil { - return absPath - } - return filepath.ToSlash(rel) -} - -// ParseGortexID extracts the file path and symbol name from a Gortex node ID. -// Gortex IDs have the format "path/to/file.go::SymbolName". -func ParseGortexID(id string) (filePath, symbolName string) { - idx := strings.LastIndex(id, "::") - if idx < 0 { - return id, "" - } - return id[:idx], id[idx+2:] -} - -func abs(x int) int { - if x < 0 { - return -x - } - return x -} diff --git a/internal/semantic/matcher_test.go b/internal/semantic/matcher_test.go index 276205266..d03a0c4af 100644 --- a/internal/semantic/matcher_test.go +++ b/internal/semantic/matcher_test.go @@ -4,8 +4,6 @@ import ( "testing" "github.com/stretchr/testify/assert" - - "github.com/zzet/gortex/internal/graph" ) func TestSymbolMap(t *testing.T) { @@ -14,163 +12,14 @@ func TestSymbolMap(t *testing.T) { m.Add("scip::Foo", "main.go::Foo") m.Add("scip::Bar", "lib.go::Bar") - assert.Equal(t, 2, m.Size()) - id, ok := m.GortexID("scip::Foo") assert.True(t, ok) assert.Equal(t, "main.go::Foo", id) - ext, ok := m.ExternalID("lib.go::Bar") + id, ok = m.GortexID("scip::Bar") assert.True(t, ok) - assert.Equal(t, "scip::Bar", ext) + assert.Equal(t, "lib.go::Bar", id) _, ok = m.GortexID("scip::Unknown") assert.False(t, ok) } - -func TestMatchNodeByFileLine(t *testing.T) { - g := graph.New() - g.AddNode(&graph.Node{ - ID: "main.go::main", Kind: graph.KindFunction, Name: "main", - FilePath: "main.go", StartLine: 10, EndLine: 20, - }) - g.AddNode(&graph.Node{ - ID: "main.go::helper", Kind: graph.KindFunction, Name: "helper", - FilePath: "main.go", StartLine: 22, EndLine: 30, - }) - g.AddNode(&graph.Node{ - ID: "main.go", Kind: graph.KindFile, Name: "main.go", - FilePath: "main.go", StartLine: 1, EndLine: 30, - }) - - // Exact start line match. - n := MatchNodeByFileLine(g, "main.go", 10) - assert.NotNil(t, n) - assert.Equal(t, "main.go::main", n.ID) - - // Within range. - n = MatchNodeByFileLine(g, "main.go", 15) - assert.NotNil(t, n) - assert.Equal(t, "main.go::main", n.ID) - - // Second function. - n = MatchNodeByFileLine(g, "main.go", 25) - assert.NotNil(t, n) - assert.Equal(t, "main.go::helper", n.ID) - - // Line in gap between functions — may or may not find a match. - n = MatchNodeByFileLine(g, "main.go", 21) - // Within tolerance of 2 lines from helper (line 22), should find it. - if n != nil { - assert.Equal(t, "main.go::helper", n.ID) - } -} - -func TestMatchNodeByQualName(t *testing.T) { - g := graph.New() - g.AddNode(&graph.Node{ - ID: "main.go::Foo", Kind: graph.KindFunction, Name: "Foo", - QualName: "github.com/test/pkg.Foo", FilePath: "main.go", - }) - - n := MatchNodeByQualName(g, "github.com/test/pkg.Foo") - assert.NotNil(t, n) - assert.Equal(t, "main.go::Foo", n.ID) - - n = MatchNodeByQualName(g, "unknown") - assert.Nil(t, n) -} - -func TestMatchNodeByNameInFile(t *testing.T) { - g := graph.New() - g.AddNode(&graph.Node{ - ID: "main.go::Foo", Kind: graph.KindFunction, Name: "Foo", - FilePath: "main.go", - }) - - n := MatchNodeByNameInFile(g, "Foo", "main.go") - assert.NotNil(t, n) - assert.Equal(t, "main.go::Foo", n.ID) - - n = MatchNodeByNameInFile(g, "Foo", "other.go") - assert.Nil(t, n) -} - -func TestParseGortexID(t *testing.T) { - tests := []struct { - id string - wantFile string - wantSym string - }{ - {"main.go::Foo", "main.go", "Foo"}, - {"pkg/auth/token.go::ValidateToken", "pkg/auth/token.go", "ValidateToken"}, - {"main.go", "main.go", ""}, - } - - for _, tt := range tests { - file, sym := ParseGortexID(tt.id) - assert.Equal(t, tt.wantFile, file, "file for %s", tt.id) - assert.Equal(t, tt.wantSym, sym, "sym for %s", tt.id) - } -} - -func TestNormalizeFilePath(t *testing.T) { - result := NormalizeFilePath("/home/user/repo/pkg/foo.go", "/home/user/repo") - assert.Equal(t, "pkg/foo.go", result) -} - -func TestMatchCallableByFileLine(t *testing.T) { - g := graph.New() - // Function with a param decoy sharing its declaration line — the - // zero-height param span wins MatchNodeByFileLine's innermost tie, - // which is exactly what the callable matcher must NOT return. - g.AddNode(&graph.Node{ - ID: "a.go::TestFoo", Kind: graph.KindFunction, Name: "TestFoo", - FilePath: "a.go", StartLine: 10, EndLine: 20, - }) - g.AddNode(&graph.Node{ - ID: "a.go::TestFoo#param:t", Kind: graph.KindParam, Name: "t", - FilePath: "a.go", StartLine: 10, EndLine: 10, - }) - g.AddNode(&graph.Node{ - ID: "a.go::TestFoo#closure:1", Kind: graph.KindClosure, Name: "closure", - FilePath: "a.go", StartLine: 14, EndLine: 16, - }) - g.AddNode(&graph.Node{ - ID: "a.go::Recv.M", Kind: graph.KindMethod, Name: "M", - FilePath: "a.go", StartLine: 25, EndLine: 27, - }) - g.AddNode(&graph.Node{ - ID: "a.go::topVar", Kind: graph.KindVariable, Name: "topVar", - FilePath: "a.go", StartLine: 40, EndLine: 40, - }) - g.AddNode(&graph.Node{ - ID: "a.go", Kind: graph.KindFile, Name: "a.go", - FilePath: "a.go", StartLine: 1, EndLine: 50, - }) - - cases := []struct { - name string - line int - want string // "" = nil expected - }{ - {"declaration line beats param decoy", 10, "a.go::TestFoo"}, - {"body line inside function", 12, "a.go::TestFoo"}, - {"innermost closure wins inside its span", 15, "a.go::TestFoo#closure:1"}, - {"method declaration line", 25, "a.go::Recv.M"}, - {"near-miss within tolerance snaps to method", 23, "a.go::Recv.M"}, - {"variable line far from any callable", 40, ""}, - } - for _, tc := range cases { - t.Run(tc.name, func(t *testing.T) { - n := MatchCallableByFileLine(g, "a.go", tc.line) - if tc.want == "" { - assert.Nil(t, n) - return - } - if assert.NotNil(t, n) { - assert.Equal(t, tc.want, n.ID) - } - }) - } -} diff --git a/internal/server/handler.go b/internal/server/handler.go index 0473f12a4..311375cc4 100644 --- a/internal/server/handler.go +++ b/internal/server/handler.go @@ -274,31 +274,33 @@ const ( // any query; a remote that does not advertise read_only is treated as // read-only (fail-safe) by the consumer. type HealthResponse struct { - Status string `json:"status"` - Indexed bool `json:"indexed"` - Nodes int `json:"nodes"` - Edges int `json:"edges"` - Version string `json:"version"` - UptimeSeconds float64 `json:"uptime_seconds"` - SchemaVersion int `json:"schema_version"` - APIVersion int `json:"api_version"` - ReadOnly bool `json:"read_only"` - Capabilities []string `json:"capabilities,omitempty"` + Status string `json:"status"` + Indexed bool `json:"indexed"` + Nodes int `json:"nodes"` + Edges int `json:"edges"` + Version string `json:"version"` + UptimeSeconds float64 `json:"uptime_seconds"` + SchemaVersion int `json:"schema_version"` + APIVersion int `json:"api_version"` + ReadOnly bool `json:"read_only"` + Capabilities []string `json:"capabilities,omitempty"` + GraphIntegrity *daemon.GraphIntegrityStatus `json:"graph_integrity,omitempty"` } func (h *Handler) handleHealth(w http.ResponseWriter, _ *http.Request) { stats := h.graph.Stats() resp := HealthResponse{ - Status: "ok", - Indexed: stats.TotalNodes > 0, - Nodes: stats.TotalNodes, - Edges: stats.TotalEdges, - Version: h.version, - UptimeSeconds: time.Since(h.startTime).Seconds(), - SchemaVersion: SchemaVersion, - APIVersion: APIVersion, - ReadOnly: h.readOnly, - Capabilities: h.advertisedCapabilities(), + Status: "ok", + Indexed: stats.TotalNodes > 0, + Nodes: stats.TotalNodes, + Edges: stats.TotalEdges, + Version: h.version, + UptimeSeconds: time.Since(h.startTime).Seconds(), + SchemaVersion: SchemaVersion, + APIVersion: APIVersion, + ReadOnly: h.readOnly, + Capabilities: h.advertisedCapabilities(), + GraphIntegrity: daemon.GraphIntegrityStatusFor(h.graph), } WriteJSON(w, http.StatusOK, resp) } diff --git a/internal/server/health_integrity_test.go b/internal/server/health_integrity_test.go new file mode 100644 index 000000000..02947baf0 --- /dev/null +++ b/internal/server/health_integrity_test.go @@ -0,0 +1,59 @@ +package server + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + "testing" + + "github.com/zzet/gortex/internal/graph" + "go.uber.org/zap" +) + +func TestHealthEndpointGraphIntegrityOptionalAndWarning(t *testing.T) { + store := graph.New() + handler := NewHandler(nil, store, "test-version", zap.NewNop()) + + requestHealth := func() (HealthResponse, string) { + t.Helper() + req := httptest.NewRequest(http.MethodGet, "/v1/health", nil) + recorder := httptest.NewRecorder() + handler.ServeHTTP(recorder, req) + if recorder.Code != http.StatusOK { + t.Fatalf("unexpected health status %d: %s", recorder.Code, recorder.Body.String()) + } + var response HealthResponse + if err := json.Unmarshal(recorder.Body.Bytes(), &response); err != nil { + t.Fatalf("decode health response: %v", err) + } + return response, recorder.Body.String() + } + + zero, body := requestHealth() + if zero.Status != "ok" || zero.GraphIntegrity != nil { + t.Fatalf("zero telemetry must leave health OK and omit graph_integrity: %+v", zero) + } + if strings.Contains(body, "graph_integrity") { + t.Fatalf("zero telemetry was not omitted: %s", body) + } + + store.AddNode(&graph.Node{ID: "source", Kind: graph.KindFunction, RepoPrefix: "repo-a"}) + store.AddEdge(&graph.Edge{ + From: "source", + To: "target#param:x", + Kind: graph.EdgeImplements, + Origin: "lsp", + }) + + degraded, body := requestHealth() + integrity := degraded.GraphIntegrity + if degraded.Status != "ok" || integrity == nil || integrity.Status != "warn" || !integrity.Degraded || integrity.WriteRejected != 1 { + t.Fatalf("integrity warning must not change endpoint health/readiness: %+v", degraded) + } + for _, forbidden := range []string{"samples", "attribution", "file_path", `"from"`, `"to"`} { + if strings.Contains(body, forbidden) { + t.Fatalf("routine health exposed audit detail %q: %s", forbidden, body) + } + } +} diff --git a/internal/serverstack/backend.go b/internal/serverstack/backend.go index 80ccee0a5..b48badc6b 100644 --- a/internal/serverstack/backend.go +++ b/internal/serverstack/backend.go @@ -36,7 +36,7 @@ import ( // schema version is incompatible. The caller must hold the store lock; // NewSharedServer passes true only in the branch where it acquired the // exclusive flock. -func OpenBackend(name, path string, bufferPoolMB uint64, logger *zap.Logger, allowRebuild bool) (graph.Store, func(), error) { +func OpenBackend(name, path string, logger *zap.Logger, allowRebuild bool) (graph.Store, func(), error) { if err := checkBackend(name); err != nil { return nil, nil, err } @@ -47,7 +47,7 @@ func OpenBackend(name, path string, bufferPoolMB uint64, logger *zap.Logger, all if logger != nil { logger.Info("opening sqlite backend", zap.String("path", resolved)) } - return openSqliteBackend(resolved, bufferPoolMB, allowRebuild) + return openSqliteBackend(resolved, allowRebuild) } // checkBackend validates a requested backend name. It is split out of @@ -96,10 +96,7 @@ func resolveBackendPath(in, filename string) (string, error) { // openSqliteBackend opens (or creates) the SQLite store at path. The // pure-Go modernc.org/sqlite driver keeps the binary CGo-free while still // getting a real query planner that drives the graph's secondary indexes. -// bufferPoolMB is accepted for signature parity with other on-disk -// backends but unused — SQLite sizes its page cache via a pragma. -func openSqliteBackend(path string, bufferPoolMB uint64, allowRebuild bool) (graph.Store, func(), error) { - _ = bufferPoolMB +func openSqliteBackend(path string, allowRebuild bool) (graph.Store, func(), error) { var opts []store_sqlite.Option if allowRebuild { opts = append(opts, store_sqlite.WithRebuild()) diff --git a/internal/serverstack/backend_test.go b/internal/serverstack/backend_test.go index 763150ece..ece65c616 100644 --- a/internal/serverstack/backend_test.go +++ b/internal/serverstack/backend_test.go @@ -14,7 +14,7 @@ import ( // database. The error has to name the replacement. func TestOpenBackend_MemoryIsRetired(t *testing.T) { for _, name := range []string{"memory", "mem", "in-memory", "In-Memory", " memory "} { - store, _, err := OpenBackend(name, "", 0, zap.NewNop(), false) + store, _, err := OpenBackend(name, "", zap.NewNop(), false) if err == nil { t.Fatalf("OpenBackend(%q): want an error, got a store", name) } @@ -35,7 +35,7 @@ func TestOpenBackend_MemoryIsRetired(t *testing.T) { // caller did not name a backend explicitly. func TestOpenBackend_EmptyNameOpensSqlite(t *testing.T) { path := filepath.Join(t.TempDir(), "store.sqlite") - store, cleanup, err := OpenBackend("", path, 0, zap.NewNop(), true) + store, cleanup, err := OpenBackend("", path, zap.NewNop(), true) if err != nil { t.Fatalf(`OpenBackend(""): %v`, err) } @@ -49,7 +49,7 @@ func TestOpenBackend_EmptyNameOpensSqlite(t *testing.T) { // creates) a store at the resolved path. func TestOpenBackend_SqliteOpensFile(t *testing.T) { path := filepath.Join(t.TempDir(), "store.sqlite") - store, cleanup, err := OpenBackend("sqlite", path, 0, zap.NewNop(), true) + store, cleanup, err := OpenBackend("sqlite", path, zap.NewNop(), true) if err != nil { t.Fatalf("OpenBackend(sqlite): %v", err) } @@ -62,7 +62,7 @@ func TestOpenBackend_SqliteOpensFile(t *testing.T) { // TestOpenBackend_Unknown asserts a stale backend name (e.g. the removed // ladybug) errors rather than silently falling back. func TestOpenBackend_Unknown(t *testing.T) { - if _, _, err := OpenBackend("ladybug", "", 0, zap.NewNop(), false); err == nil { + if _, _, err := OpenBackend("ladybug", "", zap.NewNop(), false); err == nil { t.Fatal("an unknown backend must error") } } diff --git a/internal/serverstack/shared_server.go b/internal/serverstack/shared_server.go index 3a3c89e65..d6c776adf 100644 --- a/internal/serverstack/shared_server.go +++ b/internal/serverstack/shared_server.go @@ -79,7 +79,6 @@ type SharedServerConfig struct { Logger *zap.Logger Version string Embedder EmbedderRequest - BufferPoolMB uint64 SideStores SideStores ScopeWorkspace string ScopeProject string @@ -143,10 +142,6 @@ type SharedServer struct { // default when it was empty. Entry points publish it so out-of-band // readers can find the same store instead of re-deriving the default. StorePath string - // EmbedderDims is the active embedder's vector dimensionality, or 0 - // when embeddings are off — the width every persisted vector in this - // workspace was built at. - EmbedderDims int // ResolverLSPRegistry / LSPRouter are the resolve-time LSP wiring the // entry point's warmup hooks reference; nil when semantic enrichment @@ -289,7 +284,7 @@ func NewSharedServer(cfg SharedServerConfig) (*SharedServer, error) { // allowRebuild is gated on actually holding the store lock: only then may // the sqlite backend drop and recreate an incompatible-schema DB. - g, backendCleanup, err := OpenBackend(cfg.Backend, storePath, cfg.BufferPoolMB, logger, storeLockHeld) + g, backendCleanup, err := OpenBackend(cfg.Backend, storePath, logger, storeLockHeld) if err != nil { return nil, err } @@ -541,7 +536,6 @@ func NewSharedServer(cfg SharedServerConfig) (*SharedServer, error) { idx.SetEmbeddingChunkOptions(EmbeddingChunkOptions(conf)) idx.SetEmbeddingMaxSymbols(conf.Embedding.MaxSymbols) idx.SetEmbeddingAPIConcurrency(conf.Embedding.APIConcurrency) - s.EmbedderDims = embedder.Dimensions() } cm, err := config.NewConfigManager("") @@ -582,6 +576,18 @@ func NewSharedServer(cfg SharedServerConfig) (*SharedServer, error) { } } s.MultiIndexer = mi + // Appended after backendCleanup but before MCP background drain. LIFO + // teardown therefore drains background work first, then closes per-repo + // parser workers, then the standalone Indexer, and only then the graph and + // store lock. + s.cleanup = append(s.cleanup, func() { + if mi != nil { + if err := mi.Close(context.Background()); err != nil { + logger.Warn("serverstack: multi-indexer shutdown failed", zap.Error(err)) + } + } + idx.Close() + }) toolPolicyCfg := gortexmcp.ToolPolicyConfig{ Preset: conf.MCP.Tools.Preset, diff --git a/internal/serverstack/shared_server_lifecycle_test.go b/internal/serverstack/shared_server_lifecycle_test.go new file mode 100644 index 000000000..7133b6dda --- /dev/null +++ b/internal/serverstack/shared_server_lifecycle_test.go @@ -0,0 +1,62 @@ +package serverstack + +import ( + "errors" + "os" + "path/filepath" + "strings" + "testing" + + "go.uber.org/zap" + + "github.com/zzet/gortex/internal/config" + "github.com/zzet/gortex/internal/indexer" +) + +func TestLifecycleSharedServerCloseRejectsExtractionAndMutations(t *testing.T) { + repo := t.TempDir() + if err := os.WriteFile(filepath.Join(repo, "toy.go"), []byte("package toy\n\nfunc Toy() {}\n"), 0o644); err != nil { + t.Fatal(err) + } + + ss, err := NewSharedServer(SharedServerConfig{ + Lifecycle: LifecycleOneshot, + Index: repo, + BackendPath: filepath.Join(t.TempDir(), "embedded.sqlite"), + Config: config.Default(), + Logger: zap.NewNop(), + Version: "test", + SideStores: SideStores{NotesDir: t.TempDir(), NotesRepo: "test"}, + SavingsPath: filepath.Join(t.TempDir(), "sidecar.sqlite"), + SavingsLegacyJSON: filepath.Join(t.TempDir(), "savings.json"), + }) + if err != nil { + t.Fatalf("NewSharedServer: %v", err) + } + closed := false + t.Cleanup(func() { + if !closed { + _ = ss.Close() + } + }) + if ss.MultiIndexer == nil { + t.Fatal("shared server did not construct a MultiIndexer") + } + + err = ss.Close() + closed = true + if err != nil { + t.Fatalf("Close: %v", err) + } + if _, err := ss.Indexer.ExtractBuffer("go", "after.go", []byte("package after\n")); !errors.Is(err, indexer.ErrIndexerClosed) { + t.Fatalf("post-close standalone extraction error = %v, want ErrIndexerClosed", err) + } + + afterRepo := t.TempDir() + if err := os.WriteFile(filepath.Join(afterRepo, "after.go"), []byte("package after\n"), 0o644); err != nil { + t.Fatal(err) + } + if _, err := ss.MultiIndexer.TrackRepo(config.RepoEntry{Path: afterRepo, Name: "after"}); err == nil || !strings.Contains(strings.ToLower(err.Error()), "closed") { + t.Fatalf("post-close mutation error = %v, want closed lifecycle", err) + } +} diff --git a/internal/toolref/toolref.go b/internal/toolref/toolref.go index 334acebfa..8711702d2 100644 --- a/internal/toolref/toolref.go +++ b/internal/toolref/toolref.go @@ -33,13 +33,6 @@ var cliExample = map[string]string{ "reindex_repository": `gortex call workspace_admin --arg operation=reindex --arg arguments='{"path":""}'`, } -// MCPRef renders an MCP-directed reference to a tool: "call the `read_file` MCP -// tool". Use wherever guidance assumes the agent has the Gortex MCP server -// mounted and can call the tool directly. -func MCPRef(tool string) string { - return "call the `" + tool + "` MCP tool" -} - // CLIFallback renders the compact shell invocation for one operation, for // example `gortex call read --arg target='{"file":"..."}'`. This is the single // place an internal tool reference becomes agent-facing Bash — nothing else