Documentation
¶
Overview ¶
Package vectorstore provides code indexing and vector storage operations.
The vectorstore package implements semantic code search using vector embeddings, including parallel indexing, incremental updates, chunk-based document storage, and integration with Qdrant vector database.
Index ¶
- Constants
- func GetAllBranches(repoPath string) ([]string, error)
- func GetBranchCommit(repoPath, branch string) (string, error)
- func GetChangedFilesSince(repoPath, fromCommit string) ([]string, error)
- func GetCurrentBranch(repoPath string) (string, error)
- func GetHeadCommit(repoPath string) (string, error)
- func GetKnownBranches(repoName string) ([]string, error)
- func GetMetadataPath(repoName, branch string) string
- func GetWorkerCount() int
- func IsGitRepo(repoPath string) bool
- func NeedsReindexing(repoPath, repoName, branch string) (bool, string, error)
- func SanitizeBranchName(branch string) string
- func SaveMetadata(meta *BranchMetadata) error
- type BranchMetadata
- type CodeChunk
- type EmbeddingProvider
- type FileCandidate
- type FileSelection
- type IndexJob
- type IndexStats
- type Indexer
- type IndexingConfig
- type OllamaEmbeddingProvider
- type OpenAIEmbeddingProvider
- type QdrantStore
- func (qs *QdrantStore) Close() error
- func (qs *QdrantStore) DeleteCollection(ctx context.Context) error
- func (qs *QdrantStore) DeleteFile(ctx context.Context, filePath string) error
- func (qs *QdrantStore) FetchAllChunks(ctx context.Context, basePath string) ([]SearchResult, error)
- func (qs *QdrantStore) GetStats(ctx context.Context) (*Stats, error)
- func (qs *QdrantStore) IndexFile(ctx context.Context, filePath, content string) error
- func (qs *QdrantStore) Search(ctx context.Context, query string, limit int) ([]SearchResult, error)
- func (qs *QdrantStore) SearchWithAggregation(ctx context.Context, query string, maxFiles int) ([]SearchResult, error)
- type SearchConfig
- type SearchResult
- type Stats
- type VectorStore
Constants ¶
const ( MaxTokensPerChunk = 3500 // Safe chunk size to avoid context limit OverlapTokens = 250 // Overlap between chunks for context continuity MaxTokensWholeFile = 3200 // Embed whole file if under this limit CharsPerTokenEstimate = 4.0 // Conservative estimate: 1 token ≈ 4 chars )
Token budget configuration for bge-m3 model (8192 token context)
const ( // Default Ollama model for embeddings // Can be overridden via config - see configs/repos.yaml for options DefaultOllamaModel = "bge-m3" // Model dimensions (used for Qdrant collection creation) OllamaNomicDimension = 768 // nomic-embed-text: 2K context, fastest OllamaBGEM3Dimension = 1024 // bge-m3: 8K context, best quality (recommended) OllamaMxbaiDimension = 1024 // mxbai-embed-large: 512 token context, middle speed )
const ( // OpenAI embedding models OpenAIModelTextEmbedding3Small = "text-embedding-3-small" OpenAIModelTextEmbedding3Large = "text-embedding-3-large" // Dimensions OpenAIDimensionSmall = 1536 OpenAIDimensionLarge = 3072 )
Variables ¶
This section is empty.
Functions ¶
func GetAllBranches ¶
GetAllBranches returns a list of all local branches in a repository
func GetBranchCommit ¶
GetBranchCommit returns the HEAD commit SHA for a specific branch
func GetChangedFilesSince ¶
GetChangedFilesSince returns a list of files that changed between fromCommit and HEAD This is used for incremental indexing after git pull
func GetCurrentBranch ¶
GetCurrentBranch returns the current git branch name for a repository
func GetHeadCommit ¶
GetHeadCommit returns the current HEAD commit SHA for a repository
func GetKnownBranches ¶
GetKnownBranches returns a list of branches that have been indexed by reading the .mesh/{repo}/ directory structure
func GetMetadataPath ¶
GetMetadataPath returns path to metadata file for repo+branch Example: .mesh/my-repo/main/metadata.json
func GetWorkerCount ¶
func GetWorkerCount() int
GetWorkerCount returns optimal worker count for current system
func NeedsReindexing ¶
NeedsReindexing checks if a repository needs to be re-indexed Returns true if: - No metadata exists (never indexed) - Current commit differs from last indexed commit
func SanitizeBranchName ¶
SanitizeBranchName converts a branch name to a filesystem-safe string Example: "feature/auth-v2" -> "feature-auth-v2"
func SaveMetadata ¶
func SaveMetadata(meta *BranchMetadata) error
SaveMetadata saves the metadata for a repo+branch combination
Types ¶
type BranchMetadata ¶
type BranchMetadata struct {
RepoName string `json:"repo_name"`
Branch string `json:"branch"`
CommitSHA string `json:"commit_sha"`
IndexedAt time.Time `json:"indexed_at"`
FileCount int `json:"file_count"`
}
BranchMetadata tracks indexing state per branch This allows incremental re-indexing when commits change
func LoadMetadata ¶
func LoadMetadata(repoName, branch string) (*BranchMetadata, error)
LoadMetadata loads the metadata for a repo+branch combination Returns nil if no metadata exists yet (first time indexing)
type CodeChunk ¶
type CodeChunk struct {
Content string
ChunkIndex int
StartLine int
EndLine int
Header string // Context header for better retrieval
ChunkID string // Stable identifier: hash(path + startLine + endLine)
}
CodeChunk represents a chunk of code with metadata
type EmbeddingProvider ¶
type EmbeddingProvider interface {
// CreateEmbedding creates an embedding vector from text
CreateEmbedding(ctx context.Context, text string) ([]float32, error)
// GetDimensions returns the dimensionality of the embedding vectors
GetDimensions() int
// GetModelName returns the name of the embedding model
GetModelName() string
}
EmbeddingProvider is the interface for embedding providers Allows swapping between OpenAI, Ollama, local models, etc.
type FileCandidate ¶
type FileCandidate struct {
BasePath string
Language string
// Semantic scores (from vector search chunks)
BestChunkScore float32 // Highest chunk score
AvgChunkScore float32 // Average of all chunk scores
TopKChunkScores []float32 // Top-K chunk scores for depth analysis
ChunkCount int // Number of chunks for this file
// Hybrid scores
KeywordScore float32 // Keyword matching score (length-normalized)
PathScore float32 // File path relevance score
HybridScore float32 // Final weighted hybrid score
// Token tracking
EstimatedTokens int // Estimated token count for budget management
}
FileCandidate represents a file with rich scoring information for hybrid ranking.
type FileSelection ¶
type FileSelection struct {
BasePath string
Language string
Score float32
IsPartial bool // true if only top chunks included
ChunkIndices []int // which chunks to include (empty = all chunks)
EstimatedTokens int
}
FileSelection represents either a complete file or top chunks from an oversized file.
type IndexStats ¶
type IndexStats struct {
Indexed int
Skipped int
Errors int
// contains filtered or unexported fields
}
IndexStats tracks indexing statistics (thread-safe)
type Indexer ¶
type Indexer struct {
// contains filtered or unexported fields
}
Indexer handles indexing of repository files into the vector store
func NewIndexer ¶
func NewIndexer(store VectorStore, repoPath string, logger zerolog.Logger) *Indexer
NewIndexer creates a new indexer
func NewIndexerWithBranch ¶
func NewIndexerWithBranch(store VectorStore, repoPath, repoName, branch string, logger zerolog.Logger) *Indexer
NewIndexerWithBranch creates a new indexer with branch awareness
func (*Indexer) IndexIncremental ¶
IndexIncremental performs incremental indexing based on git changes Only re-indexes files that have changed since the last indexed commit
type IndexingConfig ¶
type IndexingConfig struct {
MaxWorkers int // Concurrent file indexing workers
ChunkingEnabled bool // Enable chunking for large files
MaxChunkSize int // Maximum chunk size in bytes
OverlapSize int // Overlap between chunks in bytes
}
IndexingConfig holds configuration for indexing performance
func DefaultIndexingConfig ¶
func DefaultIndexingConfig() *IndexingConfig
DefaultIndexingConfig returns sensible defaults based on available resources
type OllamaEmbeddingProvider ¶
type OllamaEmbeddingProvider struct {
// contains filtered or unexported fields
}
OllamaEmbeddingProvider implements EmbeddingProvider using Ollama Runs embeddings locally - no data sent to external APIs
func NewOllamaEmbeddingProvider ¶
func NewOllamaEmbeddingProvider(ollamaURL, model string, logger zerolog.Logger) (*OllamaEmbeddingProvider, error)
NewOllamaEmbeddingProvider creates a new Ollama embedding provider
func (*OllamaEmbeddingProvider) CreateEmbedding ¶
func (o *OllamaEmbeddingProvider) CreateEmbedding(ctx context.Context, text string) ([]float32, error)
CreateEmbedding creates an embedding using Ollama
func (*OllamaEmbeddingProvider) GetDimensions ¶
func (o *OllamaEmbeddingProvider) GetDimensions() int
GetDimensions returns the embedding dimension IMPORTANT: If you add a new model to configs/repos.yaml, you may need to add it here if dimensions differ from 1024
func (*OllamaEmbeddingProvider) GetModelName ¶
func (o *OllamaEmbeddingProvider) GetModelName() string
GetModelName returns the model name
type OpenAIEmbeddingProvider ¶
type OpenAIEmbeddingProvider struct {
// contains filtered or unexported fields
}
OpenAIEmbeddingProvider implements EmbeddingProvider using OpenAI API
func NewOpenAIEmbeddingProvider ¶
func NewOpenAIEmbeddingProvider(apiKey, model string, logger zerolog.Logger) (*OpenAIEmbeddingProvider, error)
NewOpenAIEmbeddingProvider creates a new OpenAI embedding provider
func (*OpenAIEmbeddingProvider) CreateEmbedding ¶
func (o *OpenAIEmbeddingProvider) CreateEmbedding(ctx context.Context, text string) ([]float32, error)
CreateEmbedding creates an embedding using OpenAI API
func (*OpenAIEmbeddingProvider) GetDimensions ¶
func (o *OpenAIEmbeddingProvider) GetDimensions() int
GetDimensions returns the embedding dimension
func (*OpenAIEmbeddingProvider) GetModelName ¶
func (o *OpenAIEmbeddingProvider) GetModelName() string
GetModelName returns the model name
type QdrantStore ¶
type QdrantStore struct {
// contains filtered or unexported fields
}
QdrantStore implements VectorStore using Qdrant vector database
func NewQdrantStore ¶
func NewQdrantStore(qdrantURL string, embeddingProvider EmbeddingProvider, repoName string, logger zerolog.Logger) (*QdrantStore, error)
NewQdrantStore creates a new Qdrant vector store with an embedding provider For backward compatibility, uses "main" as default branch if branch is empty
func NewQdrantStoreWithBranch ¶
func NewQdrantStoreWithBranch(qdrantURL string, embeddingProvider EmbeddingProvider, repoName, branch string, logger zerolog.Logger) (*QdrantStore, error)
NewQdrantStoreWithBranch creates a new Qdrant vector store with branch support Collection name format: mesh-{repo}-{branch}-v1
func (*QdrantStore) DeleteCollection ¶
func (qs *QdrantStore) DeleteCollection(ctx context.Context) error
DeleteCollection removes the entire collection
func (*QdrantStore) DeleteFile ¶
func (qs *QdrantStore) DeleteFile(ctx context.Context, filePath string) error
DeleteFile removes a file and all its chunks from the index
func (*QdrantStore) FetchAllChunks ¶
func (qs *QdrantStore) FetchAllChunks(ctx context.Context, basePath string) ([]SearchResult, error)
FetchAllChunks retrieves all chunks for a file using base_path filter
func (*QdrantStore) GetStats ¶
func (qs *QdrantStore) GetStats(ctx context.Context) (*Stats, error)
GetStats returns statistics about the vector store
func (*QdrantStore) IndexFile ¶
func (qs *QdrantStore) IndexFile(ctx context.Context, filePath, content string) error
IndexFile indexes a file by creating an embedding and storing it in Qdrant
func (*QdrantStore) Search ¶
func (qs *QdrantStore) Search(ctx context.Context, query string, limit int) ([]SearchResult, error)
Search performs semantic search for relevant code
func (*QdrantStore) SearchWithAggregation ¶
func (qs *QdrantStore) SearchWithAggregation(ctx context.Context, query string, maxFiles int) ([]SearchResult, error)
SearchWithAggregation searches and reconstructs complete files from chunks Uses adaptive scoring, token budget management, and hybrid ranking for improved recall
type SearchConfig ¶
type SearchConfig struct {
// Token budget management
MaxTokenBudget int // Maximum tokens for context (default: 80000)
ReserveTokens int // Buffer for system prompt (default: 5000)
OversizeChunkLimit int // Top-K chunks for oversized files (default: 5)
// Adaptive scoring (distribution-based)
MinAbsoluteScore float32 // Hard floor for scores (default: 0.15)
ScoreDistributionPercentile float32 // Percentile for threshold (default: 0.90)
MinFilesAfterThreshold int // Ensure min survivors (default: 5)
// Search limits
InitialChunkLimit int // Initial search limit (default: 50)
MaxFilesLimit int // Maximum files to return (default: 15)
// Hybrid scoring weights (dynamically adjusted)
// These weights should sum to 1.0 for proper normalization
SemanticWeight float32 // Vector similarity weight (default: 0.70)
KeywordWeight float32 // Keyword matching weight (default: 0.15)
PathWeight float32 // Path relevance weight (default: 0.05)
AggregateWeight float32 // Multi-chunk depth weight (default: 0.10)
}
SearchConfig holds configuration for intelligent file selection with adaptive scoring, token budget management, and hybrid ranking.
func DefaultSearchConfig ¶
func DefaultSearchConfig() *SearchConfig
DefaultSearchConfig returns a balanced configuration optimized for response speed while maintaining search quality.
func (*SearchConfig) EffectiveTokenBudget ¶
func (c *SearchConfig) EffectiveTokenBudget() int
EffectiveTokenBudget returns the available token budget after reserve.
func (*SearchConfig) Validate ¶
func (c *SearchConfig) Validate() error
Validate checks if the configuration is valid and returns an error if not.
type SearchResult ¶
type SearchResult struct {
// FilePath is the relative path to the file in the repository
FilePath string
// Content is the file content
Content string
// Score is the similarity score (0.0 to 1.0, higher is better)
Score float32
// Language is the programming language
Language string
// FileHash is the SHA256 hash of the file (for change detection)
FileHash string
}
SearchResult represents a search result from the vector store
type Stats ¶
type Stats struct {
// TotalVectors is the total number of vectors in the collection
TotalVectors int64
// CollectionName is the name of the collection
CollectionName string
// IndexedFiles is the number of unique files indexed
IndexedFiles int
// LastIndexed is the timestamp of the last indexing operation
LastIndexed string
}
Stats holds statistics about the vector store
type VectorStore ¶
type VectorStore interface {
// IndexFile indexes a single file with its content
IndexFile(ctx context.Context, filePath, content string) error
// Search performs semantic search for relevant code
Search(ctx context.Context, query string, limit int) ([]SearchResult, error)
// SearchWithAggregation performs search with chunk aggregation to reconstruct complete files.
// Implementations without aggregation support should fall back to Search().
// This method is used by context builders to provide complete file context rather than fragments.
SearchWithAggregation(ctx context.Context, query string, limit int) ([]SearchResult, error)
// DeleteFile removes a file from the index
DeleteFile(ctx context.Context, filePath string) error
// DeleteCollection removes an entire collection (repository)
DeleteCollection(ctx context.Context) error
// GetStats returns statistics about the vector store
GetStats(ctx context.Context) (*Stats, error)
// Close closes the connection to the vector store
Close() error
}
VectorStore is the interface for vector database operations Implementations: QdrantStore, InMemoryStore (for testing), etc.