Documentation
¶
Overview ¶
Package indexer is the phase-2 consumer of the chunker contract in internal/index (docs/13-index.md): a background service that keeps a local any-store database with a BM25 full-text index and an IVF-SQ vector index per space, fed from the SDK's per-space change feed (Space.Changes()), and a hybrid (RRF) search over both.
Index ¶
- Constants
- Variables
- type DocUpsert
- type Embedder
- type FTSQuery
- type Group
- type Hardware
- type Hit
- type HostFilter
- type Indexer
- func (ix *Indexer) Backlinks(ctx context.Context, spaceId string, target anyuri.URI, q LinkQuery) (docs []LinkDoc, more bool, err error)
- func (ix *Indexer) BacklinksAll(ctx context.Context, target anyuri.URI, q LinkQuery) ([]SpaceBacklinks, error)
- func (ix *Indexer) ChunkRunes() int
- func (ix *Indexer) Close() error
- func (ix *Indexer) EmbedHardware() (Hardware, bool)
- func (ix *Indexer) HasEmbedder() bool
- func (ix *Indexer) Links(ctx context.Context, spaceId, objectId, dataset, recordId string, q LinkQuery) (docs []LinkDoc, more bool, err error)
- func (ix *Indexer) Search(ctx context.Context, spaceId string, req api.SearchRequest, host HostFilter) (api.SearchResponse, error)
- func (ix *Indexer) SetEmbedThreads(n int)
- func (ix *Indexer) Start(ctx context.Context)
- func (ix *Indexer) Sync(ctx context.Context) error
- func (ix *Indexer) SyncSpace(ctx context.Context, sp space.Space) error
- type LinkDoc
- type LinkOps
- type LinkQuery
- type Ollama
- type OpenAI
- type Options
- type ProcessUpdate
- type SpaceBacklinks
- type Store
- func (s *Store) Apply(ctx context.Context, spaceId string, ups []DocUpsert, dels []string, ...) error
- func (s *Store) ApplyLinks(ctx context.Context, spaceId string, ops *LinkOps) error
- func (s *Store) ApplyPage(ctx context.Context, spaceId string, ups []DocUpsert, dels []string, ...) ([]string, error)
- func (s *Store) Backlinks(ctx context.Context, spaceId, key string, byObject bool, kinds []string, ...) (docs []LinkDoc, more bool, err error)
- func (s *Store) Close() error
- func (s *Store) Cursor(ctx context.Context, spaceId string) (uint64, string, error)
- func (s *Store) Dim() int
- func (s *Store) DocHashes(ctx context.Context, spaceId, idPrefix string) (map[string]string, error)
- func (s *Store) DocHashesByRecords(ctx context.Context, spaceId string, bases []string) (map[string]string, error)
- func (s *Store) DropSpace(ctx context.Context, spaceId string) error
- func (s *Store) EnsureDim(ctx context.Context, dim int) error
- func (s *Store) EnsureVectorIndex(ctx context.Context, spaceId string) (bool, error)
- func (s *Store) FilterTerms(ctx context.Context, spaceId string, hits []Hit, require, exclude []string) ([]Hit, error)
- func (s *Store) LinkIds(ctx context.Context, spaceId, prefix string) (map[string]StoredLink, error)
- func (s *Store) LinkSpaces(ctx context.Context) ([]string, error)
- func (s *Store) Links(ctx context.Context, spaceId, prefix string, kinds []string, limit int) (docs []LinkDoc, more bool, err error)
- func (s *Store) LinksBackfillNeeded(ctx context.Context, spaceId string) (bool, error)
- func (s *Store) LinksVersion(ctx context.Context, spaceId string) (int, error)
- func (s *Store) Pending(ctx context.Context, spaceId string, limit int) (ids []string, texts []string, err error)
- func (s *Store) PendingCount(ctx context.Context, spaceId string) (int, error)
- func (s *Store) PinChunkRunes(ctx context.Context, n int) error
- func (s *Store) ResetLinks(ctx context.Context, spaceId string) error
- func (s *Store) SearchFTS(ctx context.Context, spaceId, q string, scopes []string, limit int) ([]Hit, error)
- func (s *Store) SearchFTSQuery(ctx context.Context, spaceId string, fq FTSQuery, scopes []string, limit int) ([]Hit, error)
- func (s *Store) SearchVector(ctx context.Context, spaceId string, vec []float32, scopes []string, limit int, ...) ([]Hit, error)
- func (s *Store) SetCursor(ctx context.Context, spaceId string, seq uint64, generation string) error
- func (s *Store) SetFTSParams(b, k1, titleWeight float64)
- func (s *Store) SetVectorMode(mode string)
- func (s *Store) SetVectors(ctx context.Context, spaceId string, ids []string, vecs [][]float32) error
- func (s *Store) StampLinksVersion(ctx context.Context, spaceId string) error
- func (s *Store) UnstampLinks(ctx context.Context, spaceId string) error
- type StoredLink
Constants ¶
const ( ProcessStarted = "started" ProcessProgress = "progress" ProcessDone = "done" ProcessFailed = "failed" ProcessCancelled = "cancelled" )
ProcessUpdate phases — each work unit reports started once, then progress (per landed batch / page / ~5s of download, plus a periodic heartbeat while an embed call runs long), then exactly one of done/failed/cancelled — cancelled when the owning context ends mid-work (space dropped, shutdown).
const ( // ProcessKindEmbed — a per-space vector drain: Done/Total count // docs (Total from the pending count, re-read per round). ProcessKindEmbed = "embed" // ProcessKindFTS — a per-space chunk/advance pass: Done counts // processed changes, Total is unknown (the change feed has no // backlog count). Reported only past AnnounceAfter — routine // debounced advances stay silent. ProcessKindFTS = "fts" // ProcessKindModelDownload — the embedding-model fetch: Done/Total // are bytes (Total 0 until the server reports a length), Name the // model file name. Account-global, not per-space. ProcessKindModelDownload = "model_download" // ProcessKindLinksBackfill — a per-space rebuild of the link index // from the records (a db that predates the link sink's layout): // Done counts objects, Total is unknown. Reported past // AnnounceAfter like an advance. ProcessKindLinksBackfill = "links_backfill" )
ProcessUpdate kinds — which pipeline the update reports on.
const ( // DefaultChunkRunes is the split target: ~500 tokens of prose, well // inside the local embedder's 2048-token clamp so every chunk embeds // whole, and a lexical unit small enough that a hit's data is the // matching passage rather than the whole mail. DefaultChunkRunes = 2000 )
Long records are split into several index docs. A record longer than the bound would otherwise be one oversized `data` value on every hit, and carry vector recall only for the head its embedding covered. Chunking is generic: the worker splits every chunker's entry, so chunkers keep emitting one entry per record and never learn about it.
Doc ids: chunk 0 keeps the record's id (`objectId:dataset:recordId`); chunk n > 0 is that id + chunkSep + n. chunkSep is U+001F, a control byte the SDK's id patterns never admit (auto ids are CIDs, user ids default to `[A-Za-z0-9._:-]+`), so a record's docs occupy exactly the primary-key range [base, base+" ") — every byte a real id can follow `base` with sorts at or above 0x20 — and record-level removal stays a single seek (recordUpper).
Variables ¶
ErrEmbedderUnavailable wraps query-time embedding failures: the embedder is configured but not currently reachable. Vector-mode searches surface it as a retryable condition; hybrid degrades to FTS instead.
var ErrIndexRebuildRequired = errors.New("indexer: index rebuild required")
ErrIndexRebuildRequired marks a store-open (or dimension-check) failure whose ONLY remedy is deleting the index db and letting it rebuild from the CRDT — a schema version this build can't read, or a vector dimension that contradicts the one the db was built with. Nothing in the db is salvageable and no migration exists: the index is a derived cache, so the fix is always "remove it, it re-indexes on the next change" (docs/13-index.md).
It exists so hosts can key recovery UI off errors.Is instead of the error text. The mobile bridges are the consumers: the iOS c-archive (mobile/ios) maps it to start code 4 (indexRebuildRequired), which is the whole reason a boot failure now carries a code AND a message. The Android gomobile bind (mobile/android) surfaces only the string today and its host substring-matches it; that stays working (see the wrapping note on the three call sites in store.go) until Android adopts the same code.
Wrapped as a PREFIX, so the human-readable remedy — which db to remove — stays in the message the host shows the user.
Functions ¶
This section is empty.
Types ¶
type DocUpsert ¶
type DocUpsert struct {
Entry index.IndexEntry
Chunk int // which chunk of the record Entry.Data holds (expandEntry)
Vector []float32
}
DocUpsert pairs an entry with its (optional) embedding. A nil Vector while the store has a vector index stores the doc as pending.
type Embedder ¶
type Embedder interface {
// EmbedDocs embeds passages for storage, one vector per text. On
// error it may return the vectors of a leading prefix of texts
// alongside the error; the caller lands those and retries the rest.
EmbedDocs(ctx context.Context, texts []string) ([][]float32, error)
// EmbedQuery embeds a single search query.
EmbedQuery(ctx context.Context, text string) ([]float32, error)
// Dim reports the embedding dimension (probing the backend if needed).
Dim(ctx context.Context) (int, error)
}
Embedder turns text into vectors. A nil Embedder on the Indexer means FTS-only operation: documents are stored without vectors and the vector search leg is unavailable.
Documents and queries embed through separate methods because retrieval models distinguish the two roles (task prompts / instructions); both must produce vectors of the same dimension.
func NewEmbedder ¶
func NewEmbedder(cfg config.Index, modelsDir, legacyModelsDir string, onProcess func(ProcessUpdate)) (Embedder, error)
NewEmbedder constructs the configured embedding client. Returns (nil, nil) for "none" — the indexer then runs FTS-only. A bare "" also maps to FTS-only: config.Load defaults it to "auto", so "" only survives when a config file sets it explicitly (the pre-"none" opt-out syntax). modelsDir hosts the local embedder's downloaded model (shared across accounts, <root>/models); legacyModelsDir is the pre-per-account location (<dataDir>/index/models), used instead when the model file already exists there.
type FTSQuery ¶
FTSQuery is the full-text query spec. Query is the `$search` string — phrases ("...") and prefixes (foo*) in it are honored by the engine. DefaultAnd makes bare terms required (AND) instead of OR. Require / Exclude are extra must / must-not terms ($require / $exclude); each may itself be a phrase or prefix.
type Group ¶
Group is one record in a reply: its best-ranked chunk plus, when asked for, the next best chunks that matched within the search window.
type Hardware ¶
type Hardware struct {
OS string `json:"os"`
Arch string `json:"arch"`
CPUs int `json:"cpus"`
Threads int `json:"threads"`
GpuLayers int `json:"gpuLayers"`
LibDir string `json:"libDir,omitempty"`
LibVersion string `json:"libVersion,omitempty"` // llama.cpp release + platform, from the libs' VERSION stamp
Backends []string `json:"backends,omitempty"` // "Vulkan from libggml-vulkan.so"
Devices []string `json:"devices,omitempty"` // "AMD ... (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | ..."
SystemInfo string `json:"systemInfo,omitempty"` // llama_print_system_info()
}
Hardware is what the embedder actually runs on, as llama.cpp reports it: the backends it registered, the devices they found, and the libs they came from. The child collects it once at startup and hands it to the server in the `ready` frame, which logs it and keeps it — the input for correlating speed and crashes with hardware later.
type Hit ¶
type Hit struct {
Scope string
ObjectId string
Dataset string
RecordId string
Chunk int // 0-based chunk of the record this doc holds (chunk.go)
Data string
Score float64
}
Hit is one search result row. Score semantics depend on the leg: BM25 score (higher = better) for FTS, RRF score after fusion; the vector leg's raw cosine distance is folded before it reaches callers.
type HostFilter ¶
type HostFilter interface {
// Resolve lists the matching object ids. With max > 0 it stops after
// max ids and reports more when the set continues past them.
Resolve(ctx context.Context, max int) (ids []string, more bool, err error)
// Match reports which of ids satisfy the filter, in one read.
Match(ctx context.Context, ids []string) (map[string]bool, error)
}
HostFilter is a request's object filter: a condition over the hit's host object row. Those rows live in the SDK's store, not the index, so the caller supplies the two reads and hostSet decides how to use them (docs/13-index.md § Filtering by object).
type Indexer ¶
type Indexer struct {
// contains filtered or unexported fields
}
Indexer is the phase-2 consumer of the chunker contract: per space it keeps a cursor over the SDK's change feed and mirrors chunker output into the local Store. FTS lands synchronously on the advance path; embedding runs on a parallel per-space loop so a slow embedder never delays the cursor (docs wait in `pending`).
func (*Indexer) Backlinks ¶
func (ix *Indexer) Backlinks(ctx context.Context, spaceId string, target anyuri.URI, q LinkQuery) (docs []LinkDoc, more bool, err error)
Backlinks returns the edges pointing at target in spaceId. When target names an object (an `o` reference without a record, or a `p` reference to one of its values), every edge to the object OR any of its records / values is returned — the caller splits them on Target.IsPart(). A record target, an identity or a file returns the edges to exactly that target. more reports that the read was cut at the limit.
func (*Indexer) BacklinksAll ¶
func (ix *Indexer) BacklinksAll(ctx context.Context, target anyuri.URI, q LinkQuery) ([]SpaceBacklinks, error)
BacklinksAll runs Backlinks over every indexed space — the device holds only spaces this account is a member of, so the result is access-filtered by construction. Spaces are read in name order; a space that fails to read is skipped.
func (*Indexer) ChunkRunes ¶
ChunkRunes is the resolved split target the workers write docs on — the authority for Store.PinChunkRunes, so the pin always describes the boundaries the chunker is actually using.
func (*Indexer) EmbedHardware ¶
EmbedHardware reports what the local embedder runs on — backends, devices and lib version as llama.cpp reports them — or false for an embedder that does not decode on this machine. Logged at child start; the accessor exists for hardware/error/speed statistics.
func (*Indexer) HasEmbedder ¶
HasEmbedder reports whether the vector pipeline is active.
func (*Indexer) Links ¶
func (ix *Indexer) Links(ctx context.Context, spaceId, objectId, dataset, recordId string, q LinkQuery) (docs []LinkDoc, more bool, err error)
Links returns the edges whose source is the object, one of its datasets or one of its records — forward links, "what does this link to".
func (*Indexer) Search ¶
func (ix *Indexer) Search(ctx context.Context, spaceId string, req api.SearchRequest, host HostFilter) (api.SearchResponse, error)
Search runs the requested mode over the space's local index. The caller validates mode / scopes / limit / passages; this layer only degrades hybrid→fts when the embedder is missing or the query embedding fails. limit counts records: each leg reads until its window covers enough distinct records (legCover), the legs fuse per chunk, and groupHits collapses chunks into records. Order of work: query embedding, then the vector leg (eager, may re-query), then the lexical cursor — so no read tx is ever held across the embed wait or another store call. host, when non-nil, is the request's object filter: it is probed once (newHostSet) and both legs keep only hits whose object it admits, so limit counts MATCHING records.
func (*Indexer) SetEmbedThreads ¶
SetEmbedThreads changes the CPU budget the local embedder decodes with; 0 restores the default (NumCPU()-1). It takes effect when the embedder child next spawns, and is a no-op for embedders that don't decode on this machine (ollama / openai / none). This is the seam a settings surface plugs into — index.local.threads seeds the value.
func (*Indexer) Start ¶
Start lists current spaces, spawns a worker per indexable space, and subscribes to space-list changes for live discovery. Non-blocking. The passed ctx bounds all background work — cancel it (or call Close) to stop.
type LinkDoc ¶
type LinkDoc struct {
ObjectId string
Dataset string
RecordId string
TypeId string // property-value sources only
Field string // runtime-record sources with several link fields
Kind string
Target anyuri.URI
ApplySeq uint64
}
LinkDoc is one stored edge.
type LinkOps ¶
type LinkOps struct {
Ups []index.LinkEntry
Seqs []uint64 // ApplySeq per Ups entry
Dels []string
Touched []string
}
LinkOps is one page's link changes, already diffed against the store: Dels are exact doc ids to remove, Ups the edges to write, Touched the liveness keys either side named. Structural evictions (object / dataset prefixes) ride the page's shared prefix deletes, which the sink applies to its collection too.
type LinkQuery ¶
type LinkQuery struct {
// Kinds keeps only these edge kinds (empty = all).
Kinds []string
// Limit bounds the read (0 = the store cap).
Limit int
}
LinkQuery narrows a link read.
type Ollama ¶ added in v0.2.2
type Ollama struct {
BaseURL string
Model string
HTTP *http.Client
// DocPrompt / QueryPrompt are the model's task prefixes; %s is
// replaced with the content. Empty disables prefixing.
DocPrompt string
QueryPrompt string
}
Ollama talks to a local Ollama's /api/embed endpoint (batch input).
type OpenAI ¶ added in v0.2.2
OpenAI talks to an OpenAI-compatible POST {baseURL}/embeddings API. Any gateway implementing that shape works — set BaseUrl accordingly.
func NewOpenAI ¶ added in v0.2.2
NewOpenAI returns a client for the given endpoint. baseURL and model are both required — there is no default provider — and the factory checks them before constructing one.
type Options ¶
type Options struct {
// Embedder is the vector pipeline's embedding client; nil = FTS-only.
Embedder Embedder
// BatchLimit is the ChangedSince page size, which also bounds one
// Apply transaction. Default 256: FTS-insert throughput plateaus
// there (~65k docs/s file-backed vs ~52k at 16) while one page
// stays a few ms — larger pages add latency, not throughput.
BatchLimit int
// EmbedBatch is how many pending docs one EmbedDocs call carries.
// Default 64: local Ollama (embeddinggemma) reaches ~97% of its max
// throughput there (76.5 texts/s vs 78.6 at 128) at half the
// per-call latency; vector insert (~30k vecs/s) never bottlenecks.
EmbedBatch int
// EmbedConcurrency is how many EmbedBatch chunks the embed loop embeds
// in parallel per round. Default 1 (sequential — right for the local
// child, which serves one frame at a time). Raise it for an online
// API (openai/auto) where parallel requests are the throughput win.
EmbedConcurrency int
// Debounce delays an advance after a dirty signal so write bursts
// coalesce into one page (and fuller embed batches). Default 250ms —
// a no-op advance is sub-ms, so this dial trades only freshness.
Debounce time.Duration
// RetryBackoff is the wait after a failed advance. Default 5s.
RetryBackoff time.Duration
// PendingEvery is the embed loop's catch-up/retry tick (the nudge
// channel covers the normal path). Default 1m.
PendingEvery time.Duration
// OnProcess, when set, receives indexing lifecycle updates — per
// ProcessKind*, one started/progress*/terminal sequence per unit
// of work (embed drain, advance pass, model download). The server
// bridges these onto the process view (docs/22-processes.md);
// nil = no reporting. Called from indexer goroutines — must not
// block.
OnProcess func(ProcessUpdate)
// OnLinks, when set, receives the target keys (canonical object
// references, or the target itself for identities and files) whose
// backlinks changed in a landed page — the liveness signal a client
// panel refreshes on. Called from the advance goroutine after the
// page committed; must not block.
OnLinks func(spaceId string, targets []string)
// AnnounceAfter gates fts/embed reporting on elapsed work time: a
// pass/drain announces only once it has been running this long, so
// usual indexing (one message, one edit — done in well under a
// second) never flashes through the view. Time, not a queue-size
// threshold, because cost per doc varies ~50× with text length and
// hardware (measured: local CPU embeds ~13 short chat docs/s but
// ~2 long editor windows/s). Default 3s; negative = announce
// immediately (tests). The model download always announces — it is
// long by definition and its absence is the state worth showing.
AnnounceAfter time.Duration
// FtsWeight / VectorWeight scale each leg's RRF contribution in
// hybrid mode. Default 1 each (plain RRF).
FtsWeight float64
VectorWeight float64
// FTSDefaultAnd makes the lexical leg require ALL query terms (AND)
// instead of the default any-term (OR). Higher precision, lower recall
// — off by default ($defaultOperator). Phrase/prefix and per-request
// Require/Exclude still work regardless.
FTSDefaultAnd bool
// AdaptiveWeights scales the FTS leg's weight by its per-query
// confidence (legConfidence) — auto-down-weighting BM25 when its
// scores are flat/weak (e.g. paraphrastic corpora), so a weak lexical
// leg can't drag hybrid below the dense leg. Vector weight is left
// alone (cosine is uncalibrated). Off by default.
AdaptiveWeights bool
// ChunkRunes is the split target for long records (chunk.go): an
// entry longer than this many runes is indexed as several chunk
// docs. <= 0 = DefaultChunkRunes.
ChunkRunes int
// MinVectorSim drops vector hits below this cosine similarity before
// fusion. Default 0 = the legacy "> 0" floor.
MinVectorSim float64
// StopWords strips a built-in stop list from the FTS-leg query (the
// vector leg always gets the full query). Default off in the zero
// Options; OpenIndexer turns it on unless config disables it.
StopWords bool
// QueryEmbedTimeout bounds the query embedding inside Search, for
// every embedder: past it hybrid degrades to fts (vectorStatus
// unavailable) and mode=vector fails as embedder-unavailable, so a
// cold model load, a wedged child or a slow API never holds a
// search — the local child already serves queries ahead of doc
// frames, so in normal operation the budget is far from reached.
// Default 5s.
QueryEmbedTimeout time.Duration
}
Options tunes the indexer. Zero values pick the defaults noted; the numeric defaults are measured (bench_test.go, docs/13-index.md § tuning), not guessed.
type ProcessUpdate ¶
type ProcessUpdate struct {
Kind string // ProcessKind*
SpaceId string // per-space kinds; empty for model download
Name string // model file name (model download only)
Phase string
Done int64
Total int64 // 0 = unknown
Message string
}
ProcessUpdate is one Options.OnProcess report. Message carries the failure detail on ProcessFailed (log-grade — the server never puts it on the wire), empty otherwise.
type SpaceBacklinks ¶
SpaceBacklinks is one space's share of an account-wide read.
type Store ¶
type Store struct {
// contains filtered or unexported fields
}
Store is the indexer-owned any-store database: one collection per space, each carrying a BM25 full-text index on `data` and (when dim > 0) an IVF-SQ cosine vector index on `vector`.
Doc shape: {id: dataset+"/"+recordId, scope, objectId, dataset, recordId, data, applySeq, vector?, pending?}. `pending: 1` marks a doc whose text awaits embedding — the embed loop drains them; the field is removed once the vector lands.
func OpenStore ¶
OpenStore opens (or creates) the index DB at path. dim is the configured vector dimension; 0 means "unknown — learn it from the first successful embedding" (EnsureDim). embedderConfigured turns on pending-marking even before the dimension is known. A dim change against an existing DB is a hard error — the index must be rebuilt.
func OpenStoreInMemory ¶
OpenStoreInMemory opens a throwaway in-memory store (tests).
func (*Store) Apply ¶
func (s *Store) Apply(ctx context.Context, spaceId string, ups []DocUpsert, dels []string, prefixDels []string) error
Apply lands one advance page in a single write transaction: structural prefix deletes first (object deletions / type-detach evictions — ':'-terminated id prefixes), then doc deletions — each id removes that doc AND every chunk under it (the record range, chunk.go; a chunk id has no chunks of its own, so passing one deletes exactly it; missing ids are a no-op) — then upserts (full-doc replace — a re-written doc goes back to pending until re-embedded). Atomic with the page, so eviction can never race the cursor.
A removal WINS over an upsert of the same doc in one page: collectObject appends an object-wide prefix delete mid-loop when it finds the object tombstoned, by which point earlier chunkers have already queued upserts for it. Writing those would resurrect docs of an object nothing will ever re-stream, so they are dropped rather than ordered around.
func (*Store) ApplyLinks ¶
ApplyLinks lands link ops on their own — the backfill's per-page write, text docs untouched. One transaction per call.
func (*Store) ApplyPage ¶
func (s *Store) ApplyPage(ctx context.Context, spaceId string, ups []DocUpsert, dels []string, prefixDels, textPrefixDels []string, links *LinkOps) ([]string, error)
ApplyPage is Apply plus the page's link ops, landed on the link collection in the same transaction (links_store.go): the shared structural prefixes evict edges as they evict text docs, while textPrefixDels evict text docs only (a runtime dataset that lost its search mapping but keeps link fields). Returns the target keys whose edge set changed (the liveness signal).
func (*Store) Backlinks ¶
func (s *Store) Backlinks(ctx context.Context, spaceId, key string, byObject bool, kinds []string, limit int) (docs []LinkDoc, more bool, err error)
Backlinks returns the edges pointing at key: every target that belongs to the object when byObject (the object itself, its records, its property values), or exactly the target when not. kinds narrows (empty = all); limit bounds the read (0 = the cap). Ordered by id — source object, dataset, record. more reports that the read was cut at limit.
func (*Store) Cursor ¶
Cursor returns the last indexed ApplySeq for the space (0 = never) and the SDK generation that seq belongs to ("" on rows written before the generation was tracked, and on a never-indexed space).
func (*Store) DocHashes ¶
DocHashes returns id→hash for every stored doc under the id prefix (objectId:dataset:). The reconcile diff uses it to find vanished and changed docs without trusting the chunker to enumerate deletions.
func (*Store) DocHashesByRecords ¶
func (s *Store) DocHashesByRecords(ctx context.Context, spaceId string, bases []string) (map[string]string, error)
DocHashesByRecords returns id→hash for every chunk doc of the given record ids (base doc ids, chunk.go) — one primary-key range seek per record. The incremental stream path diffs a re-streamed record's new chunk set against it.
func (*Store) DropSpace ¶
DropSpace removes the space's collection and cursor (space deleted or left).
func (*Store) EnsureDim ¶
EnsureDim records the dimension learned from the first successful embedding. A no-op when it matches the known dim; an ErrIndexRebuildRequired when the embedder's output contradicts what this DB was built with (model changed under a populated index) — the vectors already stored are unusable, so the db has to go.
Unreachable on mobile today (vector is off in both binds), wrapped anyway so the host contract doesn't depend on that staying true.
func (*Store) EnsureVectorIndex ¶
EnsureVectorIndex creates the space's vector index once at least one embedded doc exists (the default IVF-SQ mode trains quantizers from existing docs, so it can't be created empty; the others also wait so the first build sees real data). Returns whether the index exists after the call. Idempotent and cheap once created (cached). The strategy is chosen by vectorIndexParams — default IVF-SQ (cheap incremental ingest, churn-friendly); HNSW (btree) is opt-in for higher recall (docs/search/README.md § index mode).
func (*Store) FilterTerms ¶
func (s *Store) FilterTerms(ctx context.Context, spaceId string, hits []Hit, require, exclude []string) ([]Hit, error)
FilterTerms keeps only the hits that satisfy the Require / Exclude terms — the same $require / $exclude semantics SearchFTSQuery applies to the lexical leg, re-checked against the FTS index for hits that arrived some other way (the vector leg). One indexed query restricted to the hits' doc ids: any-store costs the pk restriction against the posting lists and probes the text index per candidate when that is cheaper (any-store v2.0.1 — before it, the $text predicate always drove and this cost the term's whole posting list). The analyzer decides "contains", so a phrase or prefix term behaves exactly as it does in the lexical leg. Nothing to enforce (no terms, no hits) returns hits unchanged.
Negated clauses only tombstone docs a positive clause already scored, so an exclude-only filter is run inverted: match the excluded terms as shoulds and drop whatever comes back.
func (*Store) LinkIds ¶
LinkIds returns the stored edge ids under prefix (an object, `objectId:`), each with its liveness key — the worker diffs a changed object's new edge set against it.
func (*Store) LinkSpaces ¶
LinkSpaces lists the spaces holding a link collection — the account-wide read iterates them.
func (*Store) Links ¶
func (s *Store) Links(ctx context.Context, spaceId, prefix string, kinds []string, limit int) (docs []LinkDoc, more bool, err error)
Links returns the edges whose source is under prefix — an object (`objectId:`), one of its datasets (`objectId:dataset:`) or one record (`objectId:dataset:recordId:`) — narrowed by kinds.
A record prefix ends in the control-byte separator (linkRecordPrefix) and is bounded by linkRecordUpper; the others end in ':' and by prefixUpper.
func (*Store) LinksBackfillNeeded ¶
LinksBackfillNeeded reports whether the space's edges predate the current link-sink layout: an indexed space (cursor > 0) whose stamp is behind. A never-indexed space needs none — its first advance extracts everything and stamps.
func (*Store) LinksVersion ¶
LinksVersion reads the link-sink layout version stamped on the space's cursor row (0 = never stamped).
func (*Store) Pending ¶
func (s *Store) Pending(ctx context.Context, spaceId string, limit int) (ids []string, texts []string, err error)
Pending returns up to limit docs awaiting embedding (only docs with non-empty text ever carry the pending mark — see Apply).
func (*Store) PendingCount ¶
PendingCount reports how many docs still await embedding — the denominator for embed-progress reporting (rides the sparse pending index, so it stays cheap).
func (*Store) PinChunkRunes ¶
PinChunkRunes records the chunk target this index's docs are written on, and refuses a db written on a different one — it decides every doc id and every doc's text, so mixing boundaries leaves the same text ranking differently by when it was last written, with stale trailing chunks under the old scheme.
Separate from OpenStore because the authority is the indexer's resolved Options.ChunkRunes, which the caller only has after the store exists: taking it here means the pin can never describe a boundary the chunker isn't using (OpenIndexer passes Indexer.ChunkRunes()). A db written before the pin existed carries none and is adopted — and pinned on the spot, since nothing else writes that row.
func (*Store) ResetLinks ¶
ResetLinks drops the space's edges ahead of a backfill.
func (*Store) SearchFTS ¶
func (s *Store) SearchFTS(ctx context.Context, spaceId, q string, scopes []string, limit int) ([]Hit, error)
SearchFTS runs the BM25 leg. Hits come back ranked by descending score; docs with empty data never match (nothing was indexed).
func (*Store) SearchFTSQuery ¶
func (s *Store) SearchFTSQuery(ctx context.Context, spaceId string, fq FTSQuery, scopes []string, limit int) ([]Hit, error)
SearchFTSQuery runs the BM25(F) leg with full operator support and returns the first limit hits in rank order (0 = every match).
func (*Store) SearchVector ¶
func (s *Store) SearchVector(ctx context.Context, spaceId string, vec []float32, scopes []string, limit int, minSim float64) ([]Hit, error)
SearchVector runs the ANN leg: nearest-first by cosine distance. Score is folded to similarity (1 - distance) so "higher = better" holds across legs. While the vector pipeline is frozen (dimension never learned, or no embedded docs in the space yet) it returns no hits rather than erroring — search degrades, never breaks.
minSim is the cosine-similarity floor: hits at or below it are dropped. The effective floor is max(minSim, smallest-positive) — a similarity must always be > 0 (cosine distance < 1) to carry any signal.
func (*Store) SetCursor ¶
SetCursor persists the space cursor and the generation it belongs to.
MERGES rather than replaces: an empty generation leaves a stamped one in place. A caller reaches the advance loop with no generation in hand whenever the boot-time read failed (alignIndex warns and returns) or never ran (SyncSpace builds a worker directly) — replacing the row there would erase the stamp, and a later SDK store rebuild would then pass both re-index triggers unnoticed: the cursor freezes the index on a renumbered applySeq axis with no error anywhere.
func (*Store) SetFTSParams ¶
SetFTSParams sets the BM25 tuning for FTS indexes created after the call. All three are index-creation params: b/k1 are baked in, and whether titleWeight is zero decides the index's FIELD SET (`data` alone, or `data`+`title`), so turning a title boost on for an existing index needs a rebuild. Its value is then read at query time.
func (*Store) SetVectorMode ¶
SetVectorMode picks the ANN index strategy for indexes created after the call (existing indexes keep their mode until rebuilt). Empty = default.
func (*Store) SetVectors ¶
func (s *Store) SetVectors(ctx context.Context, spaceId string, ids []string, vecs [][]float32) error
SetVectors lands one embed batch in a single write transaction: $set vector + clear pending, update-only (a doc deleted since Pending is skipped, not resurrected).
func (*Store) StampLinksVersion ¶
StampLinksVersion marks the space's edges as written on the current layout (merged into the cursor row): after a backfill, and after the first page an advance lands.
type StoredLink ¶
type StoredLink struct {
Key string // liveness key (linkTargetKey)
}
StoredLink is what the worker's diff needs of a stored edge.