Documentation
¶
Overview ¶
Package sim implements a deterministic simulation testing (DST) harness for the GoGraph engine, modelled on TigerBeetle's VOPR. The simulator is seed-reproducible, single-goroutine, and tick-driven: it drives the real cypher.Engine against an in-memory store, maintains a shadow oracle model of what the graph must contain, and verifies ACID and graph invariants after every operation.
The whole point of the harness is determinism: a given seed always produces the exact same sequence of actors, operations, parameters, and injected faults, so any violation can be reproduced bit-for-bit from its seed alone. To preserve that property, every probabilistic decision anywhere in the package must draw from a single Seed; no other source of randomness (no global math/rand, no time.Now, no map-iteration ordering decisions) may influence control flow. The import half of that rule is enforced, not merely documented: TestSim_SeedIsTheOnlyRandomnessSource fails if any file other than this one imports a randomness package, and fails on crypto/rand or the auto-seeded math/rand (v1) even here. The seed itself is mandatory at the command line for the same reason — cmd/sim refuses to run without one rather than invent a value that would make the run unreproducible.
Concurrency contract ¶
No type in this package is safe for concurrent use. The simulator runs on a single goroutine and spawns none; the determinism guarantee depends on a single, totally-ordered stream of draws from Seed. Sharing any value in this package across goroutines is a programmer error.
Index ¶
- Constants
- Variables
- func CheckCorruptImageRejected(ctx context.Context, seed uint64) error
- func DefaultVariantPair() (EngineVariant, EngineVariant)
- func ParallelAggregateVariantPair() (EngineVariant, EngineVariant)
- func ParallelScanVariantPair() (EngineVariant, EngineVariant)
- func RangeSeekVariantPair() (EngineVariant, EngineVariant)
- func ReadFixtureFile(rel ...string) ([]byte, error)
- func RecordTrace(ctx context.Context, cfg Config) (Trace, *SimReport, error)
- func ReplayInstructions(trace Trace) string
- func RunCountStore(ctx context.Context, cfg CountStoreConfig) (*CountStoreEvidence, *SimReport, error)
- func RunFluentQuery(ctx context.Context, cfg FluentQueryConfig) (*FluentQueryEvidence, *SimReport, error)
- func RunLabelIndexScoped(ctx context.Context, cfg LabelIndexScopedConfig) (*LabelIndexScopedEvidence, *SimReport, error)
- func RunMVCCSubstrateAborts(ctx context.Context, cfg MVCCContentionConfig) (*MVCCSubstrateResult, *MVCCContentionResult, error)
- func RunPageRankRanker(ctx context.Context, cfg PageRankRankerConfig) (*PageRankRankerEvidence, *SimReport, error)
- func RunSwarmWithMetricsOracle(ctx context.Context, sw *Swarm, goroutineSlack int) (SwarmResult, MetricsOracleResult, error)
- func RunTypedSchema(ctx context.Context, cfg TypedSchemaConfig) (*TypedSchemaEvidence, *SimReport, error)
- func SimEngineForServer() *cypher.Engine
- type AbuseFamily
- type AbuseOutcome
- type Actor
- type BoltAbuser
- type BoltAuthArm
- type BoltAuthEvidence
- type BoltBeginExtrasEvidence
- type BoltBookmarkArm
- type BoltCertRotationEvidence
- type BoltDBArm
- type BoltDecodeEvidence
- type BoltDecodeExchange
- type BoltDecodeHonest
- type BoltDecodeProbe
- type BoltDrainCommit
- type BoltDrainConfig
- type BoltDrainEvidence
- type BoltMetadataObs
- type BoltModeArm
- type BoltOffer
- type BoltRouteObs
- type BoltStreamEvidence
- type BoltStreamPage
- type BoltStreamRefusal
- type BoltStreamTxArm
- type BoltTimeoutArm
- type BoltTxListingRow
- type BoltTxPlanRow
- type BoltTxQuotaBegin
- type BoltTxQuotaEvidence
- type BoltTxRegistryEvidence
- type BoltTxTerminateCall
- type BoltTxTerminateEvidence
- type BoltTxTerminateRow
- type BoltTxWriteWindow
- type BoltVersionAbuse
- type BoltVersionArm
- type BoltVersionAuthProbes
- type BoltVersionEntity
- type BoltVersionMatrixEvidence
- type BoltVersionNegotiation
- type BoltVersionTemporal
- type BoltVersionTemporalRefs
- type BoltVersionZoneRef
- type BoundedChurnWriter
- type CSRHeaderFacts
- type CertRotationStep
- type CheckSelection
- type CheckpointCadenceConfig
- type CheckpointCadenceEvidence
- type CheckpointConfig
- type ConcurrentConfig
- type ConcurrentMix
- type ConcurrentResult
- type Config
- type CountStoreConfig
- type CountStoreEvidence
- type CountStoreProbes
- type CountStoreWriter
- type CounterUniqueness
- type CoverageBucket
- type CoverageSummary
- type CoverageTracker
- type CrashConfig
- type CrashKind
- type CrashSchedule
- type CrossReleaseDiffResult
- type CrossReleaseDivergence
- type CrossReleaseUpgradeResult
- type DBTeardownConfig
- type DBTeardownEvidence
- type DiffResult
- type DiskConfig
- type EdgeByName
- type EdgePropsWriter
- type EdgeState
- type Engine
- type EngineAdapter
- func (a *EngineAdapter) ConstraintNames() []string
- func (a *EngineAdapter) CountSnapshot() count.Snapshot
- func (a *EngineAdapter) CountStoreCells() int
- func (a *EngineAdapter) EdgeCount() (int64, error)
- func (a *EngineAdapter) Explain(query string, params map[string]any) (string, error)
- func (a *EngineAdapter) ListIndexes() []string
- func (a *EngineAdapter) NodeCount() (int64, error)
- func (a *EngineAdapter) Profile(ctx context.Context, query string, params map[string]any) (string, error)
- func (a *EngineAdapter) Run(ctx context.Context, query string, params map[string]any) (Result, error)
- func (a *EngineAdapter) RunWrite(ctx context.Context, query string, params map[string]any) (Result, error)
- func (a *EngineAdapter) StatsTrackedPairs() int
- type EngineVariant
- type ExecMode
- type FluentQueryConfig
- type FluentQueryEvidence
- type FluentQueryProbes
- type GenerationSwapConfig
- type GenerationSwapEvidence
- type GraphIOCSVArm
- type GraphIOCancelObservation
- type GraphIOCapObservation
- type GraphIOGuardDecl
- type GraphIOGuardResult
- type GraphIOMutation
- type GraphIOPropsObservation
- type GraphIOSurfaceResult
- type GraphOracle
- func (o *GraphOracle) ApplyCreate(cypher string, params map[string]any) OracleResult
- func (o *GraphOracle) ApplyDelete(cypher string, params map[string]any) OracleResult
- func (o *GraphOracle) ApplyMalformed(cypher string, params map[string]any) OracleResult
- func (o *GraphOracle) ApplyMatch(cypher string, params map[string]any) OracleResult
- func (o *GraphOracle) ApplyMerge(cypher string, params map[string]any) OracleResult
- func (o *GraphOracle) BeginTx() *OracleTx
- func (o *GraphOracle) DeletedKnowsInstances() []KnowsInstance
- func (o *GraphOracle) EdgeCount() int
- func (o *GraphOracle) ExistenceOnEmail() bool
- func (o *GraphOracle) HasEdge(src, dst uint64, label string) bool
- func (o *GraphOracle) HasKnowsByName(a, b string) bool
- func (o *GraphOracle) HasNode(id uint64) bool
- func (o *GraphOracle) HasPersonName(name string) bool
- func (o *GraphOracle) KnowsEdgesByName() []EdgeByName
- func (o *GraphOracle) KnowsInstancesByEID() []KnowsInstance
- func (o *GraphOracle) NodeCount() int
- func (o *GraphOracle) NodeNames() []string
- func (o *GraphOracle) Ops() []OracleOp
- func (o *GraphOracle) SetExistenceOnEmail(active bool)
- func (o *GraphOracle) SetUniqueOnName(active bool)
- func (o *GraphOracle) String() string
- func (o *GraphOracle) TypedIDs() []int64
- func (o *GraphOracle) TypedNode(id int64) (map[string]any, bool)
- func (o *GraphOracle) UniqueOnName() bool
- type GroupCommitConfig
- type GroupCommitEvidence
- type GroupCommitFailAllResult
- type HelperOpResult
- type HelperRunResult
- type HonestReader
- type HonestWriter
- type IndexIntersectProbes
- type IndexSeekResults
- type IndexSpec
- type InvariantChecker
- func (c *InvariantChecker) Check(tick int64, oracle *GraphOracle, engine Engine) []Violation
- func (c *InvariantChecker) CheckDurability(tick int64, oracle *GraphOracle, engine Engine) []Violation
- func (c *InvariantChecker) ChecksRun() int
- func (c *InvariantChecker) HasViolations() bool
- func (c *InvariantChecker) Violations() []Violation
- type KnowsInstance
- type LabelIndexScopedConfig
- type LabelIndexScopedEvidence
- type LegacyCSRVerdict
- type LivenessConfig
- type LivenessOutcome
- type MVCCContentionConfig
- type MVCCContentionResult
- type MVCCSessionsConfig
- type MVCCSessionsResult
- type MVCCSubstrateConfig
- type MVCCSubstrateResult
- type MalformedSender
- type ManifestFraming
- type MergeRelWriter
- type MetricsObservation
- type MetricsOracle
- func (o *MetricsOracle) Check(before, after MetricsSnapshot, expectedWrites, expectedWriteErrors uint64, ...) MetricsOracleResult
- func (o *MetricsOracle) CheckGoroutineBaseline(before, after MetricsSnapshot, goroutineSlack int) MetricsOracleResult
- func (o *MetricsOracle) Restore()
- func (o *MetricsOracle) Snapshot() MetricsSnapshot
- type MetricsOracleResult
- type MetricsRunStats
- type MetricsSnapshot
- type NodeState
- type NullSemanticsWriter
- type Op
- type OpKind
- type OracleOp
- type OracleResult
- type OracleSnapshot
- type OracleTx
- func (t *OracleTx) Abort()
- func (t *OracleTx) AgeOf(name string) (any, bool)
- func (t *OracleTx) ApplyCreate(cypher string, params map[string]any) OracleResult
- func (t *OracleTx) ApplyCreateKnows(params map[string]any) OracleResult
- func (t *OracleTx) ApplyDelete(cypher string, params map[string]any) OracleResult
- func (t *OracleTx) ApplyMatch(cypher string, params map[string]any) OracleResult
- func (t *OracleTx) ApplyMerge(cypher string, params map[string]any) OracleResult
- func (t *OracleTx) Commit() error
- func (t *OracleTx) CreatedNames() []string
- func (t *OracleTx) HasPerson(name string) bool
- func (t *OracleTx) NodeCount() int
- func (t *OracleTx) NodeNames() []string
- func (t *OracleTx) Ops() []OracleOp
- func (t *OracleTx) PendingKnows(a, b string) bool
- type OrderingStats
- type OverloadActor
- type OverloadFamily
- type OverloadOutcome
- type OverloadReader
- type PageRankRankerConfig
- type PageRankRankerEvidence
- type ParityProbe
- type PatternShapesWriter
- type PendingState
- type PlanBaseline
- type PlanEngine
- type PriorReleaseHelper
- type PriorSnapshotFacts
- type RefreshExpectation
- type Registry
- type RequiredCounter
- type Result
- type Scenario
- type ScenarioCounterDecl
- type ScenarioSelector
- type SchemaChangeFamily
- type SchemaChangeOutcome
- type SchemaChanger
- func (SchemaChanger) Name() string
- func (SchemaChanger) PickFamily(seed *Seed) SchemaChangeFamily
- func (SchemaChanger) PickModernForm(seed *Seed) bool
- func (a SchemaChanger) Run(c *WireClient, family SchemaChangeFamily, modern bool) (SchemaChangeOutcome, error)
- func (a SchemaChanger) RunChecked(srv *SimServer, c *WireClient, family SchemaChangeFamily, modern bool) (SchemaChangeOutcome, []Violation, error)
- type SchemaModel
- type SchemaMutationWriter
- type ScriptedResult
- type Seed
- type ShrinkConfig
- type ShrinkResult
- type SimConn
- func (c *SimConn) Close() error
- func (c *SimConn) CloseWithError(err error) error
- func (c *SimConn) LocalAddr() net.Addr
- func (c *SimConn) Read(p []byte) (int, error)
- func (c *SimConn) ReadBuffered() int
- func (c *SimConn) RemoteAddr() net.Addr
- func (c *SimConn) SetDeadline(t time.Time) error
- func (c *SimConn) SetReadDeadline(t time.Time) error
- func (c *SimConn) SetWriteDeadline(t time.Time) error
- func (c *SimConn) Write(p []byte) (int, error)
- type SimDisk
- func (d *SimDisk) AppendCount() int64
- func (d *SimDisk) ArmDirSyncFaultForPath(dir string)
- func (d *SimDisk) ArmParentDirSyncFaultForPath(childPath string)
- func (d *SimDisk) ArmRemoveRollbackForPath(path string)
- func (d *SimDisk) ArmRemoveWritebackForPath(path string)
- func (d *SimDisk) ArmRenameFaultForPath(newPath string)
- func (d *SimDisk) ArmRenameRevokeBothForPath(newPath string)
- func (d *SimDisk) ArmRenameRollbackForPath(newPath string)
- func (d *SimDisk) ArmRenameWritebackForPath(newPath string)
- func (d *SimDisk) ArmSyncFaultAt(at int64)
- func (d *SimDisk) ArmSyncGateAt(at int64) *SyncGate
- func (d *SimDisk) ArmTornAppendAt(at, keepBytes int64)
- func (d *SimDisk) CorruptRange(path string, off int64, n int) error
- func (d *SimDisk) Crash()
- func (d *SimDisk) CrashHost()
- func (d *SimDisk) CrashProcess()
- func (d *SimDisk) DirSync(dir string) error
- func (d *SimDisk) DirSyncFaultCount() int64
- func (d *SimDisk) DurableImage(path string) ([]byte, error)
- func (d *SimDisk) DurableSize(path string) (int64, bool)
- func (d *SimDisk) Exists(path string) bool
- func (d *SimDisk) FaultRate() float64
- func (d *SimDisk) FaultedSectorCount(path string) int
- func (d *SimDisk) LastCrashDiscardedBytes() int64
- func (d *SimDisk) LastCrashKind() CrashKind
- func (d *SimDisk) LastCrashRemoveOutcome() (pending, restored int)
- func (d *SimDisk) LastCrashRenameOutcome() (pending, rolledBack int)
- func (d *SimDisk) MarkDataDurable(path string) error
- func (d *SimDisk) MkdirAll(dir string, _ fs.FileMode) error
- func (d *SimDisk) OpenFile(path string, flag int) (*SimFileHandle, error)
- func (d *SimDisk) ParentDirSync(childPath string) error
- func (d *SimDisk) PendingRemoveCount() int
- func (d *SimDisk) PendingRenameCount() int
- func (d *SimDisk) ReadDir(dir string) ([]fs.DirEntry, error)
- func (d *SimDisk) ReadFile(path string) ([]byte, error)
- func (d *SimDisk) Remove(path string) error
- func (d *SimDisk) RemoveAll(path string) error
- func (d *SimDisk) RemoveHitCount() int64
- func (d *SimDisk) RemoveHitCountForPath(path string) int64
- func (d *SimDisk) RemoveRollbackCount() int64
- func (d *SimDisk) RemoveWritebackCount() int64
- func (d *SimDisk) Rename(oldPath, newPath string) error
- func (d *SimDisk) RenameFaultCount() int64
- func (d *SimDisk) RenameRevokeBothCount() int64
- func (d *SimDisk) RenameRollbackCount() int64
- func (d *SimDisk) RenameWritebackCount() int64
- func (d *SimDisk) SetCapacity(capacityBytes int64, enospcOnSync bool)
- func (d *SimDisk) Snapshot() map[string][]byte
- func (d *SimDisk) Stat(path string) (fs.FileInfo, error)
- func (d *SimDisk) SyncCount() int64
- func (d *SimDisk) TruncatePath(path string, size int64) error
- type SimFileHandle
- func (h *SimFileHandle) Close() error
- func (h *SimFileHandle) Read(p []byte) (int, error)
- func (h *SimFileHandle) Seek(offset int64, whence int) (int64, error)
- func (h *SimFileHandle) Stat() (fs.FileInfo, error)
- func (h *SimFileHandle) Sync() error
- func (h *SimFileHandle) Truncate(size int64) error
- func (h *SimFileHandle) Write(p []byte) (int, error)
- type SimListener
- type SimReport
- type SimServer
- func NewSimServer(eng *cypher.Engine, clk clock.Clock) (*SimServer, error)
- func NewSimServerAuth(eng *cypher.Engine, clk clock.Clock, auth server.AuthHandler) (*SimServer, error)
- func NewSimServerInFlight(eng *cypher.Engine, clk clock.Clock, maxInFlight int) (*SimServer, error)
- func NewSimServerInboundBudget(eng *cypher.Engine, clk clock.Clock, maxInboundDecodeBytes int64) (*SimServer, error)
- func NewSimServerOwnedCloser(eng *cypher.Engine, clk clock.Clock, closer io.Closer, ...) (*SimServer, error)
- func NewSimServerTxRegistry(eng *cypher.Engine, listenerClk, serverClk clock.Clock, ...) (*SimServer, error)
- type SimStore
- func (s *SimStore) Checkpoint() error
- func (s *SimStore) Clean() bool
- func (s *SimStore) ClockNow() uint64
- func (s *SimStore) Close() error
- func (s *SimStore) Config() simStoreConfig
- func (s *SimStore) Crash()
- func (s *SimStore) CrashHost()
- func (s *SimStore) CrashProcess()
- func (s *SimStore) Engine() *cypher.Engine
- func (s *SimStore) Graph() *lpg.Graph[string, float64]
- func (s *SimStore) RecoveredIndexPopulation() cypher.RecoveredIndexPopulation
- func (s *SimStore) RecoveredMaxCommitTS() uint64
- func (s *SimStore) ResumedTxnSeq() uint64
- func (s *SimStore) WAL() *wal.Writer
- func (s *SimStore) WALOps() int
- type Simulator
- func (s *Simulator) CheckpointCount() int
- func (s *Simulator) Close() error
- func (s *Simulator) CrashCount() int
- func (s *Simulator) Disk() *SimDisk
- func (s *Simulator) NodeIDsCompared() int
- func (s *Simulator) Oracle() *GraphOracle
- func (s *Simulator) RejectedReads() int
- func (s *Simulator) RejectedWrites() int
- func (s *Simulator) ReplayedOps() int
- func (s *Simulator) Run(ctx context.Context) (*SimReport, error)
- type SlowConsumer
- type SlowConsumerResult
- type StatsEngine
- type StatsRegime
- func (k *StatsRegime) CheckRecovered(tick int64, engine StatsEngine) []Violation
- func (k *StatsRegime) CheckRefresh(tick int64, engine StatsEngine, expect RefreshExpectation) []Violation
- func (k *StatsRegime) Finish(tick int64) []Violation
- func (k *StatsRegime) PlanChanges() int
- func (k *StatsRegime) Refreshes() int
- func (k *StatsRegime) Refusals() int
- type SurfaceWriter
- type Swarm
- type SwarmConfig
- type SwarmResult
- type SwarmRun
- type SyncGate
- type Trace
- type TraceFault
- type TracedOp
- type TxnOversizeAttempt
- type TxnOversizeConfig
- type TxnOversizeEvidence
- type TxnOversizeReplayArm
- type TxnOversizeReplayEvidence
- type TypedSchemaConfig
- type TypedSchemaEvidence
- type TypedSchemaProbes
- func (p *TypedSchemaProbes) ArmWitness(ctx context.Context, sm *Simulator, tick int64) []Violation
- func (p *TypedSchemaProbes) EngineWrite(ctx context.Context, sm *Simulator, tick int64, op Op, want tsVerdict) []Violation
- func (p *TypedSchemaProbes) Evidence() *TypedSchemaEvidence
- func (p *TypedSchemaProbes) InstallEngineSchema(g *lpg.Graph[string, float64]) error
- func (p *TypedSchemaProbes) NodeBattery(tick int64, side *typedSchemaSide, perturb tsPerturb) []Violation
- func (p *TypedSchemaProbes) RecoveryPin(sm *Simulator, tick int64, epoch int, perturb tsPerturb) []Violation
- func (p *TypedSchemaProbes) SideWrite(tick int64, side *typedSchemaSide, spec tsOpSpec, perturb tsPerturb) []Violation
- func (p *TypedSchemaProbes) VerifyWitnesses(ctx context.Context, sm *Simulator, tick int64, phase string, ...) []Violation
- type TypedWriter
- type UpgradeConfig
- type UpgradeResult
- type Violation
- func CheckAccessPathParity(tick int64, _ *GraphOracle, engine PlanEngine, probes ...ParityProbe) []Violation
- func CheckCartesianNotification(tick int64, engine *EngineAdapter) []Violation
- func CheckCounterDeclShape(d ScenarioCounterDecl) []Violation
- func CheckCypherSurface(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
- func CheckCypherSurfaceEntity(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
- func CheckCypherSurfaceExtended(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
- func CheckCypherSurfaceGrouped(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
- func CheckCypherSurfaceOrdering(tick int64, oracle *GraphOracle, engine *EngineAdapter, st *OrderingStats) []Violation
- func CheckDDLCounters(tick int64, query string, before, after ddlSchemaNames, ...) []Violation
- func CheckEdgeProperties(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
- func CheckExprLiterals(tick int64, engine *EngineAdapter) []Violation
- func CheckGraphIOGuardDeclShape(decls []GraphIOGuardDecl) []Violation
- func CheckGraphIOGuards(r *GraphIOGuardResult) []Violation
- func CheckGraphIOSurface(r *GraphIOSurfaceResult) []Violation
- func CheckGraphIOSurfaceShape(r *GraphIOSurfaceResult) []Violation
- func CheckIndexConsistency(tick int64, _ *GraphOracle, engine indexConsistencyEngine, specs ...IndexSpec) []Violation
- func CheckMergeHandleCollision(tick int64, f *mergeHandleFixture, g *lpg.Graph[string, float64], ...) []Violation
- func CheckMergePairRelProps(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
- func CheckMergeRel(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
- func CheckMergeZeroDriverAbsent(tick int64, engine *EngineAdapter) []Violation
- func CheckNonDeterministicFuncs(tick int64, engine *EngineAdapter) []Violation
- func CheckNullSemantics(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
- func CheckOpCounters(tick int64, op Op, committed bool, got *exec.QueryCounters, ...) []Violation
- func CheckPatternShapes(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
- func CheckPlanStability(tick int64, base *PlanBaseline, engine PlanEngine) []Violation
- func CheckSchemaIntrospection(tick int64, model *SchemaModel, engine *EngineAdapter) []Violation
- func CheckSchemaMutation(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
- func CheckSearch(tick int64, oracle *GraphOracle, engine Engine) []Violation
- func CheckShortestPath(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
- func CheckTypedListPredicates(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
- func CheckTypedProperties(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
- func CheckTypedTemporalOrder(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
- func CheckVarlenPaths(tick int64, oracle *GraphOracle, engine *EngineAdapter, st *vleStats) []Violation
- type ViolationKind
- type VirtualClock
- type WALContiguityConfig
- type WALContiguityEvidence
- type WALGuardResult
- type WALLifecycleResult
- type WALWatermarkEvidence
- type WireClient
- func (c *WireClient) AuthenticateAs(principal, credentials string) (any, error)
- func (c *WireClient) Begin() (any, error)
- func (c *WireClient) BeginExtras(extra map[string]packstream.Value) (any, error)
- func (c *WireClient) BeginMode(mode string) (any, error)
- func (c *WireClient) Close() error
- func (c *WireClient) Commit() (any, error)
- func (c *WireClient) Conn() *SimConn
- func (c *WireClient) Connect(ctx context.Context) error
- func (c *WireClient) ConnectAs(ctx context.Context, principal, credentials string) (any, error)
- func (c *WireClient) Discard(n int64) (records []*proto.Record, terminal any, err error)
- func (c *WireClient) DiscardQID(n, qid int64) (records []*proto.Record, terminal any, err error)
- func (c *WireClient) Goodbye() error
- func (c *WireClient) Handshake(ctx context.Context) (proto.Version, error)
- func (c *WireClient) HandshakeOffering(ctx context.Context, offers ...proto.Version) (proto.Version, error)
- func (c *WireClient) HandshakeOfferingSlots(_ context.Context, slots [4]BoltOffer) (proto.Version, error)
- func (c *WireClient) Hello(extra map[string]packstream.Value) (any, error)
- func (c *WireClient) Logoff() (any, error)
- func (c *WireClient) Logon() (any, error)
- func (c *WireClient) LogonWith(auth map[string]packstream.Value) (any, error)
- func (c *WireClient) NetConn() net.Conn
- func (c *WireClient) Pull(n int64) (records []*proto.Record, terminal any, err error)
- func (c *WireClient) PullAll() (records []*proto.Record, terminal any, err error)
- func (c *WireClient) PullQID(n, qid int64) (records []*proto.Record, terminal any, err error)
- func (c *WireClient) Recv() (any, error)
- func (c *WireClient) RecvRaw() ([]byte, error)
- func (c *WireClient) Request(msg any) (any, error)
- func (c *WireClient) Reset() (any, error)
- func (c *WireClient) Rollback() (any, error)
- func (c *WireClient) Run(query string, params map[string]any) (any, error)
- func (c *WireClient) Version() proto.Version
- func (c *WireClient) WriteChunkedRaw(payload []byte) error
- func (c *WireClient) WriteRaw(p []byte) (int, error)
- type WireExchange
- type WireTranscript
- type Workload
- type XReleaseBuildOptions
Constants ¶
const ( // ArmBoltDrainOrdered is the drain-success arm: a commit is parked inside its // WAL fsync, Shutdown drains it, and only then is the store closed. ArmBoltDrainOrdered = "ordered-drain" // ArmBoltDrainExpiryDrainTimeout is the expiry arm driven by a DEADLINE, which // takes Shutdown's clamped drain-timeout branch (see the file comment). ArmBoltDrainExpiryDrainTimeout = "expiry-drain-timeout" // ArmBoltDrainExpiryCtxCancel is the expiry arm driven by an explicit cancel on // a context with no deadline, which is the only way to reach Shutdown's // `<-ctx.Done()` branch. ArmBoltDrainExpiryCtxCancel = "expiry-ctx-cancel" // ArmBoltDrainOnce is the publication arm: the store's close is made to FAIL, // so the value the server's sync.Once caches is a non-nil, freshly allocated // error whose IDENTITY across callers is a discriminating assertion. ArmBoltDrainOnce = "once-published" // ArmBoltDrainUnordered is the SENSITIVITY control: the store is closed with a // live client mid-session and no drain at all, which is what the ordering // exists to prevent. It is expected to FAIL the ordering and wire clauses. ArmBoltDrainUnordered = "unordered-control" // ArmBoltDrainFleet is the concurrent arm: several committers in flight when // Shutdown fires. Leak/no-panic/convergence guarded, NOT bit-reproducible. ArmBoltDrainFleet = "fleet-drain" )
The arm names, carried in the evidence and used by the arm-specific clauses.
const ( ScenarioCrashStorm = "crash-storm" ScenarioWriteHeavy = "write-heavy" ScenarioReadHeavy = "read-heavy" ScenarioSchemaChaos = "schema-chaos" ScenarioSearch = "search" ScenarioSearchCrash = "search-crash" ScenarioBadActors = "bad-actors" ScenarioOverload = "overload" ScenarioBulkVsOnline = "bulk-vs-online" ScenarioLongRunning = "long-running" ScenarioDiskFull = "disk-full" ScenarioMemPressure = "mem-pressure" ScenarioCPUStarvation = "cpu-starvation" ScenarioConstraintEnforce = "constraint-enforce" ScenarioTypeCoverage = "type-coverage" ScenarioCypherPaths = "cypher-paths" ScenarioEdgeProperties = "edge-properties" ScenarioIndexDiversity = "index-diversity" ScenarioCypherSurface = "cypher-surface" ScenarioPatternShapes = "pattern-shapes" ScenarioSchemaMutation = "schema-mutation" ScenarioMergeRel = "merge-rel" ScenarioNullSemantics = "null-semantics" ScenarioBoltAuth = "bolt-auth" ScenarioBoltCertRotation = "bolt-cert-rotation" ScenarioBoltTxRegistry = "bolt-tx-registry" ScenarioBoltTxQuota = "bolt-tx-quota" // ScenarioBoltShutdownDrain is the deterministic teardown scenario of rmp // #2483: Server.Shutdown's connection drain against the Options.Closer the // server owns. ScenarioBoltShutdownDrain = "bolt-shutdown-drain" // ScenarioBoltShutdownFleet is its concurrent sibling: the same drain against a // fleet of committers in flight. Registered separately because it is not // bit-reproducible. ScenarioBoltShutdownFleet = "bolt-shutdown-fleet" // ScenarioBoltStreaming is the deterministic streaming-semantics scenario of rmp // #2484: PULL n paging against an independent reference drain, the exact window a // partial DISCARD removes, the qid and second-RUN refusals, and cursor // accumulation up to the per-connection in-flight cap. ScenarioBoltStreaming = "bolt-streaming" // ScenarioBoltStreamingStall is its concurrent sibling: the same surface behind a // slow consumer that stalls mid-stream and then disconnects. Registered // separately because it is not bit-reproducible. ScenarioBoltStreamingStall = "bolt-streaming-stall" // ScenarioBoltBeginExtras is the deterministic BEGIN-extras scenario of rmp #2485: // bookmark causality across connections (and the proof that the token is ignored // rather than honoured), a client-supplied tx_timeout against its own control, // tx_metadata, the access mode, database selection, and the ROUTE payload. ScenarioBoltBeginExtras = "bolt-begin-extras" // ScenarioBoltVersionMatrix is the deterministic protocol-version scenario of // rmp #2486: 4.4, 5.0, 5.1 and 5.6 negotiated and driven side by side, with the // entity and zoned-datetime encodings required to DIFFER across the Bolt 5 // boundary while the decoded semantics stay identical at every version. ScenarioBoltVersionMatrix = "bolt-version-matrix" // ScenarioBoltDecodePressure is the deterministic inbound-decode scenario of rmp // #2487: the engine-wide (cross-connection) decode pool adjudicated against a // closed-form model of its per-slot charges, and the nesting-depth family that // separates the wire cap, the engine's parameter cap and the pool by the three // different codes they answer with. ScenarioBoltDecodePressure = "bolt-decode-pressure" // ScenarioBoltDecodeSwarm is its concurrent sibling: K connections pushing // large-collection parameters at one shared pool while an honest client works. // Registered separately because it is not bit-reproducible — and because the // aggregate vector is only reachable concurrently, since every charge is // released before its reply is written. ScenarioBoltDecodeSwarm = "bolt-decode-swarm" )
Standard scenario names. They are the kebab-case keys the CLI and the integration tests use to select a scenario from DefaultRegistry.
const ( // ArmCheckpointCadenceClean drives the cadence with no fault injected: the // ticker/MaxAge path, the explicit-trigger age reset, and the reclaimed-bytes // counter. ArmCheckpointCadenceClean = "cadence-clean" // ArmCheckpointCadenceTransientFault is the same plan with a one-shot fsync // fault armed on one periodic fire, so the failure lands in // [checkpoint.Stats.LastError] and the NEXT cadence fire has to recover from // it. ArmCheckpointCadenceTransientFault = "cadence-transient-fault" // ArmCheckpointCadenceIntervalOnly is the CONTROL: an interval with no // MaxAge. Ticks are delivered and the loop stays alive and responsive, and // nothing is ever folded from the cadence. ArmCheckpointCadenceIntervalOnly = "cadence-interval-only" )
The arm names, carried into the evidence so the non-vacuity gate can apply each arm's own floor.
const ( // ArmDBTeardownCancelledCtx is the cancelled-context arm: a single // CloseCtx whose context is cancelled before the call. ArmDBTeardownCancelledCtx = "cancelled-ctx" // ArmDBTeardownConcurrentClosers is the N-goroutine arm: many callers race // into the same teardown. ArmDBTeardownConcurrentClosers = "concurrent-closers" // ArmDBTeardownInFlightCommit is the boundary arm: the closers run while a // commit is parked inside its WAL fsync. ArmDBTeardownInFlightCommit = "in-flight-commit" )
The arm names, used in evidence and in the non-vacuity gate's arm-specific clause.
const ( // ScenarioDurableCommitCrash is ST2 (which subsumes ST1): concurrent durable // Bolt commits with a mid-flight WAL fsync fault, then a crash and recovery. ScenarioDurableCommitCrash = "durable-commit-crash" // ScenarioCheckpointTeardown is ST3: a background checkpointer racing // concurrent committers, torn down via the crash-safe store.DB shutdown // order. ScenarioCheckpointTeardown = "checkpoint-teardown" // ScenarioReadTxIsolation is ST7: read-only transactions under concurrent // writers and a mid-read crash of other actors. ScenarioReadTxIsolation = "readtx-isolation" )
Standard names for the concurrent-durable scenarios (sprint 270).
const ( // DeclDBTeardownFaultOnClose is the db-teardown arm that arms a one-shot // fsync fault on the WAL close's own fsync (step 3 of the composed teardown). DeclDBTeardownFaultOnClose = "db-teardown[fault-on-close]" // DeclCheckpointCadenceTransientFault is the cadence arm whose one periodic // fire fails inside its snapshot publish and whose next fire must recover. DeclCheckpointCadenceTransientFault = "checkpoint-cadence[transient-fault]" )
Arm keys for the declarations whose scenario has more than one fault shape.
const ( // SchemaIndexHash is the hash index kind as reported by SHOW INDEXES and // db.indexes(). It is also the default kind of CREATE INDEX without OPTIONS. SchemaIndexHash = "hash" // SchemaIndexBTree is the btree index kind. SchemaIndexBTree = "btree" // SchemaConstraintUnique is the UNIQUE constraint kind as reported by // SHOW CONSTRAINTS and db.constraints(). SchemaConstraintUnique = "UNIQUE" // SchemaConstraintNotNull is the NOT NULL (existence) constraint kind. SchemaConstraintNotNull = "NOT_NULL" )
Schema-model kind vocabulary, matching the engine's introspection values.
const ( // ScenarioCSRFilePublishFault is ST4: the csrfile atomic tmp->fsync->rename-> // parent-fsync publish under an ENOSPC bound and an armed Sync fault. ScenarioCSRFilePublishFault = "csrfile-publish-fault" // ScenarioWALCorruptionFailStop is ST5: genuine corruption of an interior WAL // frame, then recovery fail-stop. ScenarioWALCorruptionFailStop = "wal-corruption-failstop" // ScenarioCheckpointDirFsyncFault is ST6: a checkpoint WAL-prefix truncation // whose post-rename parent-dir fsync faults, poisoning the writer. ScenarioCheckpointDirFsyncFault = "checkpoint-dirfsync-fault" // ScenarioIORoundTripFault is ST8: a graph/io export/import round-trip under a // clean run and under an ENOSPC export fault. ScenarioIORoundTripFault = "io-roundtrip-fault" )
Standard names for the mechanical storage-fault scenarios (sprint 270).
const LegacyFullSnapshotDir = XReleaseFixtureDir + "/legacy_full_snapshot"
LegacyFullSnapshotDir is the store-root of the frozen PRE-rmp-#2520, PRE-rmp-#2526 snapshot fixture: a complete manifest-v3 snapshot directory (csr.bin, labels.bin, properties.bin, mapper.bin) whose manifest.json carries NO integrity trailer and NO `integrity` key, and whose csr.bin carries the dense 8-byte-wide float64 weights column.
It is a store root, not the snapshot directory itself: the snapshot lives at LegacyFullSnapshotDir/snapshot, which is where store/recovery.Open looks, and there is deliberately NO WAL beside it. A recovery over this directory can therefore only have obtained the graph from the snapshot bytes.
The bytes are FROZEN. They are pinned by digest in the compatibility test, not by a golden-file helper, because a golden can be rewritten by `-update` — and a fixture silently regenerated in the CURRENT format would keep every assertion passing while testing nothing at all. That is the failure mode rmp #2520 avoided by leaving its own frozen fixture unframed.
const ScenarioBulkImportParity = "bulkimport-parity"
ScenarioBulkImportParity is the catalogue key of the bulk-import publication parity scenario.
const ScenarioBulkLoadOracle = "bulk-load-oracle"
ScenarioBulkLoadOracle is the catalogue key of the bulk-loader full-contract scenario.
const ScenarioCheckpointCrashStorm = "checkpoint-crash-storm"
ScenarioCheckpointCrashStorm is the catalogue key of the crash-during- checkpoint-publish scenario.
const ScenarioConstraintExistence = "constraint-existence"
ScenarioConstraintExistence is the registry name of the existence scenario.
const ScenarioCountStore = "count-store"
ScenarioCountStore is the catalogue key of the count-store oracle scenario (rmp #2494).
const ScenarioDDLCheckpointCrash = "ddl-checkpoint-crash"
ScenarioDDLCheckpointCrash is the catalogue key of the DDL-across-the- checkpoint-boundary scenario.
const ScenarioFluentQuery = "fluent-query"
ScenarioFluentQuery is the catalogue key of the fluent-engine differential scenario (rmp #2492).
const ScenarioGenerationSwap = "generation-swap"
ScenarioGenerationSwap is the catalogue key for the generation-swap scenario.
const ScenarioLabelIndexScoped = "label-index-scoped"
ScenarioLabelIndexScoped is the catalogue key for this scenario.
const ScenarioPageRankRanker = "pagerank-ranker"
ScenarioPageRankRanker is the catalogue key for the stateful-PageRanker scenario.
const ScenarioProductionProfile = "production-profile"
ScenarioProductionProfile is the catalogue key of the production-profile scenario.
const ScenarioSnapshotCorruptionFailStop = "snapshot-corruption-failstop"
ScenarioSnapshotCorruptionFailStop is the catalogue key of the snapshot component corruption battery.
const ScenarioTypedSchema = "typed-schema"
ScenarioTypedSchema is the catalogue key of the typed-schema validator scenario (rmp #2493).
const SentinelCSRWidth uint8 = 0xFF
SentinelCSRWidth is the weightSizeBytes value rmp #2526 introduced to select the variable-width, codec-encoded weights section. It is restated here rather than imported because store/snapshot keeps it unexported; the value is part of the ON-DISK contract, so a copy in the cross-release model is the same kind of restatement the wire protocol next door already makes.
const XReleaseFixtureDir = "testdata/xrelease"
XReleaseFixtureDir is the root of the frozen cross-release on-disk fixtures, relative to this package's directory.
Variables ¶
var ErrCorruptImageCheckAborted = errors.New("sim: corrupt-image: aborted before a verdict was reached")
ErrCorruptImageCheckAborted reports that CheckCorruptImageRejected could not reach a verdict because its context was cancelled or its deadline expired. It is NOT a durability failure and must never be read as one: the check simply did not finish.
It exists because the two outcomes were previously indistinguishable — both arrived as a bare error — so a cancelled run was reported to the operator as a fail-stop violation (rmp #2745).
var ErrCrashedDisk = fmt.Errorf("sim: file handle was opened before a crash and did not survive it: %w", fs.ErrClosed)
ErrCrashedDisk is returned by every SimFileHandle method whose handle was opened before a crash. It WRAPS fs.ErrClosed, so code that already treats a closed handle as terminal keeps working unchanged, while errors.Is against this sentinel still tells a scenario author which of the two happened.
It exists because the alternative was silent (rmp #2544). SimDisk.CrashHost drops entries from the name maps, but a handle holds its *simFile DIRECTLY, so a scenario that crashed while holding its own handle went on writing into an ORPHANED file. The bytes went nowhere, no error was returned, and the symptom presented as data missing from the engine — a false accusation pointing at the subject under test. Fail-stop beats silent discard: the author learns the handle is dead instead of debugging phantom data loss.
var ErrShortCSRHeader = errors.New("sim: cross-release: csr.bin shorter than its header")
ErrShortCSRHeader is returned by ReadCSRHeaderBytes when the input is too short to contain a csr.bin header.
var ErrSimConnClosed = errors.New("sim: SimConn is closed")
ErrSimConnClosed is returned by SimConn I/O after the connection is closed.
var ErrSimFault = errors.New("sim: injected disk fault")
ErrSimFault is the sentinel returned by a simulated file operation that the seed-driven fault injector chose to fail. Callers match it with errors.Is. It models a durability fault: data the caller believed it was flushing did not reach stable storage.
var ErrSimListenerClosed = errors.New("sim: SimListener is closed")
ErrSimListenerClosed is returned by SimListener.Accept and SimListener.Dial after the listener is closed, mirroring net.ErrClosed semantics for the bolt/server accept loop (which treats a closed listener as a clean shutdown).
Functions ¶
func CheckCorruptImageRejected ¶
CheckCorruptImageRejected verifies the FAIL-STOP guarantee: a durable image whose committed WAL prefix has been corrupted must be REJECTED by the reopen path, never silently opened onto. It writes a small workload, closes, corrupts the durable WAL bytes inside an already-committed frame, and asserts the reopen refuses the image (the production recovery fail-stop contract, mirrored by OpenSimStore).
What makes the verdict a verdict ¶
Three things, each of which the check previously lacked:
- The refusal must be the RIGHT refusal. Accepting any non-nil error from the reopen made the oracle pass whenever the store failed to open for a reason having nothing to do with the corruption. The error must now satisfy [isWALCorruptionFailStop].
- Something must have been WRITTEN. A workload in which nothing committed leaves a WAL with no committed frame to corrupt, so the reopen has nothing to refuse and any verdict about it is vacuous. The commit count is now a precondition, not an assumption.
- A cancellation is not a verdict. It is reported as ErrCorruptImageCheckAborted, distinguishable by the caller from a durability failure.
It returns nil when the corruption was correctly rejected, and a descriptive error otherwise.
func DefaultVariantPair ¶
func DefaultVariantPair() (EngineVariant, EngineVariant)
DefaultVariantPair returns the PRIMARY differential pair: the engine's default configuration versus the same engine with the disconnected-equi-join hash-join optimisation turned OFF. The engine documents DisableHashJoin as existing "for the differential test that proves both plans return an identical result multiset" — so the two MUST agree on every observable output. This is a real, in-process, equivalent-result toggle, not a contrived comparison.
func ParallelAggregateVariantPair ¶ added in v0.10.0
func ParallelAggregateVariantPair() (EngineVariant, EngineVariant)
ParallelAggregateVariantPair returns the PARALLEL-AGGREGATION differential pair: the morsel-parallel aggregate scan (#2111 — min / max / count and their GROUP BY forms) versus the serial EagerAggregation pipeline, proving they produce a BIT-IDENTICAL observable result over the ParallelAggregateTrace. Variant A lowers ParallelScanThreshold to 1 so the parallel aggregate reduce engages on the small trace graph; variant B forces the serial path with DisableParallelScan.
Scope of what this pair proves. The differential builds a SEPARATE graph per variant, and the LPG mapper's WalkNodeIDs order is NOT stable across independent builds (it is unspecified by openCypher — two builds from the same CREATE stream legitimately scan in different orders). min/max over a Compare-TIE (e.g. int 2^53 vs float 2^53) keeps the first-seen member, so its retained REPRESENTATION is scan-order-dependent and therefore differs between two separately-built graphs even on the SERIAL path alone. The two-graph differential thus cannot compare a tie representative — which is precisely why ParallelScanVariantPair tests only the order-independent count. Accordingly ParallelAggregateTrace uses UNIQUE extrema (no ties), so min/max/count and their grouped forms are order-independent and the two variants must agree exactly. The adversarial TIE-representative determinism (mixed int/float, ±0.0, NaN) is proven where it can be — on ONE graph with a controlled scan order — by the exec-level TestParallelAggregateScan_TieRepresentative and the same-graph engine-level TestParallelAggregate_ScalarTie_Differential / _GroupBy_Differential.
func ParallelScanVariantPair ¶ added in v0.6.0
func ParallelScanVariantPair() (EngineVariant, EngineVariant)
ParallelScanVariantPair returns the PARALLELISM differential pair: the morsel-parallel count fast path (#1672) versus the serial EagerAggregation pipeline, proving they produce a BIT-IDENTICAL observable result. Variant A lowers ParallelScanThreshold to 1 so the parallel count reduce engages on the small trace graph (it is gated on a 50k-node default that a scripted trace never reaches); variant B forces the serial path with DisableParallelScan. The post-op count(n)/count(r) signatures stepSignature computes therefore come from the parallel reduce on A and the serial path on B, so any divergence in the parallel reduce surfaces here. The int64 partial-sum reduce is associative and partition-invariant, so the two MUST agree exactly. This brings the engine's multithread/parallel count path under the DST differential.
func RangeSeekVariantPair ¶
func RangeSeekVariantPair() (EngineVariant, EngineVariant)
RangeSeekVariantPair returns a second PRIMARY pair: the default configuration versus the same engine with the range-predicate B+tree index seek turned OFF. Like DisableHashJoin, DisableRangeIndexSeek exists for the differential proof that both plans return an identical result multiset.
func ReadFixtureFile ¶ added in v0.12.0
ReadFixtureFile reads one file out of the frozen fixture tree and returns its bytes. The path is joined under this package's directory, so it resolves the same way `go test` resolves testdata.
func RecordTrace ¶
RecordTrace runs a deterministic engine-API simulation under cfg and captures the full ordered op stream (plus any crash ticks) into a Trace, alongside the run's report (nil when the run passed). Recording adds no nondeterminism: it only observes the op stream the simulator already produces from the seed, so the recorded run behaves identically to an unrecorded one and the returned Trace replays (via ReplayTrace) to the same end-state.
cfg.OnOp and cfg.OnCrash are overridden by the recorder; any caller-supplied hooks are chained AFTER the recording hook so verbose tracing still works.
func ReplayInstructions ¶
ReplayInstructions renders a human-readable, copy-pasteable description of a (possibly shrunk) trace: the seed it came from and the ordered op list, so a failure can be reproduced and inspected. It is included in the SimReport for a shrunk reproducer.
func RunCountStore ¶ added in v0.12.0
func RunCountStore( ctx context.Context, cfg CountStoreConfig, ) (*CountStoreEvidence, *SimReport, error)
RunCountStore drives the scenario once and returns the evidence it measured alongside the report of the first violation (nil when the run is clean).
It owns and closes the simulator, so no durable handle or goroutine leaks past the run. The evidence is returned even on a violation, because what the run managed to exercise before failing is part of the diagnosis.
func RunFluentQuery ¶ added in v0.12.0
func RunFluentQuery(ctx context.Context, cfg FluentQueryConfig) (*FluentQueryEvidence, *SimReport, error)
RunFluentQuery drives the scenario once and returns the evidence it measured alongside the report of the first violation (nil when the run is clean).
It owns and closes the simulator, so no durable handle or goroutine leaks past the run. The evidence is returned even on a violation, because what the run managed to exercise before failing is part of the diagnosis.
func RunLabelIndexScoped ¶ added in v0.12.0
func RunLabelIndexScoped( ctx context.Context, cfg LabelIndexScopedConfig, ) (*LabelIndexScopedEvidence, *SimReport, error)
RunLabelIndexScoped drives one whole run: the relationship sweep against the naive model, the Union arm, the serialize/deserialize round trip, the three damage families on a SimDisk, the scope-routing table, and the two boundary pins.
It returns the evidence in every case, a report when a clause or a gate fired, and an error only for a harness failure — which here means an entry point returned an error the arm did not construct, or the SimDisk refused a fixture step.
The whole run is a pure function of the seed. There is no clock, no goroutine, no process-global counter and no allocation measurement anywhere in it, so unlike most scenarios in this package nothing has to be excluded from the digest, and the determinism claim is an equality on the whole of it.
func RunMVCCSubstrateAborts ¶ added in v0.12.0
func RunMVCCSubstrateAborts(ctx context.Context, cfg MVCCContentionConfig) (*MVCCSubstrateResult, *MVCCContentionResult, error)
RunMVCCSubstrateAborts is the ABORT-HEAVY arm: it drives the contended lost-update scenario, which produces typed serialization conflicts, and reads the substrate at the drain point to assert the refused transactions' versions were WITHDRAWN rather than left to accumulate (rmp #2470, criterion 2).
Why conflicts and not rollbacks ¶
mvcc.WriteCounts.Aborts counts transactions the SUBSTRATE refused, not transactions a client rolled back. GoGraph's explicit rollback is served by the statement undo log, so a voluntary rollback publishes its inverses and is counted as a commit — measured, 50 rollbacks produced `commits +49, aborts 0`. Only a serialization conflict reaches the aborted-version path, which is why this arm drives contention rather than rollback.
What "withdrawn" means here ¶
Withdrawal is SYNCHRONOUS: [lpg.Graph.abortWake] calls `withdrawAbortedNow` before returning, because a present-time read takes the stored value directly and the aborted transaction's writes must be out of it by then. So the assertion is not "the vacuum eventually freed them" but the stronger "they are not in the live count at the drain point at all" — which is why this arm can adjudicate without waiting on a background sweep, and why it does not need to cross the vacuum's wake threshold to be meaningful.
func RunPageRankRanker ¶ added in v0.12.0
func RunPageRankRanker(ctx context.Context, cfg PageRankRankerConfig) (*PageRankRankerEvidence, *SimReport, error)
RunPageRankRanker drives one whole run of the scenario: build the fixture, drive the interleaved sequence over ONE PageRanker, drive the cross-regime arm, compare the reference window against the independent power iteration, and adjudicate.
It returns the evidence in every case, a report when a clause or a gate fired, and an error only for a harness failure — which here means a foreign GOMAXPROCS clamp landed inside a window, or an entry point returned an unexpected error.
func RunSwarmWithMetricsOracle ¶
func RunSwarmWithMetricsOracle(ctx context.Context, sw *Swarm, goroutineSlack int) (SwarmResult, MetricsOracleResult, error)
RunSwarmWithMetricsOracle runs a swarm bracketed by a metrics oracle and asserts the reliability bound: after the (concurrent) swarm spawns and joins every worker, the live goroutine count must return to its baseline (within slack). It is the swarm-level wiring of the metrics oracle. Because the metrics sink is global it must run serially with respect to other metrics-emitting work; it restores the no-op backend before returning.
The returned SwarmResult is the swarm's own aggregate; the MetricsOracleResult carries the goroutine-baseline verdict.
func RunTypedSchema ¶ added in v0.12.0
func RunTypedSchema( ctx context.Context, cfg TypedSchemaConfig, ) (*TypedSchemaEvidence, *SimReport, error)
RunTypedSchema drives the scenario once and returns the evidence it measured alongside the report of the first violation (nil when the run is clean).
It owns and closes the simulator, so no durable handle or goroutine leaks past the run. The evidence is returned even on a violation, because what the run managed to exercise before failing is part of the diagnosis.
func SimEngineForServer ¶
SimEngineForServer builds a fresh in-memory directed-multigraph engine with a finite result-row cap, suitable for backing a SimServer. The multigraph model matches openCypher's additive-CREATE relationship semantics that the Bolt e2e path expects.
Types ¶
type AbuseFamily ¶
type AbuseFamily int
AbuseFamily identifies one class of Bolt wire-protocol violation the BoltAbuser emits. The set is fixed so a unit test can assert every family is reachable and exercised.
const ( // AbuseBadHandshake sends an invalid version-handshake preamble (wrong magic). AbuseBadHandshake AbuseFamily = iota // AbuseNoCommonVersion offers only versions the server does not support, so // negotiation must fail (server responds 0.0 and closes). AbuseNoCommonVersion // AbuseTruncatedChunk sends a chunk header advertising more bytes than follow, // then closes — a partial/truncated message. AbuseTruncatedChunk // AbuseOversizedChunk sends a single well-framed message whose total size // exceeds the server's MaxMessageBytes cap (but is bounded by the harness). AbuseOversizedChunk // AbusePullBeforeRun sends PULL immediately after auth, with no preceding RUN // (wrong session state). AbusePullBeforeRun // AbuseRunBeforeLogon sends RUN before authenticating (wrong session state on // a deferred-auth version, or before HELLO on inline-auth). AbuseRunBeforeLogon // AbuseGarbageOpcode sends a correctly-framed message carrying an unknown // struct tag (a garbage opcode). AbuseGarbageOpcode // AbuseDuplicateHello sends two HELLOs back to back (a duplicate/interleaved // marker for an already-progressed session). AbuseDuplicateHello // AbuseLogoffThenRun authenticates, sends LOGOFF, then sends a WRITE RUN on // the de-authorised connection. The CWE-306 gate must refuse it. AbuseLogoffThenRun // AbuseCommitAfterLogoff authenticates, opens an explicit transaction with a // write in it, sends LOGOFF (legal from TX_READY, and it leaves the session in // TX_READY but unauthenticated), then sends COMMIT. The transaction-finalising // gate must refuse it, so the write never becomes durable. AbuseCommitAfterLogoff // AbuseBadCredentials presents a WRONG password on the authenticating message // and then attempts a write. It is the one family that needs a server whose // AuthHandler actually validates credentials — see [AbuseFamily.NeedsCredentialAuth]. AbuseBadCredentials )
Bolt abuse families.
func (AbuseFamily) NeedsCredentialAuth ¶ added in v0.12.0
func (f AbuseFamily) NeedsCredentialAuth() bool
NeedsCredentialAuth reports whether driving this family proves anything only against a server whose AuthHandler rejects a wrong credential. Against NoAuthHandler, AbuseBadCredentials is admitted — correctly, since that handler admits everything — so a battery that ran it there would be asserting the absence of a check nobody installed.
func (AbuseFamily) String ¶
func (f AbuseFamily) String() string
String renders an AbuseFamily for reports.
type AbuseOutcome ¶
type AbuseOutcome struct {
FailureMsg string // populated when GotFailure
Family AbuseFamily
GotFailure bool // server replied with a typed FAILURE
GotClose bool // server closed the connection cleanly (or it became unreadable)
}
AbuseOutcome records how the server responded to one abuse attempt. Exactly one of GotFailure or GotClose is the expected acceptable result; a third outcome (a normal SUCCESS where a violation was sent, or a hang) is a defect the checker flags. The Family and the seed that chose it are retained so a finding is reproducible.
func (AbuseOutcome) Acceptable ¶
func (o AbuseOutcome) Acceptable() bool
Acceptable reports whether the outcome is one the robustness contract allows: a typed FAILURE or a clean connection close. Anything else (no terminal response, or an unexpected SUCCESS) is a violation.
type Actor ¶
type Actor interface {
// Name returns a stable identifier for the actor (used in reports).
Name() string
// NextOp returns the next operation to execute, drawing all randomness from
// seed and reading current contents from oracle.
NextOp(seed *Seed, oracle *GraphOracle) Op
}
Actor produces operations for the workload. An actor is stateless beyond the arguments to Actor.NextOp; all randomness comes from the supplied Seed and all knowledge of current graph contents from the supplied GraphOracle, so the operation an actor emits is a pure function of (seed state, oracle state).
Concurrency contract ¶
Actors are NOT safe for concurrent use; they are invoked from the single simulation goroutine.
type BoltAbuser ¶
type BoltAbuser struct{}
BoltAbuser emits protocol-level wire abuse over a SimConn and classifies the server's response. Each abuse runs on its own fresh connection in LOCK-STEP (send the violation, then block reading the terminal response or observing the close), so a given seed reproduces the exact violation and the exact server reaction. The server must respond with a typed FAILURE or close the connection cleanly — never panic, never leak a goroutine, never corrupt state.
Concurrency contract ¶
BoltAbuser is stateless and its BoltAbuser.Abuse method may be called from any goroutine, but each call drives one connection it owns end-to-end.
func (BoltAbuser) Abuse ¶
func (a BoltAbuser) Abuse(srv *SimServer, family AbuseFamily) (AbuseOutcome, error)
Abuse opens a fresh connection to srv, emits the chosen abuse family over the wire, and returns the classified outcome. The connection is always closed before return, so no goroutine or handle leaks regardless of how the server reacted. An error is returned only for a harness-level failure (e.g. the listener is closed), never for an expected server FAILURE/close.
func (BoltAbuser) PickFamily ¶
func (BoltAbuser) PickFamily(seed *Seed) AbuseFamily
PickFamily chooses an abuse family from the seed. It draws exactly one int so the workload draw stream is stable. Only the families every server refuses are drawn (see AbuseFamily.NeedsCredentialAuth): the random abuser runs against the NoAuth SimServer the bad-actors scenario builds, where a wrong-credential attempt is legitimately admitted and would be scored as a violation of nothing.
type BoltAuthArm ¶ added in v0.12.0
type BoltAuthArm struct {
// Name identifies the arm in a violation message.
Name string
// WantCode is the failure code a REFUSE arm must be told. It is empty for an
// ADMIT arm.
WantCode string
// GotCode is the failure code actually received ("" when none was).
GotCode string
// Detail carries any harness-level note (e.g. the connection closed before a
// reply, which is an acceptable refusal shape for a terminated session).
Detail string
// GotMessage is the FAILURE's message text, and WantStateInMessage the session
// state the refusal must name.
//
// They exist because the authentication gate and the state-machine gate return
// the SAME code — failTransition's Neo.ClientError.Request.Invalid — so a code
// match cannot tell which one refused. That matters: if LOGOFF's target state
// regressed (state.go's TX_READY/READY cases), commit-after-logoff would be
// refused by the STATE check one line above the auth check, the arm would still
// see Request.Invalid, and the CWE-306 gate would be untested while the
// scenario stayed green. failTransition reports the ORIGIN state in its
// message, which is exactly the discriminator: a refusal by the auth gate names
// the LEGAL state the session was in (READY / TX_READY), a refusal by the state
// gate names FAILED or the wrong state. An empty WantStateInMessage waives the
// check for arms whose refusal is not a failTransition (a credential rejection).
GotMessage string
WantStateInMessage string
// FramesBefore/FramesAfter bracket the arm with [wal.Stats.Frames].
FramesBefore, FramesAfter uint64
// BytesBefore/BytesAfter bracket the arm with [wal.Stats.Bytes].
BytesBefore, BytesAfter uint64
// Admit records the arm's INTENT: true when the server is expected to accept
// the operation and append to the WAL, false when it must refuse it and
// append nothing.
Admit bool
// Accepted records what actually happened: the server answered SUCCESS to the
// arm's decisive message.
Accepted bool
}
BoltAuthArm is one probe of the authentication surface: what it did, whether the server admitted it, the failure code it was told, and the WAL counters bracketing it.
type BoltAuthEvidence ¶ added in v0.12.0
type BoltAuthEvidence struct {
// Arms are the probes in the order they ran.
Arms []BoltAuthArm
// GhostNodes is how many nodes carrying [abuseGhostLabel] exist at the end:
// the count of writes that a de-authorised or unauthenticated session managed
// to apply. It must be zero.
GhostNodes int
// HonestNodes is how many nodes carrying [authHonestLabel] exist at the end:
// the writes the properly authenticated arms committed. It must equal the
// number of ADMIT arms, which is what proves the harness could write at all.
HonestNodes int
// RecoveredGhosts / RecoveredHonest are the same two censuses taken after a
// crash and a real WAL replay, so a frame that was appended but not yet
// visible in the live engine cannot hide from the oracle.
RecoveredGhosts int
RecoveredHonest int
// Seed is the seed the run was built from.
Seed uint64
}
BoltAuthEvidence is everything one RunBoltAuthSurface observed. It is a plain value so the checkers are pure functions of it — which is what lets a test perturb one field and prove the corresponding check fires.
func RunBoltAuthSurface ¶ added in v0.12.0
func RunBoltAuthSurface(ctx context.Context, seed uint64) (BoltAuthEvidence, error)
RunBoltAuthSurface drives the whole authentication surface once against a WAL-backed server whose AuthHandler validates credentials, and returns the evidence. It is bit-reproducible from seed: every arm is a fixed lock-step script on its own connection, and the only seeded component is the SimDisk the WAL lives on.
The returned error is reserved for harness failures (the store would not open, the listener refused a dial); a refused credential or a rejected statement is EVIDENCE, not an error.
func (BoltAuthEvidence) String ¶ added in v0.12.0
func (e BoltAuthEvidence) String() string
String renders the evidence for a report: one line per arm with its verdict and WAL delta, then the censuses.
type BoltBeginExtrasEvidence ¶ added in v0.12.0
type BoltBeginExtrasEvidence struct {
// Bookmarks are the causal-read arms, in the order they ran (a seed-drawn
// permutation).
Bookmarks []BoltBookmarkArm
// Timeouts, Modes and DBs are the other arm families, likewise in run order.
Timeouts []BoltTimeoutArm
Modes []BoltModeArm
DBs []BoltDBArm
// Route and Metadata are single observations rather than families.
Route BoltRouteObs
Metadata BoltMetadataObs
// CommittedNodes is how many nodes the writer committed under
// [beginCausalLabel]: the seed-drawn figure every causal read must observe. It is
// the harness's own count, never read back from the engine, so it is the
// independent side of the causality comparison.
CommittedNodes int
// IssuedBookmarks are the bookmarks the writer's successive COMMITs returned, in
// order. Their literal text is NOT reachable from the seed (see the file header),
// so clauses are written over relations between them, and the rendering shows
// them positionally.
IssuedBookmarks []string
// AutocommitBookmarkFresh is the `bookmark` in the terminal PULL SUCCESS of an
// autocommit WRITE on a connection that has never committed an explicit
// transaction. Measured EMPTY.
AutocommitBookmarkFresh string
// AutocommitBookmarkAfterCommit is the same field on a connection that HAS, and
// AutocommitAfterCommitExpected is the bookmark that connection's COMMIT had
// returned. The clause is that the two are EQUAL: the autocommit statement was
// handed the earlier transaction's token rather than one minted for itself.
AutocommitBookmarkAfterCommit string
AutocommitAfterCommitExpected string
// AutocommitStatsSeen records that the autocommit write's own terminal SUCCESS
// reported it as an update, so the empty bookmark above cannot be explained by
// the statement having written nothing.
AutocommitStatsSeen bool
// Seed is the seed the run was built from.
Seed uint64
}
BoltBeginExtrasEvidence is everything one RunBoltBeginExtras observed. It is a plain value carrying observations only, so every checker is a pure function of it — which is what lets a test perturb one field and prove the corresponding clause fires.
func RunBoltBeginExtras ¶ added in v0.12.0
func RunBoltBeginExtras(ctx context.Context, seed uint64) (*BoltBeginExtrasEvidence, error)
RunBoltBeginExtras drives the whole BEGIN extras surface once and returns the evidence.
It is bit-reproducible from seed. The seed draws three things and nothing else: how many nodes the writer commits, the order the causal-read arms run in, and the order the mode and database arms run in. Every other quantity is either a fixed script or a virtual duration; no arm depends on wall time for its outcome, only for its bound.
The bookmark, mode, database, ROUTE and metadata families share ONE server, because the causal read is only meaningful against the store the writer committed to. The timeout family builds a server per arm: each needs its own bounds and its own fake clock, and a fake clock cannot be installed on a server that is already serving (see SimServer.Server).
func (*BoltBeginExtrasEvidence) String ¶ added in v0.12.0
func (e *BoltBeginExtrasEvidence) String() string
String renders the evidence for a report. Two runs of the same seed must produce BYTE-IDENTICAL output, which is asserted directly, so the whole rendering is swept for anything not reachable from the seed rather than each such field being handled as it is noticed (rmp #2483's lesson). Three classes are excluded:
- REAL-TIME durations. BoltBookmarkArm.BeginElapsed and BoltTimeoutArm.ReplyElapsed are the instruments for "does not wait" and "aborts rather than stalls"; they belong in a clause with a generous bound, never in a byte comparison.
- PROCESS-GLOBAL bookmark text. bookmarkCounter is global to the process (bolt/server/bookmark.go:13), so an issued bookmark's literal value depends on how many transactions every other test committed first — and so does the ADVANCE between two of them, which is why renderIssued shows neither. Issued bookmarks are rendered POSITIONALLY, and the stale-autocommit reading as the RELATION it is adjudicated on rather than as its text.
- Harness error TEXT. BoltBookmarkArm.ReadFailed can carry a transport message whose wording is not seed-determined, so its PRESENCE is rendered and its text is left to the violation that reports it.
type BoltBookmarkArm ¶ added in v0.12.0
type BoltBookmarkArm struct {
// Name identifies the arm in a violation message and in the rendering.
Name string
// SentToken renders the single element the arm put in the `bookmarks` list, as
// the arm built it. It is "<issued>" for the writer's own bookmark — never the
// literal text, which is not reachable from the seed — and the verbatim constant
// for a fabricated one. Empty means the key was omitted entirely.
SentToken string
// SentKey records whether the `bookmarks` key was present at all, so an arm that
// omitted it is distinguishable from one that sent an empty list.
SentKey bool
// ServerSawTokens is how many tokens [server.ExtractBookmarks] would keep from
// what the arm sent, computed by the HARNESS from its own list rather than read
// back from the server (the server reports nothing). It is 0 for the wrong-type
// arm, which is the point of that arm.
ServerSawTokens int
// Accepted is true when BEGIN was answered SUCCESS.
Accepted bool
// GotCode / GotMessage are the FAILURE's code and message, empty when none came.
GotCode, GotMessage string
// Observed is how many nodes carrying [beginCausalLabel] the arm's causal read
// returned. It is compared ACROSS arms rather than against a literal, because
// what the pin is about is that the token changed nothing.
Observed int
// ReadFailed carries a harness-level note when the causal read could not be
// completed, so a broken read is never silently indistinguishable from a read
// that returned zero.
ReadFailed string
// BeginElapsed is how long the BEGIN took in REAL time. It is the instrument for
// "an unknown bookmark is not WAITED on", and it is deliberately absent from the
// rendering: it is scheduling-dependent (rmp #2483's lesson) and belongs in a
// clause, not in a byte-identity comparison.
BeginElapsed time.Duration
}
BoltBookmarkArm is one causal read: what the reader put in its BEGIN `bookmarks` list, whether the BEGIN was accepted, and how many of the writer's committed nodes it went on to observe.
It carries observations and no verdicts. Whether a fabricated token being accepted is correct is a question for [checkBoltBookmarkCausality], not for the collector.
type BoltCertRotationEvidence ¶ added in v0.12.0
type BoltCertRotationEvidence struct {
// Steps are the rotation attempts in the order they ran.
Steps []CertRotationStep
// InitialLoadTornErr is the error [server.NewCertReloader] returned when
// constructed over a TORN key, rendered. The initial load is documented as
// mandatory, so this must be non-empty: a reloader that starts on unparseable
// material would put a server into service with no certificate at all.
InitialLoadTornErr string
// WatchErrors are the errors the reloader's onError callback delivered while a
// [server.Watch] goroutine polled a BROKEN pair.
//
// onError is the only operator-visible signal that a BACKGROUND rotation
// failed, and before this scenario nothing in the module asserted it: deleting
// the callback from Watch would have broken no test. It is the same defect
// class as the Options.Logger bypass this sprint fixed — a security-relevant
// event with no reachable destination — so the signal is now evidence.
WatchErrors []string
// UnloadedGetErr is the error a zero-value CertReloader's GetCertificate
// returned, rendered. It must be non-empty rather than a nil certificate,
// which the TLS stack would dereference.
UnloadedGetErr string
// Seed is the seed the run was built from.
Seed uint64
}
BoltCertRotationEvidence is everything one RunBoltCertRotation observed.
func RunBoltCertRotation ¶ added in v0.12.0
func RunBoltCertRotation(ctx context.Context, seed uint64) (BoltCertRotationEvidence, error)
RunBoltCertRotation drives the certificate-rotation surface once and returns the evidence. It is bit-reproducible from seed: the key material, the serials and the torn-prefix length are all drawn from it in a fixed order.
The returned error is reserved for harness failures; a refused reload is EVIDENCE, not an error.
func (BoltCertRotationEvidence) String ¶ added in v0.12.0
func (e BoltCertRotationEvidence) String() string
String renders the evidence for a report: one line per step.
type BoltDBArm ¶ added in v0.12.0
type BoltDBArm struct {
// Name identifies the arm; SentDB is the name sent, SentKey whether it was
// present.
Name string
SentDB string
SentKey bool
// ReportedDB is the `db` field of the terminal PULL SUCCESS — the one a driver
// turns into ResultSummary.Database().Name() (rmp #2172).
ReportedDB string
// ReportedOnRun is the `db` field of the RUN SUCCESS, which carries it too
// (bolt/server/session.go handleRun). Recorded so the two are compared rather
// than one assumed to follow the other.
ReportedOnRun string
// CommitMetaKeys is every key the COMMIT SUCCESS carried, sorted. It is here to
// pin what COMMIT does NOT carry: `db` is absent from it, so a driver reading the
// database name off a commit reply gets nothing.
CommitMetaKeys []string
}
BoltDBArm is one probe of the `db` extra: the name sent on BEGIN and the name reported back in the terminal PULL SUCCESS.
type BoltDecodeEvidence ¶ added in v0.12.0
type BoltDecodeEvidence struct {
// Census counts are read over the wire (live) and straight off a store reopened
// through real recovery. Keyed by label; ids are never recorded, because a
// created node's hidden key is minted from a process-global counter.
LiveCensus map[string]int
RecoveredCensus map[string]int
// ControlCensus is the same census on the CONTROL server's own engine, where
// the bytes the pressured pool refused were served. Two engines, one payload:
// the count differing between them IS the arm's claim.
ControlCensus map[string]int
// AbuserReplies is the swarm's reply census: code (or "SUCCESS") to count. It
// is an invariant SUMMARY — which code, how many — and never says which
// connection drew what, because that is the scheduler's business.
AbuserReplies map[string]int
// TransportErrors names every abuser exchange that failed at the transport
// rather than drawing a Bolt reply. A reassembly-layer budget breach tears the
// connection down (see the floor arithmetic on [boltDecodeSwarmAbusers]), so a
// non-empty list here is the honest report of the swarm having been sized wrong
// — not a silent pass.
TransportErrors []string
// Window is the boundary scan, in ascending element count.
Window []BoltDecodeProbe
// Arms are the named deterministic arms, in the order they were driven.
Arms []BoltDecodeExchange
// Honest are the swarm's honest exchanges, in order.
Honest []BoltDecodeHonest
// Seed is the run's seed.
Seed uint64
// Budget is the pressured (or swarm) server's ceiling in bytes.
Budget int64
// ControlBudget is the control server's ceiling, 0 on the swarm arm.
ControlBudget int64
// ModelBoundary is the largest element count the harness's model says the pool
// admits. MeasuredBoundary is the largest the SERVER actually admitted inside
// the scanned window, or -1 when the window did not span a transition.
ModelBoundary int
MeasuredBoundary int
// LeakProbeInitial and LeakProbeFinal are the same boundary-sized message sent
// before any pressure and after all of it. The first CALIBRATES (a pristine
// pool must admit it); the second is the no-leak oracle, since a pool that
// failed to return any of what it lent would no longer admit it.
LeakProbeInitial BoltDecodeProbe
LeakProbeFinal BoltDecodeProbe
// AbuserAccepted / AbuserRejected are the swarm's totals.
AbuserAccepted int
AbuserRejected int
// AbuserAliveAfter is how many abuser connections still served an honest
// exchange once the swarm quiesced.
AbuserAliveAfter int
// Abusers is how many pushed.
Abusers int
// RefusalsConcurrentWithHonest is how many abusive messages were refused on a
// round trip that OVERLAPPED a honest exchange's own flight.
//
// It is the overlap instrument, and it is an interval intersection rather than
// an event-in-window test: the abuser samples the honest client's in-flight
// state immediately before it writes and again at the moment it counts the
// refusal, and both endpoints are published by the goroutine that owns them.
// See [boltDecodeSwarm.abuser] for the sampling and
// [checkBoltDecodeSwarmNonVacuity] for what the clause may and may not read
// into it.
RefusalsConcurrentWithHonest int64
// RejectionsSpanningHonestWindow is how many abusive messages were refused on a
// round trip that INTERSECTED the honest window — the interval from the first
// honest exchange starting to the last one finishing.
//
// It is the quantity nv-swarm-pressure-density adjudicates, and it is an
// interval intersection rather than a counter differenced across the window: the
// abuser samples the honest window's published phase immediately before it
// writes and again at the moment it counts the refusal. See
// [boltDecodeSwarm.honestPhase] for the three phases and
// [checkBoltDecodeSwarmNonVacuity] for why the counter difference it replaced
// could not carry the claim.
//
// No FLOOR is ever put on it beyond nonzero. It is a count of events over an
// interval whose length the machine controls, so any floor above that passes on
// an idle host and fails on a busy one — which is exactly what two earlier
// versions of nv-swarm-pressure-density did (rmp #2587).
//
// It cannot carry the overlap claim on its own: a fleet that took turns with the
// honest client, never once refusing on a round trip that overlapped honest
// work, would produce a large count here and witness nothing about concurrency.
// Overlap is adjudicated on RefusalsConcurrentWithHonest above, and this total is
// what gates it — a window that held no refusals at all is this clause's own
// subject, not that one's.
RefusalsSpanningHonestWindow int64
// RejectionsDuringHonest is how many abusive messages the fleet-wide counter
// ADVANCED BY between the honest client's own sample after the start barrier and
// its deferred sample once the last exchange had finished.
//
// DIAGNOSTIC ONLY — no clause reads it any more, and rmp #2611 is why. It was
// the density clause's instrument through two earlier versions, and a nonzero
// floor on it was described here as the one floor that was safe. It is not: the
// quantity is not a count of refusals drawn inside the window, it is a
// difference between two reads of a counter an abuser increments only after
// decoding its refusal reply. A refusal drawn on a round trip that straddles the
// window's close is therefore counted AFTER the window and missed entirely, and
// under load — where an abusive round trip outlasts the whole honest window —
// that case stops being rare. Measured on a coverage-instrumented sweep of 100
// seeds, 2 seeds per sweep read zero here while the fleet had drawn 3 refusals
// and the overlap instrument had attributed 2 of them to honest FLIGHT: the
// pressure was demonstrably live and this instrument could not see it.
//
// So this is the THIRD iteration of one defect. #2587 removed a numeric floor
// (eight refusals) and then a positional one (one per segment) after both failed
// on clean engines; what it left was a nonzero floor on this counter, which is
// scheduler-dependent for a different reason — not the RATE of refusals but
// WHERE the counter records them. The floor is gone with the instrument: the
// claim now rests on RefusalsSpanningHonestWindow above.
//
// It is still rendered, and [boltDecodeSwarmInFlightRefusals] plus
// [boltDecodeSwarmGapRefusals] is still exactly this number minus its tail, so
// the three together keep showing where the counter recorded the pressure
// relative to the honest client's own samples.
RejectionsDuringHonest int64
// WideHoldExpiries is how many WIDE honest exchanges reached
// [boltDecodeSwarmWideHoldMax] before the fleet had completed the round trips
// their hold was waiting for (see [boltDecodeSwarm.wideHold]).
//
// RENDERED and adjudicated nowhere, for the same reason as GateTimeouts below: a
// threshold on it would be another floor that machine load moves. Zero is the
// ordinary reading; a nonzero one says the fleet stalled, and the refusals that
// run drew inside honest flight were coincidences again.
WideHoldExpiries int64
// GateTimeouts is how many times an abuser gave up waiting for a partner at the
// fleet's rendezvous (see [boltDecodeSwarmGate]).
//
// It is RENDERED and adjudicated nowhere. Zero is the ordinary reading and every
// regime measured for rmp #2611 returned zero; a nonzero one says the fleet
// stopped pairing, so the refusals that run drew were coincidences again rather
// than consequences of the construction. Thresholding it would put back exactly
// the kind of load-moved floor this surface has now removed three times, so it
// is reported as data and read by a human.
GateTimeouts int64
// PressureStarted records whether the honest client saw a refusal counted
// before it began, i.e. whether the start barrier was satisfied rather than
// expiring. It is what tells a failing run apart: honest exchanges that did not
// straddle because there was no pressure, versus ones that did not straddle
// while there was.
PressureStarted bool
// Swarm marks the concurrent arm, whose adjudication differs.
Swarm bool
}
BoltDecodeEvidence is everything one run observed.
func RunBoltDecodePressure ¶ added in v0.12.0
func RunBoltDecodePressure(ctx context.Context, seed uint64) (*BoltDecodeEvidence, error)
RunBoltDecodePressure drives the whole deterministic inbound-decode surface once and returns the evidence.
The subject is a WAL-backed server whose engine-wide inbound-decode ceiling is [boltDecodePressuredBudget]; the CONTROL is a second server that differs from it in that ceiling and in nothing else, so a payload the first refuses and the second serves isolates the pool as the cause rather than the payload.
It is bit-reproducible from seed: every arm is a fixed lock-step script, the element counts are computed from the harness's model of the pool, and the only seeded draws are the excessive nesting depth, the two write arms' element counts and the SimDisk the WAL lives on. Nothing it records is a node id or a WAL byte total.
The returned error is reserved for harness failures (the store would not open, a dial was refused, a reply did not arrive inside [boltDecodeArmBound]). A refused message is EVIDENCE, not an error.
func RunBoltDecodeSwarm ¶ added in v0.12.0
func RunBoltDecodeSwarm(ctx context.Context, seed uint64) (*BoltDecodeEvidence, error)
RunBoltDecodeSwarm drives the AGGREGATE arm: K abusers push large-collection parameters at one shared pool while an honest client keeps working against the same server.
It is NOT bit-reproducible, and it cannot be: which abuser wins the pool on any given round is the scheduler's decision, and the whole point of the arm is that two charges are outstanding at once — a state a deterministic script cannot reach at all (see the file comment). The seed drives the honest client's interleave delay and nothing else. The oracle is what EVERY interleaving must share:
- every abuser reply is either SUCCESS or the typed retryable refusal, never a third code and never a dropped connection;
- at least one message was refused (the pressure was real) and at least one was served (the pool is not simply broken — a pool stuck at zero would satisfy "every refusal is typed" perfectly);
- every honest exchange returned the value the harness computed for it, and a refusal was drawn on an abusive round trip that overlapped one of them IN FLIGHT, so honest service demonstrably overlapped the pressure window rather than merely preceding or following it (see [boltDecodeSwarmInFlightRefusals] for the instrument that could NOT carry that claim, and why);
- the pool comes back: a message sized to the whole ceiling is admitted once the swarm quiesces.
Each abuser's message is sized at [boltDecodeSwarmChargeNum]/[boltDecodeSwarmChargeDen] of the pool, so one fits and two cannot. See the floor arithmetic on [boltDecodeSwarmAbusers] for why that sizing also keeps the honest client clear of the reassembly reader, whose own budget breach is NOT the connection-preserving refusal the decode layer's is.
func (*BoltDecodeEvidence) String ¶ added in v0.12.0
func (e *BoltDecodeEvidence) String() string
String renders the evidence.
Everything here is a function of the seed. Node ids never appear (a created node's hidden internal key is minted from a process-global counter, so an id is not seed-reachable — the trap rmp #2486's determinism test caught); the census is counts by label; the engine's internal-error text has its per-session id redacted before it is ever recorded; and the honest client's elapsed times, which are a property of the machine, are not rendered at all.
The SWARM rendering is deliberately a census and not a sequence: which abuser won the pool on a given round is the scheduler's business and would differ run to run. It is rendered only when the arm FAILS, and the determinism test covers the deterministic scenario alone.
type BoltDecodeExchange ¶ added in v0.12.0
type BoltDecodeExchange struct {
// Name is the arm's stable name; it is what a violation cites.
Name string
// Vehicle is the message kind the abuse rode in on: "run" or "hello". The
// HELLO arms matter on their own because the wire nesting cap is reachable
// during the FIRST HELLO decode, before any authentication.
Vehicle string
// Composite is the nesting arm's chain kind — "list" or "map" — or "" for the
// pool arms. Both kinds must hit the same cap: readValue's bound is on
// composite depth, not on lists.
Composite string
// Reply is the terminal reply kind: "SUCCESS", "FAILURE", "IGNORED", or
// "RECORD+SUCCESS" when records preceded it.
Reply string
// Code and Message are the FAILURE's, with the per-session id redacted
// ([boltDecodeRedactSession]) so the rendering stays a function of the seed.
Code string
Message string
// Depth is the nesting chain's length for a nesting arm, 0 otherwise.
Depth int
// Elements is the parameter list's element count for a pool arm, -1 otherwise.
Elements int
// WireBytes is the message's size on the wire. For the nesting family it is the
// evidence that SIZE cannot be what refused it.
WireBytes int
// ModelHeld is the harness's model of the pool hold, for pool arms only.
ModelHeld int64
// SessionUsable is whether the SAME connection served a follow-up exchange
// with NO RESET in between. It is the third discriminator between the caps:
// the two decode-layer refusals happen above the session state machine and
// leave it READY; the engine's parameter cap travels through it and FAILS it.
SessionUsable bool
// FollowUpValue is what the follow-up RUN returned, or -1 when it was not
// served. The harness picks the expected value, so a connection that answered
// with someone else's row is a failure and not a pass.
FollowUpValue int64
}
BoltDecodeExchange is one named arm: a message sent, the reply it drew, and whether the connection and the session survived it.
type BoltDecodeHonest ¶ added in v0.12.0
type BoltDecodeHonest struct {
// Index identifies the exchange and IS the value its query must return, so the
// oracle is arithmetic rather than the server's own claim.
Index int
// RejectionsBefore/RejectionsAfter sample the fleet-wide rejection counter
// immediately before the request is written and immediately after the reply is
// read. After > Before means abusive traffic was REFUSED, AND COUNTED, while
// this exchange was in flight.
//
// That is weaker than it looks and it is REPORTED, never adjudicated: an abuser
// increments the counter only after decoding its reply, so a refusal issued
// before this exchange opened can be counted inside it. See
// [boltDecodeSwarmInFlightRefusals] for the measurement that establishes how
// badly, and [BoltDecodeEvidence.RefusalsConcurrentWithHonest] for the
// instrument nv-swarm-overlap uses instead.
RejectionsBefore int64
RejectionsAfter int64
// Value is what the exchange returned, or -1 when it returned nothing.
Value int64
// Elapsed bounds liveness. It is deliberately NOT rendered: it is a property of
// the machine, not of the seed.
Elapsed time.Duration
// OK is whether the exchange completed and returned exactly one record.
OK bool
// Wide marks an exchange that deliberately held its stream open for
// [boltDecodeSwarmWideHold] between RUN and PULL. Its in-flight window is wide
// enough to hold several refusals when they are arriving quickly, and the run
// reports how many the narrowest one held; nothing thresholds it (rmp #2596).
Wide bool
}
BoltDecodeHonest is one honest exchange run against the server while the swarm pushes. It carries its own overlap evidence.
type BoltDecodeProbe ¶ added in v0.12.0
type BoltDecodeProbe struct {
// Code is the failure code when the pool refused, else "".
Code string
// Elements is the parameter list's element count, the only thing that varies
// across a scan.
Elements int
// ModelHeld is what [boltDecodeModelHeld] says this message's decode holds from
// the shared pool at its peak. It is computed by the harness, never read from
// the server.
ModelHeld int64
// Slack is Budget - ModelHeld: positive when the model says the message fits.
// It is what makes the leak probe's SENSITIVITY explicit.
Slack int64
// Accepted is whether the server decoded and ran the message.
Accepted bool
}
BoltDecodeProbe is one modelled message and the pool's verdict on it. It is the unit the boundary scan and the leak probes are made of.
type BoltDrainCommit ¶ added in v0.12.0
type BoltDrainCommit struct {
// Name is the node name the statement would create.
Name string
// Phase is "pre" (issued and answered before the teardown began), "inflight"
// (parked inside its WAL fsync when the teardown began) or "post" (issued after
// the teardown).
Phase string
// RunCode is the failure code the RUN was answered with, "" for a SUCCESS.
// RunAcked records an EXPLICIT [proto.Success], classified with [isSuccess]. For
// an auto-commit write the RUN reply is the durability acknowledgement (see
// [boltDrainAckIsRunSuccess]).
RunCode string
RunAcked bool
// RunIgnored records a [proto.Ignored] reply ([isIgnored]): the session was
// already FAILED, so the statement was never dispatched and nothing was written.
//
// It has its own field because reading the acknowledgement as "not a FAILURE"
// counts an IGNORED as a durable commit. That is not hypothetical: it is the
// defect this file's FIRST oracle had, and it manufactured an ACID_DURABILITY
// violation in 8 of 30 concurrent runs. The mechanism, measured: the shutdown
// answered a committer's RUN with Neo.ClientError.Transaction.Terminated, which
// puts the session in FAILED; the committer's NEXT RUN was answered IGNORED; the
// harness recorded an acknowledgement; and the name it then demanded from
// recovery had never been written at all — confirmed by finding it neither in the
// live engine nor anywhere in the raw WAL image. bolt_auth_surface.go (rmp #2481)
// added [isIgnored] against exactly this class of false positive and noted that
// its own arms did not reach the branch; these arms do.
RunIgnored bool
// PullAcked records whether the client also drained the result stream. It is a
// WITNESS, not a clause: a graceful Shutdown closes the connection as soon as
// the in-flight statement's reply is flushed, so an acknowledged in-flight
// commit routinely never gets its PULL through (measured; see the file
// comment).
PullAcked bool
// PullCode is the failure code the PULL terminal carried, "" when there was
// none, and PullIgnored records an IGNORED terminal.
PullCode string
PullIgnored bool
// Transport records that the round trip failed below the protocol (the
// connection was closed under the client). Rendered as a boolean in the
// evidence's String because the underlying text names a connection object.
Transport bool
}
BoltDrainCommit is one write attempt over the wire and what the client was told. Phase is what dates it against the teardown, which is the whole basis of the wire clause: a statement issued BEFORE the teardown must not be answered with a storage failure, while one issued after it legitimately is.
type BoltDrainConfig ¶ added in v0.12.0
type BoltDrainConfig struct {
// Seed is the master seed. Every node name and the SimDisk sub-stream derive
// from it.
Seed uint64
// Arm names the variant (see the ArmBoltDrain* constants).
Arm string
// ParkCommit arms a [SyncGate] on the next fsync and holds one commit inside
// it, so the teardown genuinely races an in-flight writer.
ParkCommit bool
// ShutdownBudget bounds the first Shutdown call. Zero passes
// [context.Background], so Shutdown drains to completion.
ShutdownBudget time.Duration
// ExpiryByCancel bounds Shutdown with an explicit cancel on a context that has
// NO deadline, instead of with a deadline. It is the only way to reach
// Shutdown's `<-ctx.Done()` branch: with a deadline, Shutdown's own clamped
// time.After wins (measured 12/12; see the file comment).
ExpiryByCancel bool
// ShutdownCalls is how many times Shutdown is called (minimum 1). Two is what
// makes the published-identity clause a statement about the server's sync.Once.
ShutdownCalls int
// FailClose arms a one-shot fsync fault on the WAL close, so the store's
// teardown FAILS and the value the server caches is non-nil.
FailClose bool
// CloseOutOfOrder closes the store DIRECTLY, with a live client mid-session and
// no drain at all. It is the sensitivity seam: it reproduces what the ordering
// exists to prevent, so the ordering and wire clauses are proved to fire.
CloseOutOfOrder bool
// Fleet drives several concurrent committers instead of the deterministic
// single-commit geometry. NOT bit-reproducible.
Fleet bool
// SkipCheckpointer stands the store up with NO checkpointer at all: no loop, no
// published snapshot, and recovery through WAL replay alone. It is the A/B seam
// for attributing a lost commit to the checkpoint interaction rather than to the
// drain, and it is deliberately a configuration rather than a separate code path
// so the two sides differ in exactly one thing.
//
// It is not a shipped arm: with no snapshot and no loop to join, the coverage gate
// fires on nonvacuity-snapshot and nonvacuity-loop by design.
SkipCheckpointer bool
}
BoltDrainConfig parameterises one arm. The zero value is not usable: RunBoltShutdownDrain normalises Arm and derives the rest of the geometry from it, and the flags below are what a control varies.
type BoltDrainEvidence ¶ added in v0.12.0
type BoltDrainEvidence struct {
// Arm is the variant that produced this evidence; Seed is the seed it was
// built from.
Arm string
Seed uint64
// CloserWired records that the server was actually given an
// [server.Options.Closer]. Without it every ordering clause below is inert.
CloserWired bool
// ConnDecorated records that the connection tracker was installed, which is
// what the ordering clause reads.
ConnDecorated bool
// CloseBodies is every invocation of the store's teardown, in call order. The
// contract is exactly one.
CloseBodies []boltDrainCloseBody
// CloseBodiesWhileParked is how many teardown bodies had been entered while a
// commit was still parked inside its WAL fsync AND Shutdown had demonstrably
// closed the listener. It must be zero.
CloseBodiesWhileParked int64
// CloseBodiesAtShutdownReturn is how many teardown bodies had been entered at
// the instant the FIRST Shutdown call returned. On an expiry arm it must be
// zero: neither of Shutdown's failure branches closes the owned store
// (bolt/server/serve.go:874-879), so the store is closed later, by Serve's exit
// path — which is what makes the attribution clause below adjudicable there and
// only a witness on the success path.
CloseBodiesAtShutdownReturn int64
// CloseFaultArmed records that a one-shot fsync fault was armed on the WAL
// close, so the value the server caches is non-nil and its identity across
// callers is a discriminating assertion.
CloseFaultArmed bool
// The four declarative echoes of the arm's configuration. They are in the
// evidence so the adjudicators are pure functions of it and never re-derive an
// expectation from the arm's NAME — which is what lets a falsifiability table
// build a healthy value for any arm and perturb one field of it.
//
// ParkExpected: the arm intended to park a commit inside its fsync.
// ExpiryExpected: the arm bounded Shutdown so it could not drain in time.
// OutOfOrderArm: the arm closed the store directly, with no drain at all.
// FleetArm: the arm drove concurrent committers and is not bit-reproducible.
ParkExpected bool
ExpiryExpected bool
OutOfOrderArm bool
FleetArm bool
// AckedAfterCheckpoint is how many acknowledged commits landed after the arm's
// one checkpoint and therefore live ONLY in the WAL suffix. Without at least
// one, every acked name could come back from the published snapshot and a WAL
// close that flushed nothing would satisfy the durability clause — the same
// measurement db_teardown.go records for its own arms.
AckedAfterCheckpoint int
// ConnsAccepted / ConnsClosed / ConnsPeak / ConnsLiveAtEnd are the connection
// tracker's totals for the whole arm.
ConnsAccepted int64
ConnsClosed int64
ConnsPeak int64
ConnsLiveAtEnd int64
IdleConnsOpened int
// GateArmed records that a fsync rendezvous was armed, GateFired that it was
// actually entered — the reachability observable rmp #2465 established, since
// an ordinal that never matched is a silent no-op.
GateArmed bool
GateFired bool
// ParkedLiveConns is how many server-side connections were live while the
// commit was parked. It is SAMPLED at the instant the fsync gate fires, so it is
// a coverage witness and not a function of the seed: like ConnsPeak it is read
// by the non-vacuity gate but deliberately absent from
// [BoltDrainEvidence.String], for the reason set out there.
// ListenerClosedWhileParked records that the listener was
// already closed in that window, which is what proves Shutdown had passed
// ln.Close() and was inside its drain wait rather than not yet started.
// DialRefusedAfterTeardown is the same observable from a client's point of view,
// taken once the teardown is over (where a probe Dial can create nothing).
ParkedLiveConns int64
ListenerClosedWhileParked bool
DialRefusedAfterTeardown bool
// ShutdownErrs renders what each Shutdown call returned, in call order.
ShutdownErrs []string
// ShutdownFirstNil records whether the FIRST Shutdown drained cleanly, and
// LastShutdownNil whether the LAST one did. They differ on the expiry arms: the
// first reports the expiry, and the last — made after the drain finally
// completed — observes the cached, successful close result.
ShutdownFirstNil bool
LastShutdownNil bool
// ShutdownCalls is how many Shutdown calls the arm made, and
// DistinctShutdownErrs how many distinct error VALUES they returned — compared
// by identity, never by class, because the class cannot tell one published
// value from N re-derived ones.
ShutdownCalls int
DistinctShutdownErrs int
// ShutdownErrIsDrainTimeout / ShutdownErrIsCtx classify the FIRST call's error:
// which of Shutdown's two failure branches was taken.
ShutdownErrIsDrainTimeout bool
ShutdownErrIsCtx bool
// ShutdownErrIsCloseFault records that the first call returned the store's own
// close failure, which is what the publication arm asserts identity over.
ShutdownErrIsCloseFault bool
// ShutdownExpiryBudget is the bound the arm gave Shutdown (zero when it gave
// none), and ShutdownExpiryByCancel records that the bound was an explicit
// cancel on a deadline-free context rather than a deadline.
ShutdownExpiryBudget time.Duration
ShutdownExpiryByCancel bool
// ServeExitErr renders what [server.Server.Serve] returned, and
// ServeExitErrIsCloseFault whether it carried the store's close failure —
// which is how Serve's exit path is shown to observe the SAME cached result
// without comparing a joined error by pointer.
ServeExitErr string
ServeExitErrIsCloseFault bool
// Commits is every write attempt over the wire, in the order it was made.
Commits []BoltDrainCommit
// InterruptedRuns and TerminatedRuns count the writes answered with each of the
// two codes a shutdown-interrupted statement can receive. They are MEASUREMENTS,
// carried so the difference between them is visible in a report: one is transient
// and a driver retries it, the other is demoted to a client error and is never
// retried (see [boltDrainTerminatedCode]).
InterruptedRuns int
TerminatedRuns int
// IgnoredReplies counts the RUN and PULL replies that were IGNORED because the
// session was already FAILED. Such a reply is never an acknowledgement and never
// enters the acked set; it is counted so a run in which the arms measured nothing
// but ignored traffic is visible rather than silently green.
IgnoredReplies int
// PostCloseCommitErr renders what a commit attempted through the transaction
// layer AFTER the teardown returned, and PostCloseCommitRefused whether it was
// refused with [wal.ErrWriterClosed]. This is the one place the writer-closed
// error is identifiable at all (over the wire it is sanitised), so it is both
// the proof the WAL closed and the proof the detector is not blind.
PostCloseCommitErr string
PostCloseCommitRefused bool
// LoopAliveBeforeTeardown is the liveness probe taken BEFORE the teardown (a
// checkpoint request succeeded, so there was a goroutine to join), and
// LoopStoppedAfterTeardown the join verdict (a request afterwards returned
// [checkpoint.ErrCheckpointerStopped]). A WAL-closed error instead would mean
// the loop outlived the WAL — the failure the composed teardown exists to
// prevent.
LoopAliveBeforeTeardown bool
LoopStoppedAfterTeardown bool
// AckedNames are the names whose RUN was acknowledged, IssuedNames every name
// the arm sent, RecoveredNames what a reopen through real recovery found.
// MissingAcked is acked minus recovered (a durability defect) and
// PhantomNames recovered minus issued (a consistency defect).
AckedNames []string
IssuedNames []string
RecoveredNames []string
MissingAcked []string
PhantomNames []string
// PartialNames are recovered nodes missing their age property: a torn
// transaction resurrected by recovery.
PartialNames []string
// RecoveredWALOps is how many WAL ops the reopen's recovery replayed. It is
// what stops the durability clause being answered by a snapshot underneath it,
// and it is a pure function of the seed: an op count does not depend on how wide
// any encoded value happened to be.
RecoveredWALOps int
// WALBytes is the size of the durable WAL image the reopen read. It is a
// DIAGNOSTIC ONLY and deliberately absent from [BoltDrainEvidence.String] and
// from every violation message: a created node's hidden internal key is minted
// as "__cx_"+hex(n) from a PROCESS-GLOBAL counter (cypher/exec/create_node.go)
// and travels inside the WAL frame, so the same seed produces frames of
// different widths depending on how many nodes every other test in the process
// created first. The same limitation, from the same counter, is documented in
// bolt_auth_surface.go and schema_mutation.go.
WALBytes int
// SnapshotPublished records whether a snapshot manifest sits beside the WAL,
// and ReopenClean whether recovery reported a clean image.
SnapshotPublished bool
ReopenClean bool
}
BoltDrainEvidence is what one arm OBSERVED. It carries measurements and no verdicts, so the adjudicators below are pure functions of it and can be falsified by a doctored value rather than by hoping a real run misbehaves.
It is passed by POINTER everywhere: the value is far over gocritic's hugeParam threshold, and nothing mutates it after the arm returns.
func RunBoltShutdownDrain ¶ added in v0.12.0
func RunBoltShutdownDrain(ctx context.Context, cfg BoltDrainConfig) (*BoltDrainEvidence, error)
RunBoltShutdownDrain drives one arm end to end: stand up a WAL-backed store with a running checkpointer, hand it to a real Bolt server as its server.Options.Closer, acknowledge a durable prefix over the genuine wire, tear the server down as the arm's configuration dictates, and reopen the SimDisk image through real recovery.
It returns evidence and no verdict; adjudicate it with [checkBoltShutdownDrain] and [checkBoltShutdownDrainNonVacuity]. An error means the arm could not be DRIVEN, which is a harness failure and not a report.
func (*BoltDrainEvidence) String ¶ added in v0.12.0
func (e *BoltDrainEvidence) String() string
String renders the evidence for a failure message or a test log.
It is deliberately ID-FREE: no session id, no error text that embeds one, no wall-clock duration and no goroutine identity, so two runs of one seed produce byte-identical output. The sanitised failure message a storage error produces carries a crypto-random session id (bolt/server/session.go:530), which is why only the CODE is rendered.
type BoltMetadataObs ¶ added in v0.12.0
type BoltMetadataObs struct {
// SentKeys are the keys the arm put in the `tx_metadata` map, sorted.
SentKeys []string
// Accepted is true when the BEGIN carrying them was answered SUCCESS.
Accepted bool
// GotCode carries a refusal, empty when none came.
GotCode string
// BeginMetaKeys, TerminalMetaKeys and CommitMetaKeys are every key the BEGIN
// SUCCESS, the terminal PULL SUCCESS and the COMMIT SUCCESS carried, each sorted.
// The clause is that no SentKey appears in any of them: `tx_metadata` is read
// nowhere in the module, so there is nothing to echo, and pinning the absence is
// the honest assertion.
BeginMetaKeys, TerminalMetaKeys, CommitMetaKeys []string
}
BoltMetadataObs is the `tx_metadata` probe: what was sent, and every metadata key that came back anywhere.
type BoltModeArm ¶ added in v0.12.0
type BoltModeArm struct {
// Name identifies the arm; SentMode is the value sent, and SentKey whether the
// key was present at all.
Name string
SentMode string
SentKey bool
// RegistryMode is [server.TransactionInfo.Mode] read off the server's own
// open-transaction registry while the transaction was open. It is the INDEPENDENT
// observable: BEGIN's SUCCESS carries no mode, so without the registry the only
// evidence of the server's decision would be the write attempt itself, and a
// clause resting on one observable could not tell a mis-recorded mode from a
// mis-enforced one.
RegistryMode string
// WriteAccepted is true when a write statement inside the transaction was
// answered SUCCESS; WriteCode and WriteMessage carry the refusal when it was not.
WriteAccepted bool
WriteCode, WriteMessage string
// BeginRefused records that the BEGIN ITSELF was refused, with its code and
// message. It is separate from WriteCode because the two are different events
// and were previously conflated: before rmp #2564 no arm could be refused at
// BEGIN, so the driver reused the write fields, and a clause reading them
// could not tell "the transaction never opened" from "the write inside it was
// refused".
BeginRefused bool
BeginCode, BeginMessage string
}
BoltModeArm is one probe of the `mode` extra: what was sent, what the server's own registry says the transaction's mode is, and whether a write inside it was allowed.
type BoltOffer ¶ added in v0.12.0
type BoltOffer struct {
// Version is the top of the offered range.
Version proto.Version
// MinorRange is how far below Version.Minor the client will accept. Zero
// offers that exact version and no fallback.
MinorRange uint8
}
BoltOffer is one version slot of the Bolt client preamble: a version and the MINOR RANGE the client will accept below it.
A slot's wire form is [0x00, minor_range, minor, major], and a minor_range of r means the client accepts [minor-r, minor] (bolt/proto/handshake.go:39-45). The zero value is an EMPTY slot: the preamble always carries four, and a client offering fewer zero-fills the rest, which Negotiate skips (bolt/proto/handshake.go:103-104).
type BoltRouteObs ¶ added in v0.12.0
type BoltRouteObs struct {
// ListenerAddr is [SimServer.ListenerAddr] — the independent reference the
// advertised addresses are compared against.
ListenerAddr string
// SentDB and SentBookmark are what the populated ROUTE asked for, so the clause
// that the reply ignores them names what was ignored.
SentDB, SentBookmark string
// Roles are the roles the table advertised, in the order the server listed them.
Roles []string
// AddressesByRole maps each advertised role to its address list. A role listed
// twice would collide here, so RoleCount records the raw entry count separately.
AddressesByRole map[string][]string
RoleCount int
// TTL is the table's `ttl`, and TTLPresent whether the key was there and of the
// expected type at all.
TTL int64
TTLPresent bool
// TableDB is the table's own `db` field. RoutingTable hardcodes it to the empty
// string (bolt/server/route.go:30), so it does NOT echo SentDB — which is what
// the arm pins.
TableDB string
// ZeroMessageRender is the rendering of the table returned for the ZERO ROUTE
// message, and PopulatedRender the one for the populated message. They are
// compared for equality, which is how "the routing context and the bookmarks are
// ignored" is asserted rather than described.
PopulatedRender, ZeroMessageRender string
// Accepted records that both ROUTE messages were answered SUCCESS.
Accepted bool
// GotCode carries a refusal, empty when none came.
GotCode string
}
BoltRouteObs is the ROUTE payload, read twice: once from a ROUTE carrying a routing context, a bookmark and a database name, and once from the zero message.
type BoltStreamEvidence ¶ added in v0.12.0
type BoltStreamEvidence struct {
// Seed is the seed the run was built from.
Seed uint64
// Reference is the record set the reference drain produced, decoded.
Reference [][]packstream.Value
// ReferenceTerminal / ReferenceHasMore are its terminal reply's kind and
// has_more. A single PULL -1 always exhausts the stream, so has_more must be
// false here; recording it makes the paged arm's true readings comparable
// against a measured false rather than against an assumption.
ReferenceTerminal string
ReferenceHasMore bool
// PageSizes are the seed-drawn n values, in the order they were sent.
PageSizes []int64
// Pages is one entry per PULL {n} of the paged drain.
Pages []BoltStreamPage
// Paged is the concatenation of every RECORD the paged drain delivered.
Paged [][]packstream.Value
// PagingReusable records that a fresh RUN+PULL on the SAME connection was
// acknowledged after the paged drain completed, i.e. the session returned to
// READY.
PagingReusable bool
// WindowPageSizes are the drawn n values of the prefix pages, in order.
WindowPageSizes []int64
// WindowPrefixPages is how many drawn pages were pulled before the DISCARD.
WindowPrefixPages int
// WindowPrefixPagesSeen is one entry per prefix page.
WindowPrefixPagesSeen []BoltStreamPage
// WindowPrefix is what those pages delivered, concatenated.
WindowPrefix [][]packstream.Value
// WindowDiscardN is the seed-drawn n the DISCARD asked to drop.
WindowDiscardN int64
// WindowDiscard is the DISCARD exchange itself.
WindowDiscard BoltStreamPage
// WindowSuffix is what the closing PULL -1 delivered.
WindowSuffix [][]packstream.Value
// WindowSuffixPage is that closing exchange.
WindowSuffixPage BoltStreamPage
// WindowReusable records that a fresh RUN+PULL on the same connection was
// acknowledged afterwards.
WindowReusable bool
// EffectDiscard is the DISCARD exchange that abandoned the write's delivery.
EffectDiscard BoltStreamPage
// EffectStats is the "stats" map the DISCARD's terminal SUCCESS reported,
// flattened to int64 counters (a bool counter is recorded as 1).
EffectStats map[string]int64
// EffectLive / EffectRecovered are how many nodes carrying
// boltStreamDiscardLabel exist in the live engine and in a graph reopened
// through real recovery after a crash.
EffectLive, EffectRecovered int
// EffectReusable records that a fresh RUN+PULL on the same connection was
// acknowledged after the mid-stream DISCARD, which is the AC's "a post-DISCARD
// RUN succeeds" stated on the connection that did the discarding.
EffectReusable bool
// Refusals are the refusal probes in the order they ran.
Refusals []BoltStreamRefusal
// ProbedQIDs are the qid values the qid probes actually sent. Recorded so the
// non-vacuity gate can require them to be POSITIVE: a probe that sent -1 would
// be asserting the refusal of the current-stream qid.
ProbedQIDs []int64
// RunQIDs are the qid values every RUN SUCCESS in the run reported. Every one
// must be -1, which is what makes "no positive qid is ever minted" an assertion
// rather than an argument.
RunQIDs []int64
// QIDControlRows is how many RECORDs the qid=-1 control drain delivered on a
// stream identical to the refused probes'. Zero would mean "every PULL is
// refused", under which the refusal clauses say nothing about the qid.
QIDControlRows int
// TxArms are the committed and the doomed transaction, in that order.
TxArms []BoltStreamTxArm
// DoomedLive / DoomedRecovered are how many nodes carrying
// boltStreamDoomedLabel exist live and after recovery. Both must be 0.
DoomedLive, DoomedRecovered int
// CommittedLive / CommittedRecovered are the same census for
// boltStreamCommitLabel. Both must equal boltStreamInFlightCap: the cap allows
// exactly that many cursors, and the arm writes one node per cursor.
CommittedLive, CommittedRecovered int
// StallArm marks this evidence as the concurrent slow-consumer arm rather than
// the deterministic battery, so the checkers adjudicate the stall clauses and
// skip the deterministic ones.
StallArm bool
// StallRecordsPulled is how many RECORDs the slow consumer drained before it
// stopped, StallParked whether the stream was still open when the consumer
// stalled, and StallClosedCleanly whether the connection tore down without an
// unexpected transport fault.
StallRecordsPulled int
StallParked bool
StallClosedCleanly bool
// StallBufferedPeak is the MAXIMUM number of bytes observed queued toward the
// stalled consumer, sampled by [boltStreamPollParked].
//
// What it can and cannot show was verified in the harness's own code rather
// than assumed. [halfPipe.write] fills to exactly [simConnBufferSize] and then
// parks (internal/sim/simconn.go:96-124: it chunks each write to the space
// remaining, so the buffer NEVER exceeds the bound). So "the server did not
// buffer past the bound" is an invariant of the pipe, not a property of the
// server, and a clause asserting it cannot fail against a real server. It is
// kept as a guard on the harness itself and is labelled as such.
//
// What the peak DOES establish is that the writer really was blocked when the
// consumer tore the connection down: a peak at the bound means the server had
// rows it could not hand over. The server-side HEAP bound — that the page was
// not materialised into a second in-memory copy ahead of the wire — is measured
// where it can be measured, by the live-heap gate in
// bolt/server/streaming_backpressure_test.go, and is not restated here.
StallBufferedPeak int
// StallSurfaceValues are the integers a FRESH connection's paged drain
// delivered after the stalled connection was torn down mid-stream, and
// StallSurfacePages the pages that drain took.
//
// The values are stored rather than a comparison verdict, because the
// reference here is ARITHMETIC — the drain runs UNWIND range(1, N), so the
// expected sequence is 1..N and the checker derives it without asking the
// server anything. That keeps the stall arm's oracle independent without
// paying for a second full reference drain.
StallSurfaceValues []int64
StallSurfacePages []BoltStreamPage
}
BoltStreamEvidence is everything one run observed. It holds OBSERVATIONS only: every expectation is derived by the checkers from the reference drain, which is what lets a test perturb one field and prove the matching clause fires.
func RunBoltStreamSemantics ¶ added in v0.12.0
func RunBoltStreamSemantics(ctx context.Context, seed uint64) (*BoltStreamEvidence, error)
RunBoltStreamSemantics drives the whole deterministic streaming surface once against a WAL-backed server whose in-flight cursor cap is boltStreamInFlightCap, and returns the evidence.
It is bit-reproducible from seed: every arm is a fixed lock-step script on its own connection, the page sizes are drawn from one seeded stream in a fixed order, and the only other seeded component is the SimDisk the WAL lives on. No field it records is a byte total of a committed write; see the file comment.
The returned error is reserved for harness failures (the store would not open, a dial was refused, a reply did not arrive inside boltStreamReplyBound). A refused message or an unexpected reply shape is EVIDENCE, not an error.
func RunBoltStreamStall ¶ added in v0.12.0
func RunBoltStreamStall(ctx context.Context, seed uint64) (*BoltStreamEvidence, error)
RunBoltStreamStall drives the slow-consumer fault arm: a consumer opens a large stream, drains a tiny prefix, then stalls and finally tears the connection down mid-stream — and the streaming surface must be intact for everyone else afterwards.
It reuses the harness's existing SlowConsumer actor rather than a second implementation of stalling, so the backpressure this arm exercises is the same one bounded by [simConnBufferSize] that the actor was built for.
It is NOT bit-reproducible: how many records reach the bounded buffer before the consumer stops depends on the scheduler. The seed drives the stall DURATION; the oracle is what every interleaving must share — the writer parked, the buffer stayed bounded, the teardown leaked nothing, and a fresh connection's paged drain still matches plain range arithmetic.
func (*BoltStreamEvidence) String ¶ added in v0.12.0
func (e *BoltStreamEvidence) String() string
String renders the evidence for a report.
It carries no session id, no connection id, no address and no timing, and it renders no WAL BYTE total that could be non-zero: byte totals embed a created node's "__cx_"+hex(n) internal key, whose width depends on a process-global counter, so a rendering containing one is not byte-identical across two runs of the same seed. Frame COUNTS are seed-pure and are rendered. Every map is walked in sorted key order for the same reason.
The receiver is a pointer because a BoltStreamEvidence is far past the size at which gocritic flags a value copy, and the renderer is called on every failing run.
type BoltStreamPage ¶ added in v0.12.0
type BoltStreamPage struct {
// Terminal is the terminal reply's kind: "SUCCESS", "FAILURE", "IGNORED", or
// "" when the exchange produced no terminal at all. It is recorded as a kind
// rather than as a bool because a RUN SUCCESS is not an acknowledgement and an
// IGNORED is not one either; only the terminal reply of the drain is.
Terminal string
// Code is the failure code when Terminal is "FAILURE", else "".
Code string
// Requested is the n the client asked for.
Requested int64
// Delivered is how many RECORDs actually arrived before the terminal reply.
Delivered int
// HasMore is the has_more the terminal SUCCESS reported.
HasMore bool
// Bookmarked reports whether the terminal SUCCESS carried a "bookmark" key.
// The bookmark rides on the TERMINAL reply only, so it doubles as an
// independent witness of which page the server considered final.
Bookmarked bool
// Discard records that this exchange was a DISCARD rather than a PULL.
Discard bool
}
BoltStreamPage is one stream-consuming exchange: a PULL {n} or a DISCARD {n} and the terminal reply it drew.
type BoltStreamRefusal ¶ added in v0.12.0
type BoltStreamRefusal struct {
// Name identifies the probe in a violation message.
Name string
// WantCode / GotCode are the failure code the probe must be told and the one it
// received ("" when the probe was not refused at all).
WantCode, GotCode string
// WantMessage is the exact FAILURE text the probe must be told, or "" to waive
// the exact-text check in favour of WantStateInMessage.
WantMessage string
// GotMessage is the FAILURE text actually received.
GotMessage string
// WantStateInMessage is the session state failTransition must name, or "" when
// the refusal is not a failTransition.
//
// It exists because handleRun's authentication gate (session.go:1072) and its
// state gate (:1075) both return failTransition's
// Neo.ClientError.Request.Invalid, so a code match cannot say which refused.
// failTransition reports the ORIGIN state (:1885), which is the discriminator:
// a refusal by the state gate on a live stream names STREAMING or TX_STREAMING,
// whereas the auth gate would name whatever state the de-authorised session sat
// in.
WantStateInMessage string
// NextTerminal is the terminal kind the NEXT request on the same connection
// drew. Measured "IGNORED": the refusal routes through enterFailed, and a
// FAILED session soft-ignores request-phase messages until RESET.
NextTerminal string
// Delivered is how many RECORDs the refused message delivered. It must be 0.
Delivered int
// Accepted records what actually happened: the decisive message drew a SUCCESS.
Accepted bool
// PriorAck records that an earlier statement on the SAME connection was
// acknowledged through its terminal reply. Without it, "the message was
// refused" is equally explained by a connection that never worked.
PriorAck bool
// RecoveredAfterReset records that a RUN+PULL after RESET was acknowledged on
// the same connection, which is what makes the refusal SCOPED rather than
// terminal.
RecoveredAfterReset bool
}
BoltStreamRefusal is one probe whose decisive message must be refused, together with everything needed to attribute the refusal to the gate under test.
type BoltStreamTxArm ¶ added in v0.12.0
type BoltStreamTxArm struct {
// Name identifies the arm.
Name string
// RefusalCode / RefusalMessage are the typed FAILURE the over-cap RUN drew.
RefusalCode, RefusalMessage string
// FramesBefore / FramesAfter bracket the arm with wal.Stats.Frames.
FramesBefore, FramesAfter uint64
// BytesBefore / BytesAfter bracket it with wal.Stats.Bytes. Only a ZERO delta
// is adjudicated or rendered; see the file comment on seed-purity.
BytesBefore, BytesAfter uint64
// Cycles is how many RUN+PULL cycles the server accepted inside the
// transaction. Each accepted RUN appends one cursor to tx.results.
Cycles int
// CursorsAtRefusal is the "open=N" figure the cap's own diagnostic reported,
// parsed back out of the message. It is the server's own count of accumulated
// cursors, so agreeing with Cycles is a cross-check between two independent
// accountings.
CursorsAtRefusal int
// RefusalObserved records that a typed FAILURE arrived within
// boltStreamReplyBound rather than the round trip stalling.
RefusalObserved bool
// Committed records that the arm's COMMIT was acknowledged.
Committed bool
}
BoltStreamTxArm is one explicit transaction driven either to COMMIT under the in-flight cursor cap or to the cap's refusal, bracketed by the WAL counters.
type BoltTimeoutArm ¶ added in v0.12.0
type BoltTimeoutArm struct {
// Name identifies the arm.
Name string
// SentTimeoutMS is the `tx_timeout` value the arm put on BEGIN, in
// milliseconds. SentKey distinguishes "sent zero" from "omitted".
SentTimeoutMS int64
SentKey bool
// IdleBound / DefaultTotalBound are the two server-side bounds installed, in
// VIRTUAL time. They are recorded because the reaper fires at the earlier of the
// two and the whole attribution of the arm rests on which one that is.
IdleBound, DefaultTotalBound time.Duration
// Advanced is how much VIRTUAL time the arm delivered, in the order it delivered
// it. An arm may advance twice: once to show a bound is NOT yet reached, once to
// reach it.
Advanced []time.Duration
// ReapedAfter is the index into Advanced of the advance that reaped the
// transaction, or -1 when no advance did.
ReapedAfter int
// TimersArmed is how many timers the server registered on the injected clock
// across the arm. It is what makes a NOT-reaped reading evidence: a reaper that
// declined is a different thing from a reaper that was never armed, and only
// this counter separates them. syncTxTimer's NewTimer is the only clock timer
// registration in bolt/server (verified exhaustively for rmp #2482), so the count
// is attributable.
TimersArmed int64
// GotCode / GotMessage are what the client was told on its next request-phase
// message after the reap. SecondReplyKind is what the message AFTER that drew,
// which must be IGNORED.
GotCode, GotMessage string
SecondReplyKind string
// Committed is true when the arm's COMMIT succeeded, which is what a NOT-reaped
// control must report.
Committed bool
// ReplyElapsed is the REAL time from delivering the reaping advance to holding
// the abort in hand. It is the "aborted rather than stalled" instrument and, being
// scheduling-dependent, is kept out of the rendering.
ReplyElapsed time.Duration
}
BoltTimeoutArm is one probe of the transaction-timeout reaper: the bounds it was run under, whether the advance reaped the transaction, and what the client was told when it next spoke.
type BoltTxListingRow ¶ added in v0.12.0
type BoltTxListingRow struct {
// IDSuffix, Principal, Mode, Remote, State and Query are the listing's fields
// verbatim, except that the id is reduced to its sequence suffix.
IDSuffix string
Principal string
Mode string
Remote string
State string
Query string
// StartedAt is TransactionInfo.StartedAt as an offset from the probe's base
// instant. Because no advance is ever in flight while a transaction registers
// — the arm advances only from its single controlling goroutine, and only
// after the arming barrier — this must equal the plan's OpenedAt EXACTLY, not
// approximately.
StartedAt time.Duration
// Elapsed is TransactionInfo.Elapsed verbatim. txRegistry.list computes it as
// clk.Now() minus its own StartedAt (bolt/server/txregistry.go:174,185), so on
// a still fake clock it must equal the listing instant minus StartedAt exactly.
Elapsed time.Duration
}
BoltTxListingRow is one entry of server.Server.Transactions as OBSERVED at the rendezvous, reduced to reproducible values: the id keeps only its sequence suffix and the two instants are offsets from the probe's base instant.
type BoltTxPlanRow ¶ added in v0.12.0
type BoltTxPlanRow struct {
// Principal is the identity the connection authenticated as. It is unique per
// connection and is the only field that can attribute a listing entry to a
// connection, because Remote is a constant.
Principal string
// Mode is the BEGIN access mode, "r" or "w", exactly as sent on the wire.
Mode string
// IDSuffix is the registry's per-server sequence suffix ("-1", "-2", …) taken
// from the observed id. The id's prefix is crypto/rand and is never recorded,
// so a report is reproducible; the suffix IS seed-reproducible, because it is
// the ordinal of the BEGIN on this server.
IDSuffix string
// Query is the single statement run inside the transaction before it went
// silent, which is what the listing must report back.
Query string
// OpenedAt is the fake instant the transaction was opened at, as a duration
// since the probe's base instant.
OpenedAt time.Duration
// Conn is the connection's index in the plan, 0-based.
Conn int
}
BoltTxPlanRow is one transaction the arm opened, as the HARNESS built it. It carries what was sent, never what was observed: it is the independent side of every listing comparison.
type BoltTxQuotaBegin ¶ added in v0.12.0
type BoltTxQuotaBegin struct {
// Phase names the BEGIN's place in the roster.
Phase string
// Principal is the identity the connection authenticated as, which under
// server.NoAuthHandler is exactly the quota key.
Principal string
// Mode is the access mode sent, "r" or "w".
Mode string
// Accepted records that the server answered SUCCESS. Code and Message are the
// FAILURE's, verbatim, when it did not.
Accepted bool
Code string
Message string
// IDSuffix is the per-server sequence suffix of the id the registry minted,
// empty for a refused BEGIN; WantSuffix is what the harness predicted from the
// BEGIN's ordinal among the ACCEPTED ones, because only an accepted BEGIN
// reaches txRegistry.nextID.
IDSuffix string
WantSuffix string
// RegistryBefore/RegistryAfter bracket the exchange with the WHOLE server's
// listing size, and TimersBefore/TimersAfter with the counted timer
// registrations. A refused BEGIN must move neither: it must leave the registry
// exactly as it found it and arm nothing.
//
// The whole-server size is deliberately not the same quantity as the cap —
// the control principal holds transactions of its own — so PrincipalOpenAfter
// carries how many entries the BEGIN's OWN principal held once the answer was
// in hand, which is the number the cap is actually about. The harness counts
// it from the listing rather than from txQuota, which is unexported: they are
// the same set, because Session.txClosed releases the slot and unregisters the
// entry together (bolt/server/session.go:813-814).
RegistryBefore int
RegistryAfter int
TimersBefore int64
TimersAfter int64
PrincipalOpenAfter int
// WithinCeiling records that the exchange completed inside
// [txQuotaRefusalCeiling] of REAL time — the clause that separates a typed
// refusal from a stall. It is a boolean rather than the duration because the
// duration is wall time and not seed-pure.
WithinCeiling bool
}
BoltTxQuotaBegin is one BEGIN the arm sent and what the server answered. It is the arm's ledger: every clause about the cap is graded against a named row of it rather than against a position.
type BoltTxQuotaEvidence ¶ added in v0.12.0
type BoltTxQuotaEvidence struct {
// Arm names the arm; Seed is the seed it was built from.
Arm string
Seed uint64
// Limit is server.Options.MaxOpenTxPerPrincipal as installed; PrincipalA is
// the identity driven to it and PrincipalB the control.
Limit int
PrincipalA string
PrincipalB string
// IdleBound, TotalBound and Step are the virtual geometry; Advances is how
// many advances the reap composition made.
IdleBound time.Duration
TotalBound time.Duration
Step time.Duration
Advances int
// TimersTotal, Untils and Tickers are the counted registrations on the
// injected clock, WantTimers the harness's own prediction of the total, and
// Nows the read counter (not seed-pure; see the type comment). Touches is how
// many open transactions the arm deliberately re-armed by sending them a
// message after the clock had moved.
TimersTotal int64
WantTimers int64
Untils int64
Tickers int64
Nows int64
Touches int
// Begins is the BEGIN ledger, in the order the arm sent them.
Begins []BoltTxQuotaBegin
// CapModes are the access modes the cap-filling transactions used. Both kinds
// must appear: the cap counts a read transaction exactly as it counts a write.
CapModes []string
// RegistryAtCap is the listing size once the principal held its cap.
RegistryAtCap int
// WantRefusalMessage is [txQuotaRefusalMessage] recomputed from PrincipalA and
// Limit; the refusal's own text is on its ledger row.
WantRefusalMessage string
// PostRefusalAccepted records that an autocommit statement sent down the
// REFUSED connection was served. It pins rmp #2561: the quota branch returns
// before Transition and never enters FAILED, so the session is still READY.
// PostRefusalCode and PostRefusalMessage carry what it was answered instead.
PostRefusalAccepted bool
PostRefusalCode string
PostRefusalMessage string
// PredictedReapOrdinals is [idleReapModel.reapOrdinalsWithin] computed BEFORE
// the advances ran, over the three transactions open at that point;
// ObservedReapOrdinals is the ordinal at which each was first absent, in the
// same order. ReapProbeFires are the ordinals at which the arm's own uncounted
// timer received the advance.
PredictedReapOrdinals []int
ObservedReapOrdinals []int
ReapProbeFires []int
// ReapCode and ReapMessage are what the reaped connection was told on its next
// message — the attribution that turns "it is no longer listed" into "the idle
// reaper ended it".
ReapCode string
ReapMessage string
// TerminateOutcome is what server.Server.TerminateTransaction returned for the
// slot the operator reclaimed, and TerminatedGone that the entry left the
// registry.
TerminateOutcome txTerminateOutcome
TerminatedGone bool
// The de-authorised-refusal composition (the rmp #2482 carry-over clause).
// LogoffCommitCode and LogoffCommitMessage are what the refused COMMIT was
// answered with, and LogoffEntryGone that the registry entry was reclaimed
// rather than merely left behind by a refusal that wrote nothing.
LogoffCommitCode string
LogoffCommitMessage string
LogoffEntryGone bool
// Windows are the honest autocommit writes bracketing the run.
Windows []BoltTxWriteWindow
// GhostsLive/GhostsRecovered count [txQuotaGhostLabel] in the live engine and
// after a crash and a real WAL replay; Honest* do the same for the witness.
GhostsLive int
GhostsRecovered int
HonestLive int
HonestRecovered int
}
BoltTxQuotaEvidence is everything one run of the quota arm observed, and NO verdict. The checkers are pure functions of it.
Every field is seed-pure except Nows (it counts how often the HARNESS polled the registry) and the honest windows' byte deltas (the frame payload embeds a node key from a process-global counter). Neither is rendered as a number by BoltTxQuotaEvidence.String.
func RunBoltTxQuota ¶ added in v0.12.0
func RunBoltTxQuota(ctx context.Context, seed uint64) (*BoltTxQuotaEvidence, error)
RunBoltTxQuota drives the per-principal quota arm once and returns the evidence. It is bit-reproducible from seed: the whole arm runs on ONE goroutine, every advance is made by it, and the arm makes no seeded choice — the roster, the modes and the statements are fixed — so the seed reaches only the SimDisk's (unused) fault stream.
The returned error is reserved for harness failures. A cap that refused the wrong principal, a slot that was never returned, or a ghost that survived recovery is EVIDENCE, not an error.
func (*BoltTxQuotaEvidence) String ¶ added in v0.12.0
func (e *BoltTxQuotaEvidence) String() string
String renders the evidence for a report. It renders no raw transaction id — only the per-server sequence suffix — and no quantity that is a function of the process rather than of the seed, so two runs of one seed render byte-identically.
type BoltTxRegistryEvidence ¶ added in v0.12.0
type BoltTxRegistryEvidence struct {
// Arm names the arm; Seed is the seed it was built from.
Arm string
Seed uint64
// TimersAtRendezvous is the counted timer registrations observed once every
// transaction was open and its arming barrier had been reached. TimersTotal is
// the count at the end of the run. Untils and Tickers are the other two
// registration counters, and Nows is the read counter (not seed-pure; see the
// type comment).
TimersAtRendezvous int64
TimersTotal int64
Untils int64
Tickers int64
Nows int64
// IdleBound and TotalBound are the two bounds installed on the SERVER.
// ModelIdleBound is the bound the harness's own [idleReapModel] was built
// from; a control desynchronises the two deliberately, and in the arm proper
// they are equal.
IdleBound time.Duration
TotalBound time.Duration
ModelIdleBound time.Duration
// Step is one advance. SetupElapsed is the virtual time the staggered opens
// consumed before the measured sequence began, and SimElapsed the total the
// arm advanced. RendezvousAt is the virtual instant the listing was taken at.
Step time.Duration
SetupElapsed time.Duration
SimElapsed time.Duration
RendezvousAt time.Duration
// Advances is how many advances the measured sequence made.
Advances int
// Plan is what the harness opened, in the order it opened it.
Plan []BoltTxPlanRow
// Listing is [server.Server.Transactions] at the rendezvous, in the order it
// was returned. ListingOrder repeats that order as id suffixes alone, so a
// re-ordering is legible in a report without reading the rows.
Listing []BoltTxListingRow
ListingOrder []string
// RegistryPeak is the largest listing the run ever observed. Accepted and
// Refused count BEGINs by the server's answer — Refused is the per-principal
// quota's refusal, which this arm installs no cap for and must therefore be
// zero. Reaped counts the transactions observed to leave the registry across
// the measured advances.
RegistryPeak int
Accepted int
Refused int
Reaped int
// PredictedReapOrdinals is [idleReapModel.reapOrdinals] computed BEFORE the
// measured sequence ran; ObservedReapOrdinals is the ordinal at which each
// transaction was first absent from the listing, in the same order as Plan,
// or [reapNever] for one that never left.
PredictedReapOrdinals []int
ObservedReapOrdinals []int
// ProbeFireOrdinals are the ordinals at which the arm's own UNCOUNTED timer
// received the advance. An ordinal missing here is an advance that never
// reached a waiter, which would make "no reap at that ordinal" a statement
// about the harness rather than about the reaper.
ProbeFireOrdinals []int
// SettleTimeoutOrdinals are the ordinals at which the registry never shrank to
// the size the model predicts within [txRegistryObserveTimeout].
SettleTimeoutOrdinals []int
// ReapCodes and ReapMessages are what each abandoned connection was told on the
// one message it was sent AFTER the measured sequence, in plan order: the typed
// FAILURE the reaper arms, or "" when the server answered something else. They
// are what attributes a transaction's ABSENCE from the registry to the reaper
// rather than to a rollback, a dropped connection, or a harness mistake.
ReapCodes []string
ReapMessages []string
// Windows are the four honest autocommit writes, in the order they ran.
Windows []BoltTxWriteWindow
// ReapFrames and ReapBytes are the SUM of the per-advance WAL brackets. Each
// bracket opens before an advance and closes when that advance has settled,
// and the honest windows run BETWEEN advances, outside every bracket — which
// is why this sum can be zero in a run that appended plenty of frames.
//
// The bracket is a LIVE instrument, measured rather than assumed: moving the
// during-reap window inside an advance bracket made this read 4 frames and 195
// bytes and fired the reap-wal clause, so a clean zero here is a statement
// about the reap and not about a bracket that cannot see anything.
ReapFrames uint64
ReapBytes uint64
// SnapshotsBaseline, SnapshotsPeak and SnapshotsFinal are
// [lpg.MVCCStats.ActiveSnapshots] before the first BEGIN, at its highest
// reading, and once every transaction has been reaped. The final reading is
// race-free: Session.abortTx rolls the engine transaction back — releasing the
// horizon slot — BEFORE txClosed unregisters it (bolt/server/session.go:830),
// so a registry the arm has watched empty cannot still be holding a slot.
SnapshotsBaseline int
SnapshotsPeak int
SnapshotsFinal int
// GhostsLive and GhostsRecovered count [txAbandonLabel] in the live engine and
// after a crash and a real WAL replay. HonestLive and HonestRecovered do the
// same for [txAbandonHonestLabel].
GhostsLive int
GhostsRecovered int
HonestLive int
HonestRecovered int
}
BoltTxRegistryEvidence is everything one run of a transaction-registry arm observed, and NO verdict. The checkers are pure functions of it, which is what lets a test perturb exactly one field and prove the corresponding clause fires.
Every field is either seed-pure or documented here as not being so. Two are not: Nows, whose value counts how many times the HARNESS polled the registry (every txRegistry.list call reads the clock, bolt/server/txregistry.go:174) and is therefore a function of scheduling; and the byte deltas of the honest windows, whose frame payload embeds a node's hidden key "__cx_"+hex(n) taken from a process-global counter in cypher/exec, so the frame's SIZE depends on how many nodes the rest of the process created first. Neither is rendered as a number by BoltTxRegistryEvidence.String, which is what keeps a report byte-identical across runs of one seed.
func RunBoltTxRegistryAbandoned ¶ added in v0.12.0
func RunBoltTxRegistryAbandoned(ctx context.Context, seed uint64) (*BoltTxRegistryEvidence, error)
RunBoltTxRegistryAbandoned drives the abandoned-registry arm once and returns the evidence. It is bit-reproducible from seed: the whole arm runs on ONE goroutine — the abandoned connections are silent by construction and need none — every advance is made by that goroutine, and the only seeded choice is each transaction's access mode.
The returned error is reserved for harness failures (the store would not open, a connection would not negotiate, a barrier was never reached). A reap that lands on the wrong ordinal, a listing that misreports a field, or a ghost that survived recovery is EVIDENCE, not an error.
func (*BoltTxRegistryEvidence) String ¶ added in v0.12.0
func (e *BoltTxRegistryEvidence) String() string
String renders the evidence for a report.
It renders NO raw transaction id — only the per-server sequence suffix, which is the reproducible half — and no quantity that is a function of the process rather than of the seed. Two such quantities exist and are rendered as presence rather than as numbers: the Now count, which counts how often the HARNESS polled the registry, and each honest window's byte delta, whose frame payload embeds a node key drawn from a process-global counter. Everything else is seed-pure, so two runs of one seed render byte-identically.
type BoltTxTerminateCall ¶ added in v0.12.0
type BoltTxTerminateCall struct {
// Name is the call's place in the roster: one of [txTerminateCalls].
Name string
// Target is the id SUFFIX the call aimed at.
Target string
// Got is what TerminateTransaction returned; Want is what the contract
// requires for this call.
Got txTerminateOutcome
Want txTerminateOutcome
}
BoltTxTerminateCall is one server.Server.TerminateTransaction call and what it returned, classified so the record is reproducible.
type BoltTxTerminateEvidence ¶ added in v0.12.0
type BoltTxTerminateEvidence struct {
// Arm names the arm; Seed is the seed it was built from.
Arm string
Seed uint64
// IdleBound and TotalBound are the two bounds installed on the SERVER, and
// Advances is how many times the arm advanced the fake clock. The arm's whole
// attribution rests on Advances being zero.
IdleBound time.Duration
TotalBound time.Duration
Advances int
// Begins counts the explicit BEGINs the arm made; TimersArmed, Untils and
// Tickers are the counted registrations on the injected clock, and Nows the
// read count (not seed-pure; see the type comment).
Begins int
TimersArmed int64
Untils int64
Tickers int64
Nows int64
// Plan is the harness's id ledger, in the order it opened the transactions.
Plan []BoltTxTerminateRow
// ListedAtCap is every id suffix the registry held with the whole ledger open;
// ListedAfterTerminate is what it held once the victim had gone. Both are
// SORTED, not in listing order — see the file comment on why the order is not
// deterministic in an arm that never advances the clock.
ListedAtCap []string
ListedAfterTerminate []string
// RegistryPeak is the largest listing the run observed.
RegistryPeak int
// Calls are the terminate calls in the order they were made.
Calls []BoltTxTerminateCall
// SuccessorSuffix is the id suffix of the SECOND transaction the victim's
// connection opened; WantSuccessorSuffix is what the harness predicted from
// its BEGIN ordinal. SuccessorListed and BystanderListed record whether each
// was still listed after the stale-vs-successor terminate had settled.
SuccessorSuffix string
WantSuccessorSuffix string
SuccessorListed bool
BystanderListed bool
// VictimCode and VictimMessage are what the terminated connection was told on
// the one message sent to it after the termination. They are what attributes
// its departure to the operator call rather than to a rollback or a dropped
// connection — and they are adjudicated against [txTerminateFailureCode] and
// [txTerminateFailureMessage], which since rmp #2560 name the OPERATOR event
// specifically. They used to be adjudicated against the reap pair, both of
// whose halves were false here: no timeout had elapsed, and the writer lock the
// message named was retired by rmp #2305/#2306.
VictimCode string
VictimMessage string
// TermFrames and TermBytes bracket the termination itself with [wal.Stats].
// A rollback must append neither.
TermFrames uint64
TermBytes uint64
// Windows are the honest autocommit writes that bracket the termination.
Windows []BoltTxWriteWindow
// GhostsLive/GhostsRecovered count [txTerminateGhostLabel] in the live engine
// and after a crash and a real WAL replay; Honest* and Committed* do the same
// for the two witness labels.
GhostsLive int
GhostsRecovered int
HonestLive int
HonestRecovered int
CommittedLive int
CommittedRecovered int
}
BoltTxTerminateEvidence is everything one run of the operator-termination arm observed, and NO verdict. The checkers are pure functions of it.
Every field is seed-pure except Nows, which counts how many times the HARNESS polled the registry (txRegistry.list reads the clock once per call, bolt/server/txregistry.go:174) and is therefore scheduling-dependent, and the honest windows' byte deltas, whose frame payload embeds a node key drawn from a process-global counter in cypher/exec. Neither is rendered as a number by BoltTxTerminateEvidence.String.
func RunBoltTxTerminate ¶ added in v0.12.0
func RunBoltTxTerminate(ctx context.Context, seed uint64) (*BoltTxTerminateEvidence, error)
RunBoltTxTerminate drives the operator-termination arm once and returns the evidence. It is bit-reproducible from seed: the whole arm runs on ONE goroutine and makes no seeded choice at all — the ledger's roles, modes and statements are fixed — so the seed reaches only the SimDisk's (unused) fault stream.
The returned error is reserved for harness failures. A terminate that ended the wrong transaction, a stale id that was accepted, or a ghost that survived recovery is EVIDENCE, not an error.
func (*BoltTxTerminateEvidence) String ¶ added in v0.12.0
func (e *BoltTxTerminateEvidence) String() string
String renders the evidence for a report. It renders no raw transaction id — only the per-server sequence suffix — and no quantity that is a function of the process rather than of the seed, so two runs of one seed render byte-identically.
type BoltTxTerminateRow ¶ added in v0.12.0
type BoltTxTerminateRow struct {
// Role is the row's part in the arm: one of [txTerminateRoles].
Role string
// Principal is the identity the connection authenticated as, and the only
// field that can attribute a listing entry to a connection (Remote is the same
// constant on every SimConn).
Principal string
// Mode is the BEGIN access mode, "r" or "w", exactly as sent.
Mode string
// Query is the single statement run inside the transaction.
Query string
// IDSuffix is the per-server sequence suffix taken from the OBSERVED id;
// WantSuffix is the suffix the harness predicted from the BEGIN's ordinal on
// this server. The id's prefix is crypto/rand and is never recorded.
IDSuffix string
WantSuffix string
// Conn is the connection's index in the ledger, 0-based.
Conn int
}
BoltTxTerminateRow is one transaction in the harness's own id ledger — the independent reference every listing comparison is made against. It carries what the harness SENT and what it PREDICTED, never a verdict.
type BoltTxWriteWindow ¶ added in v0.12.0
type BoltTxWriteWindow struct {
// Name is the window's position in the run: one of [txAbandonWindows].
Name string
// Node is the name property the window wrote, unique per window.
Node string
// FramesBefore/FramesAfter and BytesBefore/BytesAfter bracket the window with
// [wal.Stats].
FramesBefore, FramesAfter uint64
BytesBefore, BytesAfter uint64
// OpenTx is how many abandoned transactions the registry listed when the
// window ran. It is what proves the window sat where its name says.
OpenTx int
// Ordinal is the advance ordinal the window ran after; 0 before the measured
// sequence began.
Ordinal int
// Committed records that the server answered SUCCESS to both the RUN and the
// PULL.
Committed bool
}
BoltTxWriteWindow is one honest autocommit write, bracketed by the WAL counters. The four windows straddle the reap so that "the reap appended nothing" is measured against a counter that demonstrably moves.
type BoltVersionAbuse ¶ added in v0.12.0
type BoltVersionAbuse struct {
// Name identifies the probe.
Name string
// Kind classifies the reply (`FAILURE`, `SUCCESS`, `IGNORED`, `CLOSED`).
Kind string
// Code and Message are the FAILURE's fields, empty when there was none.
Code string
Message string
}
BoltVersionAbuse is one malformed or ill-timed message and the refusal it drew.
type BoltVersionArm ¶ added in v0.12.0
type BoltVersionArm struct {
// Name is the target's name (`4.4`).
Name string
// Asked is the version the arm set out to negotiate.
Asked proto.Version
// Canonical is what the CANONICAL spelling (exact version, slot 0, range 0)
// negotiated, and Spelled what the SEED-CHOSEN spelling negotiated. Both must
// equal Asked; recording them separately is what makes
// `arm-spelling-invariant` a comparison rather than a restatement.
Canonical proto.Version
Spelled proto.Version
// Spelling renders the seed-chosen offer preamble, and SpellingDiffers
// reports whether it differed from the canonical one at all — the non-vacuity
// question for the invariance clause.
Spelling string
SpellingDiffers bool
// NegotiateErr classifies a handshake failure, empty on success.
NegotiateErr string
// Auth holds the authentication script's outcomes.
Auth BoltVersionAuthProbes
// Entity holds the entity-encoding capture.
Entity BoltVersionEntity
// Temporals holds one entry per temporal kind, in a fixed order.
Temporals []BoltVersionTemporal
// Params are the decoded round-trip values for each parameter kind, in the
// fixed order of [boltVersionParamCases]. They are compared across versions
// by value AND by concrete Go type, following rmp #2484: the dynamic type IS
// the wire encoding, so an Integer silently replaced by the
// identically-rendered Float must fail.
Params []any
// ParamErr classifies a round-trip failure, empty on success.
ParamErr string
// TxCommitted reports whether the arm's explicit BEGIN/RUN/COMMIT committed,
// TxBookmark the bookmark its COMMIT returned (rendered positionally; see the
// file comment), and TxCountAfter the number of marker nodes visible
// immediately afterwards.
TxCommitted bool
TxBookmark string
TxCountAfter int
// TxErr classifies a transaction-script failure, empty on success.
TxErr string
// Abuse holds one entry per malformed/ill-timed probe, in the fixed order of
// [boltVersionAbuseCases]: the code and message the server answered.
Abuse []BoltVersionAbuse
}
BoltVersionArm is everything one negotiated version produced.
type BoltVersionAuthProbes ¶ added in v0.12.0
type BoltVersionAuthProbes struct {
// HelloGoodKind / HelloGoodCode: a HELLO carrying the CORRECT credentials.
HelloGoodKind string
HelloGoodCode string
// RunAfterHello*: a RUN sent immediately after that HELLO, with no LOGON.
RunAfterHelloKind string
RunAfterHelloCode string
RunAfterHelloMessage string
// HelloWrong*: a HELLO carrying a WRONG password, on its own connection.
HelloWrongKind string
HelloWrongCode string
// WrongHelloConnAlive reports whether the connection survived that HELLO. At
// the inline-auth versions a failed HELLO makes the session DEFUNCT and the
// serve loop closes the socket (bolt/server/session.go:911), so the next write
// fails; at the deferred-auth versions HELLO never authenticated, so it
// cannot have refused and the connection is intact.
WrongHelloConnAlive bool
// HelloBareKind / HelloBareCode: a credential-LESS HELLO, on its own
// connection.
HelloBareKind string
HelloBareCode string
// LogonKind / RunAfterLogonKind: the LOGON that follows that bare HELLO where
// the version allows one, and a RUN after it.
LogonKind string
RunAfterLogonKind string
// ResetAfterHelloKind / RunAfterReset*: a RESET sent on a session that has
// completed HELLO but not LOGON, and the RUN that follows it.
ResetAfterHelloKind string
RunAfterResetKind string
RunAfterResetCode string
RunAfterResetMessage string
}
BoltVersionAuthProbes records what each step of the authentication script drew at one negotiated version. Every field is a CLASSIFICATION (`SUCCESS`, `FAILURE`, `CLOSED`) plus, where a refusal is expected, the code and message, so nothing process-dependent reaches the rendering.
type BoltVersionEntity ¶ added in v0.12.0
type BoltVersionEntity struct {
// Census is every (tag, arity) in the record, in encounter order, as the
// independent reader saw it.
Census []boltWireWidth
// CodecCensus is the same, as the module's own decoder saw it. The two must
// agree; see `encoding-walker-agrees-with-codec`.
CodecCensus []boltWireWidth
// Bytes is the record's length on the wire. It is a coarse but completely
// independent witness that the two versions did not emit the same message.
Bytes int
// NodeID / NodeLabels / NodeProps are the returned node's version-INVARIANT
// core.
NodeID int64
NodeLabels []string
NodeProps map[string]any
// RelID / RelStart / RelEnd / RelType / RelProps are the returned
// relationship's version-invariant core.
RelID int64
RelStart int64
RelEnd int64
RelType string
RelProps map[string]any
// PathNodeIDs / PathRelIDs / PathIndices are the returned path's
// version-invariant core.
PathNodeIDs []int64
PathRelIDs []int64
PathIndices []int64
// ElementIDs are the trailing element_id strings, in encounter order: the
// node's, then the relationship's three, then the path's. It is EMPTY at the
// Bolt 4 layout, which is itself the assertion.
ElementIDs []string
// Err classifies a capture failure, empty on success.
Err string
}
BoltVersionEntity is the wire shape of one graph entity captured at one version: the struct census of the whole RECORD plus the semantic core the checker compares ACROSS versions.
type BoltVersionMatrixEvidence ¶ added in v0.12.0
type BoltVersionMatrixEvidence struct {
// Negotiations are the raw-preamble probes in the order they ran.
Negotiations []BoltVersionNegotiation
// Supported is what [proto.SupportedVersions] advertised at collection time,
// compared against [boltVersionSupportedTripwire].
Supported []proto.Version
// Arms are the per-version arms in the order of [boltVersionTargets].
Arms []BoltVersionArm
// TemporalRef is the harness's OWN reference for every temporal value the
// scenario asks for, computed with Go's time package from the same literals
// the query carries. It is what makes the temporal clauses independent of the
// encoder they adjudicate. NamedZoneResolved is false when the zone database
// was unavailable, which the non-vacuity gate reports as a shortfall rather
// than letting the named-zone clause pass unexercised.
TemporalRef BoltVersionTemporalRefs
NamedZoneResolved bool
// LiveMarkers and RecoveredMarkers are the marker nodes each version
// committed, as counted in the live engine and in a graph reopened through
// real WAL recovery after a crash. The maps are keyed by the version name the
// client itself wrote into the node.
LiveMarkers map[string]int
RecoveredMarkers map[string]int
// Seed is the seed the run was built from.
Seed uint64
}
BoltVersionMatrixEvidence is everything one run of the version matrix observed. It is a pure record: the checkers below adjudicate it and it holds no verdict of its own.
Concurrency contract ¶
It is written by a single goroutine during collection and read-only afterwards; it carries no synchronisation and must not be shared with a running collector.
func RunBoltVersionMatrix ¶ added in v0.12.0
func RunBoltVersionMatrix(ctx context.Context, seed uint64) (*BoltVersionMatrixEvidence, error)
RunBoltVersionMatrix drives the whole version matrix once against a WAL-backed server whose AuthHandler validates credentials, and returns the evidence.
It is bit-reproducible from seed: every arm is a fixed lock-step script on its own connections, the only seeded draws are the offer SPELLINGS (which must not change any outcome — that is the claim), and the only other seeded component is the SimDisk the WAL lives on.
The returned error is reserved for HARNESS failures (the store would not open, a dial was refused, a record never arrived). A refused message, a rejected credential or an unexpected reply shape is EVIDENCE, not an error.
func (*BoltVersionMatrixEvidence) String ¶ added in v0.12.0
func (e *BoltVersionMatrixEvidence) String() string
String renders the evidence as a stable, human-readable block.
What is deliberately EXCLUDED, and why ¶
The rendering is compared byte for byte by [TestBoltVersionMatrix_Deterministic], so anything not reachable from the seed must stay out of it or that test measures the machine rather than the server:
- the COMMIT bookmark's literal text, whose counter is process-global (bolt/server/bookmark.go:13) and therefore depends on how many transactions every other test in the process committed first. It is rendered as a positional token and asserted only on its shape, exactly as rmp #2485 does;
- connection ids, which are per-connection random;
- wall-clock durations, of which this scenario records none at all: no clause here is a deadline, so there is nothing to render;
- entity IDS and the record's byte LENGTH. This one was MEASURED, not assumed, and an earlier draft of this godoc claimed the opposite. Two runs of the same seed produced node ids 38/215 and then 227/48, and records of 138 and then 140 bytes: the id is derived from a node key minted from a PROCESS-GLOBAL counter, so it depends on how many nodes every other test in the process created first, and the byte length follows it through the decimal element_id strings. Both are rendered POSITIONALLY instead — `n0`, `n1`, `e0` in first-encounter order — which keeps the structural information (which entity appears where, and which element_id belongs to which id) while dropping the process-dependent value.
The CHECKERS still read the raw ids and the raw byte length, and are entitled to: every clause over them is a DERIVED relation — the element_id equals the decimal of the id in the same structure, the two majors differ in length — and never a literal.
type BoltVersionNegotiation ¶ added in v0.12.0
type BoltVersionNegotiation struct {
// Name identifies the case in a violation message.
Name string
// Slots renders the four offer slots as the client wrote them, so a failure
// says what was actually sent rather than what the case was named.
Slots string
// Accepted is true when the server answered a non-zero version.
Accepted bool
// Got is the version the server answered (zero when it refused).
Got proto.Version
// ReadErr classifies a failure to read the 4-byte reply, empty when it
// arrived. It is a CLASSIFICATION and not the raw error text, so the
// rendering stays a function of the seed.
ReadErr string
}
BoltVersionNegotiation is one raw-preamble probe: the four slots the client wrote and what the server answered.
Concurrency contract ¶
It is a plain value written by one goroutine during collection and read-only afterwards; it carries no synchronisation of its own.
type BoltVersionTemporal ¶ added in v0.12.0
type BoltVersionTemporal struct {
// Kind names the value (`date`, `datetime-offset`, …).
Kind string
// Tag is the PackStream structure signature.
Tag byte
// Ints are the structure's integer fields in order.
Ints []int64
// Zone is the trailing IANA zone name for a named-zone datetime, empty
// otherwise.
Zone string
// Err classifies a capture failure, empty on success.
Err string
}
BoltVersionTemporal is one temporal value captured at one version: the struct tag and its integer fields, plus the string field a named zone carries.
type BoltVersionTemporalRefs ¶ added in v0.12.0
type BoltVersionTemporalRefs struct {
// EpochDay is date('2020-01-02') as whole days since the Unix epoch.
EpochDay int64
// NanosOfDay is '03:04:05.000000006' as nanoseconds since midnight, shared by
// the localtime and time values.
NanosOfDay int64
// LocalAsUTCSec is the localdatetime's wall clock read as if it were UTC,
// which is what a zone-less local datetime carries on the wire.
LocalAsUTCSec int64
// OffsetUTCSec / OffsetNanos / OffsetSeconds describe the offset-zone
// datetime: the true UTC epoch second, the sub-second part, and the zone
// offset.
OffsetUTCSec int64
OffsetNanos int64
OffsetSeconds int64
// Named describes the named-zone datetime the same way. It is meaningful only
// when the zone database resolved; see
// [BoltVersionMatrixEvidence.NamedZoneResolved].
Named BoltVersionZoneRef
// Duration is the duration's four fields in wire order.
Duration []int64
}
BoltVersionTemporalRefs is the harness's independently computed expectation for every temporal value the scenario asks the server to return. Each field is derived with Go's own time package from the SAME literal the query carries, so the temporal clauses compare the server's encoding against an outside computation rather than against the encoder that produced it.
type BoltVersionZoneRef ¶ added in v0.12.0
type BoltVersionZoneRef struct {
// UTCSeconds is the true UTC epoch second, Nanos the sub-second part, and
// OffsetSeconds the zone's offset from UTC at that instant.
UTCSeconds int64
Nanos int64
OffsetSeconds int64
}
BoltVersionZoneRef is the harness-computed reference for one zoned instant.
type BoundedChurnWriter ¶
type BoundedChurnWriter struct{}
BoundedChurnWriter is an honest writer whose create/delete bias is steered by the current modelled node count so the graph stays BOUNDED near [churnHighWater] over a very long run: below the high-water mark it favours creates and links; at or above it, it favours deletes. It reuses HonestWriter's well-formed statement builders, so every op it emits is a statement the engine accepts. It is the long-running scenario's writer.
Concurrency contract ¶
BoundedChurnWriter is NOT safe for concurrent use; it is invoked from the single simulation goroutine.
func (BoundedChurnWriter) Name ¶
func (BoundedChurnWriter) Name() string
Name returns the actor's identifier.
func (BoundedChurnWriter) NextOp ¶
func (BoundedChurnWriter) NextOp(seed *Seed, oracle *GraphOracle) Op
NextOp steers create-vs-delete by the current node count to keep the working set bounded. Below the high-water mark it creates (and occasionally links); at or above it, it deletes (and occasionally updates), so the modelled graph oscillates around [churnHighWater] indefinitely.
type CSRHeaderFacts ¶ added in v0.12.0
type CSRHeaderFacts struct {
// NVertices and NEdges are the two little-endian uint64 counts that open the
// file. NVertices includes the trailing sentinel offset.
NVertices uint64
NEdges uint64
// Width is the weightSizeBytes header byte: 1, 2, 4 or 8 for the dense native
// weights column, 0 when there are no weights, and 0xFF
// (store/snapshot's weightSizeCodec) for the variable-width codec-encoded
// section rmp #2526 added.
Width uint8
// HasWeights is the hasWeights header byte, decoded as a bool.
HasWeights bool
}
CSRHeaderFacts is a csr.bin header decoded INDEPENDENTLY of the store's own reader.
The independence is the point. "The current reader accepts this fixture and reports width 8" is one component agreeing with itself; a defect in the header parse would move both halves together and stay invisible. A second decoder, written from the layout documented on store/snapshot.WriteCSR, turns that into two sources that can disagree.
func ReadCSRHeaderBytes ¶ added in v0.12.0
func ReadCSRHeaderBytes(b []byte) (CSRHeaderFacts, error)
ReadCSRHeaderBytes decodes the first [csrHeaderBytes] bytes of a csr.bin.
type CertRotationStep ¶ added in v0.12.0
type CertRotationStep struct {
// Name identifies the step in a violation message.
Name string
// WantCN is the Common Name that MUST be in service after this step: the newly
// installed pair for a step expected to succeed, the previous one for a step
// expected to fail.
WantCN string
// ServedCN is the Common Name actually served by GetCertificate afterwards.
ServedCN string
// ReloadErr is the error Reload returned, rendered ("" when it succeeded).
ReloadErr string
// HandshakeErr is the error the verifying TLS handshake returned, rendered
// ("" when the handshake completed).
HandshakeErr string
// KeyBytes is the size of the key file this step left on disk, and KeyWantBytes
// the size of the key the step was WRITING. They exist so the fault a step
// claims to inject is proven distinct from its neighbour's: a torn key is
// SHORTER than the key it truncates, a garbled one is exactly as long. Without
// them two steps could quietly inject the same fault — they already produce the
// same parse error — and the roster would overstate what was covered.
KeyBytes, KeyWantBytes int
// WantReloadOK records the step's INTENT: true when the pair on disk is
// complete and valid, false when the step deliberately broke it.
WantReloadOK bool
// ReloadOK records what actually happened.
ReloadOK bool
// CertMTimeUnixNano / KeyMTimeUnixNano are the mtimes the step's mutation left
// on disk, or 0 when the file is absent. They exist for the preserved-mtime arm:
// that arm's whole claim is that the rotation did NOT advance the mtime, and
// without recording it the arm would silently degrade into an ordinary rotation
// the moment the projection changed.
CertMTimeUnixNano int64
KeyMTimeUnixNano int64
// ValidityRefused records whether the reload was refused BECAUSE the leaf on
// disk was outside its validity window — matched with [errors.Is] against
// [server.ErrCertOutsideValidity], not by reading the message. Without it the
// expired arms would be satisfied by ANY refusal, including a parse failure,
// and a shared failure signal cannot attribute a refusal to its cause.
ValidityRefused bool
// WantValidityRefused is the step's INTENT for [CertRotationStep.ValidityRefused].
WantValidityRefused bool
// Handshook records whether a real TLS handshake completed against the served
// certificate. It must be true after EVERY step, successful or not: a broken
// rotation may not change the certificate, and may not break the one in
// service either.
Handshook bool
}
CertRotationStep records one rotation attempt and what the reloader served afterwards.
type CheckSelection ¶
type CheckSelection struct {
// IndexSpecs are the (Label, Property) indexes the consistency check walks.
// They are declared by the scenario because the engine's index manager
// exposes only opaque names, not (label, property) pairs.
IndexSpecs []IndexSpec
// IndexConsistency runs the full index-vs-base-data consistency check
// ([CheckIndexConsistency]) at the end of the run (and, for the schema-chaos
// scenario, after DDL churn). It is meaningful only for modes that exercise
// indexes. The set of indexes to cross-check is [CheckSelection.IndexSpecs].
IndexConsistency bool
// Search runs the search-algorithm battery ([CheckSearch]) — structural
// parity between the engine graph and the oracle model, per-algorithm
// correctness against independent naive references, and the
// context-cancellation contract of every public context-accepting entry point
// in the five search packages — once at the end of the run. The search scenario also sets [Scenario.SearchEvery] for periodic
// in-loop checks. It is meaningful only for the deterministic mode (the
// battery needs a consistent, quiescent view of the graph).
Search bool
}
CheckSelection chooses which extra invariant checks a scenario runs beyond the always-on per-tick parity check (InvariantChecker.Check). The zero value runs none of the extras.
type CheckpointCadenceConfig ¶ added in v0.12.0
type CheckpointCadenceConfig struct {
// Seed is the master seed for the SimDisk sub-stream.
Seed uint64
// Arm names the variant under test. Empty defaults to
// [ArmCheckpointCadenceClean].
Arm string
// MaxAge and Interval are the cadence geometry handed to
// [checkpoint.Config]. Zero values take [cadenceMaxAge] and
// [cadenceInterval]; a NEGATIVE MaxAge is how the interval-only control asks
// for MaxAge: 0 without colliding with the "unset" meaning of zero.
MaxAge time.Duration
Interval time.Duration
// FaultOnCadenceFire arms a one-shot fsync fault immediately before the
// faulted window, so exactly one periodic fire fails inside its snapshot
// publish — before the WAL is synced and long before the prefix truncate.
FaultOnCadenceFire bool
// SkipTriggerPhase omits the explicit-trigger phase (and therefore the
// age-timer question). The interval-only control uses it: with no MaxAge there
// is no age timer to reset.
SkipTriggerPhase bool
}
CheckpointCadenceConfig parameterises one cadence run. The zero value is not meaningful: RunCheckpointCadence normalises Arm, MaxAge and Interval, but the arm-selecting flags are deliberately explicit.
type CheckpointCadenceEvidence ¶ added in v0.12.0
type CheckpointCadenceEvidence struct {
// Arm is the variant that produced this evidence.
Arm string
// MaxAgeNS and IntervalNS are the geometry read back off the CONSTRUCTED
// checkpointer's behaviour plan, not off the request: [checkpoint.New]
// derives Interval from MaxAge when it is left at zero.
MaxAgeNS int64
IntervalNS int64
// TickersRegistered is how many tickers the loop took from the injected
// clock. It must be 1: the cadence is on the seam, not on wall time.
TickersRegistered int64
// ClockNowCalls and ClockSinceCalls are the loop's total use of the seam, and
// GateReadsOnTicks the part of it made while a TICK was being serviced. The
// split matters: with MaxAge unset the age gate short-circuits before reading
// the clock, so GateReadsOnTicks is 0 on that arm while the run's own explicit
// trigger still moves the totals.
ClockNowCalls int64
ClockSinceCalls int64
GateReadsOnTicks int64
// SimulatedElapsedNS is how much FAKE time the run advanced, and
// RealElapsedNS how long it took on the wall clock. The pair is the witness
// that the cadence is virtual: the simulated total is a large multiple of
// MaxAge whatever the real one happens to be.
SimulatedElapsedNS int64
RealElapsedNS int64
// The frozen-clock control, taken with the loop running and the ticker
// registered: FrozenWindowNS of REAL time in which the fake clock did not
// move.
FrozenWindowNS int64
FrozenWindowFires uint64
FrozenWindowSinces int64
// Ticks is how many one-interval advances the run made and TicksDelivered how
// many of them the fake clock actually delivered to a waiter, measured with the
// harness's own probe ticker on the same clock at the same period. The pair is
// what makes "nothing fired" a statement about the AGE GATE rather than about a
// tick that never arrived.
Ticks int
TicksDelivered int
// TriggerCalls is how many explicit [checkpoint.Checkpointer.Trigger] calls the
// run issued, and TriggeredFires how many of those returned nil (and so
// incremented the checkpoint counter).
TriggerCalls int
TriggeredFires int
// Checkpoints is the checkpointer's own lifetime counter at the end, and
// CadenceFires the arithmetic attribution: the checkpoints this run did not
// ask for.
Checkpoints uint64
CadenceFires int
// FireTicks are the tick ordinals at which the loop RAN a checkpoint —
// successful or failed — observed as a change in the checkpoint counter or in
// LastError. PredictedFireTicks are the ordinals the independent model
// predicts, and PredictedFireTicksNoReset the ordinals the rival hypothesis
// (an explicit trigger does NOT reset the age timer) predicts.
FireTicks []int
PredictedFireTicks []int
PredictedFireTicksNoReset []int
// TriggerTick is the tick ordinal after which the explicit trigger was
// issued (0 when the arm skipped that phase), and UnresetFireTick the ordinal
// at which the rival hypothesis would have fired. FiredAtUnresetTick records
// whether it did.
TriggerTick int
UnresetFireTick int
FiredAtUnresetTick bool
TicksFromTriggerToCadenceFire int
// The transient failure. FailedFireTick is the ordinal whose fire failed (0
// if none did), FailedFireErr what [checkpoint.Stats.LastError] then held, and
// RetryFireTick the ordinal at which the next cadence fire ran.
FaultArmed bool
FailedFireTick int
FailedFireErr string
FailedFireIsSimFault bool
TicksFromFailureToRetry int
RetryFireTick int
// LastErrorAfterRetry is what LastError held once the retry had succeeded:
// the answer to "does a success clear it".
LastErrorAfterRetry string
// The counters bracketing the failure. A failed fire must advance neither.
CheckpointsBeforeFailure uint64
CheckpointsAfterFailure uint64
WALTruncBeforeFailure uint64
WALTruncAfterFailure uint64
WALTruncAfterRetry uint64
// WALBytesBeforeFailure and WALBytesAfterFailure are the DURABLE WAL image on
// the SimDisk across the failed fire. It must not shrink: a checkpoint that
// did not publish must not have reclaimed anything.
//
// On a fault-free arm these three pairs bracket the same window with no failure
// in it — the fire inside it succeeds and legitimately reclaims the prefix — so
// they are informational there and the clauses that read them are gated on
// FailedFireTick.
WALBytesBeforeFailure int
WALBytesAfterFailure int
// CheckpointsNonMonotonic and WALTruncNonMonotonic record whether either
// lifetime counter was ever observed to DECREASE across the run's samples.
CheckpointsNonMonotonic bool
WALTruncNonMonotonic bool
// WALTruncTotal is the reclaimed prefix at the end — the checkpoint redo
// position the WAL control file records (see cpCadenceEnv.walTrunc) — and
// WALTruncAdvances how many samples it strictly increased at. The
// WALTrunc* fields above read the same measure.
WALTruncTotal uint64
WALTruncAdvances int
// AckedKeys are every key acknowledged during the run and MissingAckedKeys
// those the reopen could not find. MissingAckedKeys non-empty is a durability
// defect.
AckedKeys []string
MissingAckedKeys []string
// AckedDuringFailure is how many commits were acknowledged while the
// checkpointer was in its failed state, and AckedAfterLastCheckpoint how many
// landed after the final fire and so live only in the WAL suffix.
AckedDuringFailure int
AckedAfterLastCheckpoint int
// RecoveredWALOps is how many WAL ops the reopen's recovery replayed, WALBytes
// the durable image it read, and SnapshotPublished whether a snapshot manifest
// sits beside it.
RecoveredWALOps int
WALBytes int
SnapshotPublished bool
// ReopenClean is whether the reopen's recovery found no genuine corruption.
ReopenClean bool
}
CheckpointCadenceEvidence is what one cadence run OBSERVED. It carries measurements and no verdict, so the adjudicators below are pure functions of it and can be falsified by a doctored value rather than by hoping a real run misbehaves.
func RunCheckpointCadence ¶ added in v0.12.0
func RunCheckpointCadence(ctx context.Context, cfg CheckpointCadenceConfig) (CheckpointCadenceEvidence, error)
RunCheckpointCadence drives one cadence variant end to end: acknowledge a durable prefix, start the loop, hold the fake clock still to show real time fires nothing, then advance it one interval at a time through a plan that crosses several MaxAge windows — optionally failing one periodic fire with a one-shot fsync fault and optionally interrupting the cadence with an explicit trigger — and finally CRASH the store and reopen it through real recovery.
It installs no global state and may be run beside other work; the fake clock, the SimDisk and the store are all owned by this call.
func (*CheckpointCadenceEvidence) RetryWaitedAFullWindow ¶ added in v0.12.0
func (e *CheckpointCadenceEvidence) RetryWaitedAFullWindow() bool
RetryWaitedAFullWindow reports whether the fire that followed a FAILED one was a whole MaxAge window away rather than the next tick. It is the measurement of the retry cadence the package documentation leaves unstated, and it is a witness in the log as well as a clause.
func (*CheckpointCadenceEvidence) String ¶ added in v0.12.0
func (e *CheckpointCadenceEvidence) String() string
String renders the evidence for a failure message or a test log.
func (*CheckpointCadenceEvidence) TriggerResetTheAgeTimer ¶ added in v0.12.0
func (e *CheckpointCadenceEvidence) TriggerResetTheAgeTimer() bool
TriggerResetTheAgeTimer reports whether the observed fire sequence matched the hypothesis that an explicit trigger postpones the next periodic fire, and not the rival one.
type CheckpointConfig ¶ added in v0.6.0
type CheckpointConfig struct {
// Dir is the SimDisk directory the snapshot and WAL live under. A non-empty
// dir places the WAL at dir/wal and the snapshot at dir/snapshot. When empty
// it falls back to [defaultCheckpointDir].
Dir string
// Every is the tick cadence between checkpoints. A non-positive value falls
// back to [defaultCheckpointEvery]. The first checkpoint fires at the first
// tick that is a positive multiple of Every.
Every int
// Enabled turns in-loop checkpointing on. When true the durable store is
// opened in full-stack mode and the loop checkpoints on the cadence below.
Enabled bool
}
CheckpointConfig parameterises deterministic in-loop checkpointing. The zero value disables it (Enabled == false), the safe default: a run that does not opt in never checkpoints and keeps the legacy WAL-only durable layout.
type ConcurrentConfig ¶
type ConcurrentConfig struct {
// Mix selects the per-connection actor behaviour. When nil, an honest
// read/write mix is used.
Mix *ConcurrentMix
// Seed controls WHAT each connection sends and WHEN its faults fire. Goroutine
// interleaving is NOT seed-controlled (see the package note on the hybrid
// determinism model): this mode is robustness/liveness/leak-checked, not
// bit-reproducible.
Seed uint64
// Connections is the number of concurrent client connections (one goroutine
// each). Values <= 0 are normalised to 1.
Connections int
// OpsPerConn is the number of operations each connection performs before it
// closes. Values <= 0 are normalised to 1. For the transactional roles one
// operation is one whole transaction.
OpsPerConn int
// ContendedCounters is the size of the shared counter key space the
// contended transactional role collides on (rmp #2439). Values <= 0
// normalise to 2. Fewer counters means more conflicts.
ContendedCounters int
// ContendedNamespace separates this run's shared counters from every other
// run's (rmp #2729). The counters are shared by every CONNECTION of this
// run — that sharing is what the contended role exists to create — and by no
// other run, whose acknowledged tally would otherwise be adjudicated against
// a value another run had also incremented.
//
// Name it when several sequential [RunConcurrent] calls form ONE logical run
// that must keep the same counters — [runProductionProfileEvidence] does,
// because it accumulates each counter's total across crash cycles over one
// durable store. Leave it empty and the namespace is derived from Seed, which
// is already distinct per call in every caller that drives one shared server
// concurrently.
//
// A namespace is leased to one run at a time per server: an overlapping run
// that asks for a namespace already held is refused with a typed error rather
// than left to corrupt both oracles.
ContendedNamespace string
}
ConcurrentConfig parameterises a concurrent multi-connection run. Every field is bounded: the harness spawns exactly Connections goroutines, each performing at most OpsPerConn operations, so total work is Connections×OpsPerConn and the connection count and per-connection work are both explicit upper bounds (the reliability mandate's bounded-resources rule).
type ConcurrentMix ¶
type ConcurrentMix struct {
// WriterWeight, ReaderWeight, OverloadWeight are the relative weights for the
// three honest-ish roles. They need not sum to 1.
WriterWeight float64
ReaderWeight float64
OverloadWeight float64
// TxWriterWeight selects the DISJOINT transactional writer (rmp #2439):
// explicit multi-statement transactions over the real Bolt wire (BEGIN, one
// to three uniquely-named creates, COMMIT or a seed-chosen ROLLBACK), whose
// keys never collide with another connection's.
TxWriterWeight float64
// TxContendedWeight selects the CONTENDED transactional writer: an explicit
// read-modify-write on a small shared counter space (the lost-update shape)
// plus a unique marker create, so serialization conflicts are an expected,
// typed outcome the run accounts for separately from failures.
TxContendedWeight float64
// BatchWriterWeight selects the atomic-batch writer (rmp #2440): one
// explicit transaction of exactly isoBatchSize tagged creates, always
// committed, so readers can assert batch-multiple visibility.
BatchWriterWeight float64
// IsoReaderWeight selects the during-run oracle reader (rmp #2440):
// per-connection monotonic reads over the shared counters and the
// batch-multiple atomic-visibility check.
IsoReaderWeight float64
// RYOWWriterWeight selects the read-your-own-writes probe role
// (rmp #2440): write then immediately read back on the same connection.
RYOWWriterWeight float64
// OverloadHeavyWrites lets the overload role draw the heavy-WRITE family
// [OverloadLargeCreateTx] instead of mapping it to a read (rmp #2736). It
// is the write half of the graceful-degradation mandate: a read-only
// overload population cannot establish that a large legitimate WRITE is
// either served in full or refused with a typed error.
//
// It is opt-in rather than the default for a measured reason, not a
// cautious one. A heavy write is ~6-10 ms and ~1.2 MiB of engine heap
// against ~5 us for a read family whose connection is already in the Bolt
// FAILED state, so enabling it everywhere would multiply the cost of every
// caller that leaves Mix nil — including bench/contention's
// dst-concurrent-bolt, whose operation count is calibrated on a documented
// ~2.6 ms per operation. Callers whose subject IS heavy-write saturation
// ask for it by name; the catalogue's "overload" scenario does.
//
// The population stays neutral either way: an enabled run tags, adjudicates
// and then removes each heavy write's nodes, so the node-count oracle of
// [ConcurrentResult.Consistent] holds unchanged and nothing accumulates.
OverloadHeavyWrites bool
}
ConcurrentMix is the per-connection actor selection for a concurrent run. Each connection draws one role from its own seed-derived sub-stream and plays it for the whole connection, so the population is a deterministic function of the master seed even though interleaving is not.
type ConcurrentResult ¶
type ConcurrentResult struct {
// AckedNames is the UNION, across every writer connection, of the unique
// node names whose create was acknowledged (a SUCCESS-terminated PULL). It is
// the durability oracle at the NAME granularity the durable-commit crash
// scenario needs: every name here MUST survive a crash+recovery (recovered ⊇
// acked). Each writer op uses a globally-unique name and is never retried, so
// the union is a set with no duplicates. Populated only for the writer role;
// nil-safe for read/overload-only runs (stays empty).
AckedNames []string
// IssuedNames is the UNION of every create name a writer connection SENT
// (called RUN for), regardless of outcome. It is the phantom oracle: a
// recovered name absent from IssuedNames would be a phantom the durable layer
// invented (recovered ⊆ issued).
IssuedNames []string
// FailedNames is the UNION of every create name whose client observed an
// explicit typed FAILURE (a [proto.Failure] at RUN or as the PULL terminal).
// It is the atomicity oracle: a commit the client saw fail must have applied
// nothing durable, so every FailedNames entry MUST be absent after recovery.
// A name that received an IGNORED terminal (the connection was already in the
// Bolt FAILED state) is deliberately NOT recorded here — its outcome is
// ambiguous, so it is left as merely issued, never asserted-absent.
FailedNames []string
// TxMarkersAcked / TxMarkersRefused are the transaction-granular ledgers
// (rmp #2439): the unique marker names created inside transactions whose
// COMMIT the server acknowledged, and inside transactions the server
// REFUSED (a typed conflict, an explicit rollback, or a failed commit).
// At quiescence every acked marker must be present and every refused
// marker absent — the all-or-nothing assertion at transaction granularity.
TxMarkersAcked []string
TxMarkersRefused []string
// ContendedAcked is, per shared counter, the number of read-modify-write
// increments acknowledged; ContendedFinal is each counter's value read at
// quiescence. Equality is the zero-lost-updates oracle.
//
// The equality holds only because the counters are private to this run's
// namespace (rmp #2729): a counter another concurrent run also incremented
// answers for increments this run never acknowledged, and the oracle would
// then report a lost update that never happened — or, with a run that
// acknowledged nothing readable, report nothing at all.
//
// Both are ALWAYS ConcurrentCounters long, on every path, so a caller may
// index them by the same k without checking (rmp #2552). A counter no
// connection touched reads 0 rather than being absent.
ContendedAcked []int64
ContendedFinal []int64
// ContendedConnections is how many connections the seeded role draw gave the
// contended-writer role. It is POPULATION evidence, not a result: a run that
// drew none never touches a shared counter, so any counter oracle over it is
// vacuous — and the harness must still produce well-formed counter evidence
// for it. It is what lets a test assert it really entered that case instead
// of replicating the draw and hoping (rmp #2552).
ContendedConnections int
Seed uint64
Connections int
AckedCreates int64 // nodes connections committed (eventual oracle)
EngineNodeCount int64 // engine's live node count at quiescence
Panics int64 // recovered panics across all connection goroutines (must be 0)
TransportErrors int64 // unexpected transport errors (must be 0 on a healthy run)
BoundedRejects int64 // typed bound errors (overload caps) — acceptable, not a fault
BaselineRoutines int // goroutine count captured before the run
FinalRoutines int // goroutine count after teardown
// Transaction outcome accounting (rmp #2439). TxIssued counts explicit
// transactions BEGUN; every one ends in exactly one bucket: committed
// (COMMIT acknowledged), conflicted (the typed retriable serialization
// conflict, classified by its Bolt code), rolled back (a deliberate
// client ROLLBACK), failed (any other explicit FAILURE), or ambiguous
// (the connection stopped mid-transaction on a transport error or
// cancellation — its outcome is unknowable client-side and is never
// asserted). Conservation: TxIssued == the sum of the five.
TxIssued int64
TxCommitted int64
TxConflicted int64
TxRolledBack int64
TxFailed int64
TxAmbiguous int64
// TxMissingAcked / TxPhantomRefused are the quiescence verification
// outcomes: acknowledged markers absent from the engine, and refused
// markers present in it. Both must be zero.
TxMissingAcked int64
TxPhantomRefused int64
// During-run isolation oracle tallies (rmp #2440), reconciled from the
// per-connection state after the join and asserted zero on HEAD:
// a monotonic-read regression (a shared counter or the batch population
// observed going backwards on one connection), a same-connection
// read-your-own-writes miss, and a torn atomic batch (a count that is not
// a multiple of the batch size).
IsoMonotonicViolations int64
IsoRYOWViolations int64
IsoBatchViolations int64
// IsoReads counts the oracle observations actually made, so a green run
// is provably non-vacuous.
IsoReads int64
// Heavy-write overload ledger (rmp #2736), populated only when the mix set
// [ConcurrentMix.OverloadHeavyWrites]:
//
// - OverloadHeavyIssued / OverloadHeavyAcked — heavy-write transactions
// SENT, and those whose commit the server acknowledged. Issued is the
// non-vacuity evidence: a run that reports zero drew the family never,
// so any heavy-write claim over it is empty.
// - OverloadHeavyIgnored — heavy writes the server DISCARDED because the
// connection was still in the Bolt FAILED state from an earlier bound.
// Nothing reached the engine, so such an attempt exercises nothing; it
// must be zero, and it is the tally that distinguishes a heavy write
// the engine refused from one that was never issued.
// - OverloadHeavyAdjudications — how many of them had their committed
// population actually counted. It is to the violations below what
// IsoReads is to the isolation violations: a green run with zero
// adjudications proved nothing.
// - OverloadHeavyViolations — heavy writes whose committed population did
// not match their acknowledgement (all of the batch on an acknowledged
// commit, none of it on a refusal). Must be zero, and breaks
// [ConcurrentResult.Consistent] when it is not.
OverloadHeavyIssued int64
OverloadHeavyAcked int64
OverloadHeavyIgnored int64
OverloadHeavyAdjudications int64
OverloadHeavyViolations int64
// WireParamFailures holds one description per divergence found by the Bolt
// parameter type matrix ([probeWireParamTypes], rmp #2462): every PackStream
// kind a driver can bind — String, Integer, Float, Boolean, Null, List, Map —
// is sent over the real wire and verified by read-back before any connection
// spawns. Empty on a clean run; a non-empty slice breaks [Consistent].
WireParamFailures []string
}
ConcurrentResult summarises a concurrent run for assertions and reports. It is the eventual-consistency oracle at quiescence: AckedCreates is the number of node-creating operations connections acknowledged as committed, which must equal the engine's live node count once every goroutine has drained (no committed write lost, no phantom write gained).
func RunConcurrent ¶
func RunConcurrent(ctx context.Context, srv *SimServer, cfg ConcurrentConfig) (ConcurrentResult, error)
RunConcurrent drives cfg.Connections concurrent client connections through the real Bolt server srv, one goroutine per connection, each performing cfg.OpsPerConn seed-derived operations, then waits for every goroutine to finish (quiescence) and reconciles the eventual-consistency oracle against the engine. It honours ctx cancellation: a cancelled context stops connections at their next op boundary and the harness still drains every goroutine before returning.
Determinism ¶
Per the hybrid model, this mode is NOT bit-reproducible: the SEED fixes each connection's role and op sequence, but goroutine interleaving is real and non-deterministic. Correctness is guarded by the returned ConcurrentResult (no panic, no unexpected transport error, eventual oracle==engine) plus the caller's goleak check — not by replay.
Concurrency contract ¶
RunConcurrent spawns exactly cfg.Connections goroutines, each with a defined lifecycle bounded by cfg.OpsPerConn and ctx; all are joined before return, so no goroutine outlives the call. Every goroutine recovers a panic (recording it in the result and terminating cleanly) so one connection's bug cannot crash the harness or mask a leak.
SEVERAL CALLS MAY DRIVE ONE srv AT THE SAME TIME — bench/contention's dst-concurrent-bolt does exactly that — provided each is given a distinct Seed. Every fixture a call owns is private to it: the Bolt parameter probe's node (rmp #2728) and, when the population includes contended writers, the shared counter key space (rmp #2729). The counters are the one fixture whose privacy cannot be structural on its own, because a caller may legitimately want several sequential calls to keep the same counters; a run that asks for a ConcurrentConfig.ContendedNamespace another run is already holding against this server is therefore REFUSED with an error, never allowed to share it.
func (*ConcurrentResult) Consistent ¶
func (r *ConcurrentResult) Consistent() bool
Consistent reports whether the eventual-consistency oracle holds: the engine's node count equals the acknowledged creates, with no panics and no unexpected transport errors, every Bolt parameter kind round-tripped as specified, and every heavy overload write all-or-nothing (rmp #2736). Bounded rejects (overload caps) are expected and do not break consistency because a rejected write is never acknowledged and so is never counted in AckedCreates.
func (*ConcurrentResult) TxConserved ¶ added in v0.12.0
func (r *ConcurrentResult) TxConserved() bool
TxConserved reports whether every issued transaction landed in exactly one outcome bucket.
type Config ¶
type Config struct {
// Workload is the actor mix. When nil, [DefaultWorkload] is used.
Workload *Workload
// OnOp, when non-nil, is called synchronously with each tick and the
// operation about to run, before it is executed. It is an observation hook
// (e.g. for verbose tracing); it must not mutate state or draw from any
// randomness, or it would break reproducibility. It runs on the simulation
// goroutine.
OnOp func(tick int64, op Op)
// OnCrash, when non-nil, is called synchronously after each crash+recovery
// cycle with the crash tick and how many WAL ops recovery replayed. Like
// OnOp it is an observation hook and must not mutate state or draw
// randomness.
OnCrash func(tick int64, replayedWALOps int)
// Checkpoint configures deterministic, in-loop checkpointing of the
// SimDisk-backed store. The zero value disables it (Enabled == false), so a
// run that does not opt in is byte-identical to before. When enabled, the
// durable store is opened in FULL-STACK mode (WAL + snapshot under a checkpoint
// directory) and the loop publishes a real snapshot + truncates the WAL prefix
// every [CheckpointConfig.Every] ticks, so a subsequent crash recovers through
// the full snapshot+WAL path. It requires the durable store, so it implies the
// SimDisk-backed stack even when [Config.Crash] is disabled. See
// [CheckpointConfig].
Checkpoint CheckpointConfig
// EngineOpts configures the in-memory engine the deterministic loop drives
// (the non-crash, non-disk path). The zero value is byte-identical to
// [cypher.NewEngine], so a scenario that does not set it is unaffected. The
// mem-pressure scenario clamps the logical-resource budgets here
// (MaxResultRows / MaxResultBytes / MaxCollectItems) so over-budget ops are
// refused with a typed error, exercising graceful degradation deterministically.
// It is applied only on the in-memory path; the durable (crash/disk) path uses
// the recovery-config engine.
EngineOpts cypher.EngineOptions
// Crash configures deterministic crash/recovery injection. The zero value
// disables it (Enabled == false), which is the safe default: a run that does
// not opt in drives a plain in-memory engine exactly as before, byte for
// byte. When enabled, the simulator instead drives a real SimDisk-backed
// persistence stack (WAL append+sync + recovery replay) so a scheduled crash
// drops the live engine and the store is reopened from the durable image.
Crash CrashConfig
// Disk, when its CapacityBytes > 0, bounds the SimDisk-backed durable store
// to a finite size so the run drives the engine through a disk-full (ENOSPC)
// condition on the real WAL append+sync path. A non-zero capacity implies the
// durable store even when Crash is disabled; the zero value leaves the disk
// unbounded (the prior behaviour). See [DiskConfig].
Disk DiskConfig
// Multigraph, when true, opens the engine's graph (durable and in-memory
// alike) as a directed MULTIGRAPH, so a repeated CREATE between the same
// endpoints adds a parallel edge instance instead of being rejected. Only a
// scenario whose oracle models edges per instance may set it (the
// edge-properties scenario, rmp #2449, discriminates instances by a unique
// eid property); the default (false) keeps the simple-graph shape every
// other scenario's (src,dst,label)-keyed oracle model requires, byte for
// byte.
Multigraph bool
// Seed is the master seed; the entire run is a pure function of it.
Seed uint64
// MaxTicks is the number of ticks (operations) the safety phase runs.
MaxTicks int
// CheckEvery is the invariant-check cadence in ticks. Values <= 0 are
// normalised to 1 (check every tick).
CheckEvery int
// contains filtered or unexported fields
}
Config parameterises a simulation run.
type CountStoreConfig ¶ added in v0.12.0
type CountStoreConfig struct {
Crash CrashConfig
Checkpoint CheckpointConfig
Seed uint64
MaxTicks int
// ParityEvery is the tick cadence of the full cell-by-cell parity walk.
ParityEvery int
// ShapesEvery is the tick cadence of the four count(*) shapes.
ShapesEvery int
// Persons is how many Persons the prologue creates before the loop.
Persons int
// Hubs is how many additional Persons carry the never-churned [csLabelHub]
// label. They are the nodes whose IN-side count cells stay EXACT under churn,
// so the live DIn and T clauses have something to compare.
Hubs int
// MinEdgesForBoundClaim, when > 0, requires the run to have modelled at least
// this many edges at some point, so the boundedness clause's "never |E|" half
// rests on an |E| that actually grew. The short arm leaves it 0 (its budget
// cannot grow |E| far); the soak arm sets it.
MinEdgesForBoundClaim int64
}
CountStoreConfig parameterises one count-store run.
func DefaultCountStoreConfig ¶ added in v0.12.0
func DefaultCountStoreConfig(seed uint64) CountStoreConfig
DefaultCountStoreConfig returns the SHORT-layer configuration for seed, including the crash and checkpoint schedules.
The schedules live HERE and not in [CountStoreConfig.normalise] on purpose. A normalise that re-enabled a disabled schedule would make it impossible for a caller to drive the forced-crash-only arm — the arm that proves the forced crash is not merely a fallback the catalogue seed never takes — and would silently override an explicit choice. normalise fills only the budgets a zero value leaves meaningless.
type CountStoreEvidence ¶ added in v0.12.0
type CountStoreEvidence struct {
// LiveChecks / RecoveredChecks count the parity observations in each phase.
// They are separate because the two phases assert DIFFERENT things: a run
// with no recovered check never evaluated the self-heal claim at all.
LiveChecks int
RecoveredChecks int
// DirtyLiveChecks is how many live observations found a non-empty dirty set.
// Without one, "the reopen cleared the dirty sets" is satisfied by a run that
// never dirtied anything.
DirtyLiveChecks int
// NegativeLiveChecks is how many live observations held a negative cell, and
// NegativeCellsSeen the total number of such cells across the run.
NegativeLiveChecks int
NegativeCellsSeen int
// NegativeArms is how many negative-cell fixtures the run constructed: one in
// the prologue and one after every recovery, so the heal is witnessed on every
// reopen rather than only on the first.
NegativeArms int
// HealedFromDirty counts recoveries whose IMMEDIATELY PRECEDING live
// observation was dirty and whose post-recovery observation was clean — the
// clause that makes the heal a measured transition rather than a static
// property. HealedNegative is the same for a negative cell.
HealedFromDirty int
HealedNegative int
// CellFamilies[i] counts the live observations in which family i held at least
// one cell, indexed by [csFamily]. A run that never populated `T` proved
// nothing about the pair fan-out.
CellFamilies [csFamilyCount]int
// ComparedLive[i] / ComparedRecovered[i] count the model cells of family i that
// were actually COMPARED — not skipped as dirty-covered — in each phase.
//
// These are the counters that matter, and they are separate from CellFamilies
// for a measured reason: after a single DETACH DELETE the DIn and T families
// are dirty for the deleted node's labels for the rest of the session, so a
// live check can hold plenty of cells and compare none of them. The Hub label
// exists to keep some of them exact; these counters are what prove it worked.
ComparedLive csCompared
ComparedRecovered csCompared
// DistinctTLabelsMax is the largest number of distinct labels ever seen in the
// a-position of a `T` cell. One means the multi-label fan-out never happened,
// so the `T` clause only ever compared the trivial single-label case.
DistinctTLabelsMax int
// ShapeProbes[i] counts the adjudications of shape i, ShapeCellChecks[i] the
// subset that also compared the serving count-store cell, and
// ShapeCellSkipped[i] the subset where that cell was dirty-covered and so
// could not be compared. The skip counter is recorded rather than dropped: a
// skip is a coverage deletion, and the only way to see how much of the run it
// removed is to count it.
ShapeProbes []int
ShapeCellChecks []int
ShapeCellSkipped []int
// MaxCells / MaxEdges / MaxBound are the footprint measurements the
// boundedness clause rests on. MaxEdgesAtMaxCells records the modelled edge
// count at the observation that held the most cells, so the report can state
// the ratio the design claims.
MaxCells int
MaxEdges int64
MaxBound int
MaxEdgesAtMaxCells int64
// Relabels is how many `SET`/`REMOVE` label ops the churn actor emitted (the
// only route to `countRelabel`, hence to any dirty marking).
Relabels int
// Crashes / ForcedCrashes / Checkpoints are the recovery coverage the
// recovered-phase clauses depend on.
Crashes int
ForcedCrashes int
Checkpoints int
// Digest folds every clause's (tick, clause, measured pair). It is the
// scenario's reproducibility claim: same seed, same digest. It folds no
// NodeID and no mapper key, both of which come from a process-global counter
// and are not a function of the seed.
Digest uint64
}
CountStoreEvidence is what the run measured, and the basis of its terminal assert-something-was-seen gate.
func (*CountStoreEvidence) Finish ¶ added in v0.12.0
func (e *CountStoreEvidence) Finish(tick int64) []Violation
Finish is the terminal assert-something-was-seen gate: it reports a violation for every clause the run did NOT reach.
It is deliberately unconditional on the configuration — it fires just as loudly when a budget was lowered as when a clause was deleted — so it cannot be silenced by the very change it exists to catch.
func (*CountStoreEvidence) ReproducibleSummary ¶ added in v0.12.0
func (e *CountStoreEvidence) ReproducibleSummary() string
ReproducibleSummary renders exactly the fields that are a pure function of the seed. Every field of this evidence is, so it is the full rendering; it exists so the determinism test compares what the scenario CLAIMS is reproducible, and so a future field that is NOT reproducible has an obvious place to be excluded from.
func (*CountStoreEvidence) String ¶ added in v0.12.0
func (e *CountStoreEvidence) String() string
String renders the evidence for a report and for the run's own output.
type CountStoreProbes ¶ added in v0.12.0
type CountStoreProbes struct {
// contains filtered or unexported fields
}
CountStoreProbes is the stateful checker: it takes parity observations and query-shape adjudications on demand, tracks the monotone vocabulary the boundedness ceiling is computed from, and accumulates the evidence its terminal gate reads.
Concurrency contract ¶
CountStoreProbes is NOT safe for concurrent use; the deterministic loop drives it from a single goroutine.
func NewCountStoreProbes ¶ added in v0.12.0
func NewCountStoreProbes(seed *Seed) *CountStoreProbes
NewCountStoreProbes returns a fresh probe set drawing from seed.
func (*CountStoreProbes) Evidence ¶ added in v0.12.0
func (p *CountStoreProbes) Evidence() *CountStoreEvidence
Evidence returns the accumulated evidence.
func (*CountStoreProbes) Parity ¶ added in v0.12.0
func (p *CountStoreProbes) Parity(sm *Simulator, tick int64, phase csPhase, perturb csPerturb) []Violation
Parity takes one parity observation in the given phase, records the evidence it implies, and returns the clauses that fired.
The recovered phase additionally credits a HEAL when the immediately preceding live observation was dirty (or held a negative cell) and this one is not: that is the transition the self-heal claim describes, and crediting it only on the transition is what stops a clean-all-along run from claiming it.
func (*CountStoreProbes) Shapes ¶ added in v0.12.0
func (p *CountStoreProbes) Shapes( ctx context.Context, sm *Simulator, tick int64, perturb csPerturb, ) []Violation
Shapes adjudicates the six `count(*)` pattern shapes against references derived from the shadow model, and — where the shape is served by an EXACT count-store T cell — against that cell too.
Three numbers per fully-labelled shape: what the query returned, what the model says, and what the store cell holds. The three-way comparison is the point: it is what connects "the count store is wrong" to something an operator can see, since a wrong store changes the PLAN and not the rows.
type CountStoreWriter ¶ added in v0.12.0
type CountStoreWriter struct {
// contains filtered or unexported fields
}
CountStoreWriter drives exactly the write families the count-store's maintenance path hooks, and nothing else: a Person CREATE, a KNOWS CREATE between two modelled Persons that are not already linked, a `SET n:Vip` / `REMOVE n:Vip` relabel, and a `DETACH DELETE`.
Every template it emits is one the shared GraphOracle already models ([tmplCreatePerson], [tmplCreateKnows], [tmplAddVip], [tmplRemoveVip], [tmplDetachDelete]), so this scenario adds no modelling of its own and cannot drift from the oracle.
It never emits a duplicate KNOWS. MEASURED on this harness's simple (non-multigraph) durable store, a second `CREATE (a)-[:KNOWS]->(b)` on an already-linked pair is REFUSED — `committed=false`, the store unchanged — so emitting one would spend a tick on a rejection instead of on a count delta. GraphOracle.HasKnowsByName is the same guard the edge-property writer uses.
The relabel is the load-bearing family: it is the ONLY route to `countRelabel`, and therefore the only way a dirty marking or a negative cell can exist at all.
Concurrency contract ¶
CountStoreWriter is NOT safe for concurrent use; it is invoked from the single simulation goroutine.
func (*CountStoreWriter) Name ¶ added in v0.12.0
func (*CountStoreWriter) Name() string
Name returns the actor's identifier.
func (*CountStoreWriter) NextOp ¶ added in v0.12.0
func (w *CountStoreWriter) NextOp(seed *Seed, oracle *GraphOracle) Op
NextOp picks the next write. It is a pure function of (seed state, oracle state): every branch either draws from seed or reads the model, never both a clock nor a map iteration order.
type CounterUniqueness ¶ added in v0.12.0
type CounterUniqueness int
CounterUniqueness records how many code paths in the module emit a declared counter. It is the property that decides whether the counter, on its own, identifies the path the fault took.
const ( // CounterUniqueToPath means every emission site for the name lives on the one // code path the declaration is about, so an increment IS that path running. CounterUniqueToPath CounterUniqueness = iota // than one path, so the counter moving does NOT by itself say the declared // fault fired. Such a counter is only admissible with a // [RequiredCounter.Discriminator]. CounterSharedWithOtherPaths )
func (CounterUniqueness) String ¶ added in v0.12.0
func (u CounterUniqueness) String() string
String renders a uniqueness classification for a report.
type CoverageBucket ¶
CoverageBucket is one reported bucket: its dimension, key, and hit count.
func (CoverageBucket) Exercised ¶
func (b CoverageBucket) Exercised() bool
Exercised reports whether the bucket has been hit at least once.
type CoverageSummary ¶
type CoverageSummary struct {
// Buckets lists every tracked bucket (dimension-then-key sorted).
Buckets []CoverageBucket
// Exercised is the number of buckets with a non-zero count.
Exercised int
// Unexplored is the number of tracked buckets still at zero.
Unexplored int
}
CoverageSummary is the tracker's report: every bucket across every dimension, in a deterministic order, plus the count of exercised vs unexplored buckets.
func (CoverageSummary) String ¶
func (s CoverageSummary) String() string
String renders the coverage summary as a human-readable block (one line per dimension with its bucket counts), suitable for the CLI -coverage-report output. It ends without a trailing newline.
type CoverageTracker ¶
type CoverageTracker struct {
// contains filtered or unexported fields
}
CoverageTracker accumulates which coverage buckets a swarm has exercised and biases new-run scenario selection toward under-covered scenarios. It is fed one SwarmRun at a time (typically via the swarm's Observe hook) and is the completeness-critic that stops the swarm from re-testing the same happy path.
The tracker derives every signal from already-observable sim-side data — the scenario name, the run outcome, and (for a failing run) the report's failed op kind and violation classes. It adds NO production hook. Signals that would require instrumenting production code (which Cypher exec operators a query used, which crashpoint sites a run hit) are NOT observable from the test side and are reported by CoverageTracker.UnobservableSignals rather than faked.
Concurrency contract ¶
CoverageTracker is safe for concurrent use: every method takes the internal mutex. Swarm workers feed and query it from many goroutines.
func NewCoverageTracker ¶
func NewCoverageTracker(scenarios []string) *CoverageTracker
NewCoverageTracker builds a tracker that biases selection over the given scenario names (typically a registry's Registry.Names). The scenario set is the bounded universe Select chooses from; it is copied so the caller may mutate its slice afterwards. Every dimension starts at zero coverage.
func (*CoverageTracker) Record ¶
func (ct *CoverageTracker) Record(run SwarmRun)
Record folds one completed run into the coverage tally. It tallies the scenario, the coarse outcome, and — for a failing run — the failed op kind and every violation class in the report. It is safe to call from many goroutines (the swarm's Observe hook runs under the aggregator lock, but Record takes its own lock so it is also safe to call directly).
func (*CoverageTracker) ScenarioCoverage ¶
func (ct *CoverageTracker) ScenarioCoverage() map[string]int
ScenarioCoverage returns the per-scenario hit counts (a copy), for tests that assert the bias steered runs toward under-covered scenarios.
func (*CoverageTracker) Select ¶
func (ct *CoverageTracker) Select(_ int, defaultScenario string) string
Select implements ScenarioSelector: it returns the scenario name the next run should execute, biased toward the least-covered scenario in the tracked universe. Ties are broken round-robin (by the Select call counter) so no tied scenario is starved, keeping the bias fair and deterministic for a given call order. When the tracker has no scenario universe it falls back to the default.
func (*CoverageTracker) Summary ¶
func (ct *CoverageTracker) Summary() CoverageSummary
Summary returns a snapshot of the current coverage across every dimension, in a deterministic (dimension, key) order. Zero-count scenario buckets are included so the report distinguishes "tracked but never hit" from "unknown".
func (*CoverageTracker) UnobservableSignals ¶
func (ct *CoverageTracker) UnobservableSignals() []string
UnobservableSignals reports the coverage signals the task brief names that CANNOT be observed from the test side without adding a production hook, so they are deliberately NOT tracked (the constitution forbids adding a hook to production code from the DST harness). It documents the boundary rather than faking the signal.
Returned, in order:
"cypher-exec-operators": which physical operators (Expand, NodeByLabelScan, hash join, index seek, …) a query's plan used. The engine does not export a per-run operator-usage counter through any public API; observing it would require instrumenting cypher/exec. The differential test (#1567) instead exercises operator EQUIVALENCE via the DisableHashJoin / DisableRangeIndexSeek toggles, which is the observable proxy.
"crashpoint-sites": which internal/crashpoint sites a run armed/hit. Crash points are a test-only injection seam with no public hit-counter; the crash-storm scenario exercises them but does not expose which fired. The metrics oracle (#1568) reads only already-exported metrics for the same reason.
The seam is also structurally unbridgeable into the in-process simulation, not merely unobservable: crashpoint.Breakpoint is compiled out without the gograph_crashinject build tag, and with it the hook SIGKILLs the calling process — which would kill the test binary rather than produce the harness's crash (drop the engine, revoke the not-yet-fsync'd dirents, reopen in the same process). Bridging it would require a pluggable handler in internal/crashpoint, i.e. a new hook in a production-callable package, which is exactly what this list exists to refuse.
What the DST does instead is reproduce the WINDOW each site marks using the SimDisk fault primitives, and observe the branch the site guards through surfaces that already exist. The checkpoint-crash-storm scenario (rmp #2465) does this for "recovery.snapshot-promote-post-rename-pre-fsync": it strands a snapshot backup exactly as an interrupted publish would, and adjudicates the promote repair on the durable image (backup-only before the reopen, live-only after it) and on store/recovery's own exported store.recovery.snapshot.promoteParentFsync counter.
type CrashConfig ¶
type CrashConfig struct {
// Enabled turns crash injection on. When false the schedule never fires and
// never draws from the seed, so the workload stream is unperturbed.
Enabled bool
// CrashProb is the per-eligible-tick crash probability, clamped to [0,1].
// A non-positive value falls back to [defaultCrashProb].
CrashProb float64
// StabilityWindow is the minimum tick gap enforced after a restart before
// another crash may fire. A non-positive value falls back to
// [defaultStabilityWindow].
StabilityWindow int64
}
CrashConfig parameterises a CrashSchedule. The zero value disables crashes entirely (Enabled == false), which is the safe default: an existing run that does not opt in behaves exactly as before. When Enabled is true, non-positive fields fall back to their defaults.
type CrashKind ¶ added in v0.12.0
type CrashKind uint8
CrashKind identifies which crash primitive a SimDisk last ran. It is the observable that lets a scenario state — and a gate check — which of the two physically distinct events it intended to model.
type CrashSchedule ¶
type CrashSchedule struct {
// contains filtered or unexported fields
}
CrashSchedule decides, deterministically from the seed, at which ticks a crash (a HOST crash: drop the in-memory engine and keep only what a successful fsync placed on the SimDisk — see SimDisk.CrashHost, which SimStore.Crash aliases) occurs. After a crash the simulator reopens the store from the durable image via real recovery; CrashSchedule then enforces a stability window during which no further crash is scheduled, so recovery is given time to settle and be re-validated before the next fault (mirroring TigerBeetle VOPR's crash probability plus replica_stability).
The decision is a pure function of (seed, tick, last-crash tick): given the same seed it produces the identical crash tick sequence on every run, which is what lets a failure be replayed bit-for-bit. CrashSchedule draws from its own sub-seed (derived from the master seed via [crashSeedMix]) so toggling crashes on or off never shifts the workload's op stream.
Concurrency contract ¶
CrashSchedule is NOT safe for concurrent use; it is consulted from the single simulation goroutine and its draw order is load-bearing for reproducibility.
func NewCrashSchedule ¶
func NewCrashSchedule(seed *Seed, cfg CrashConfig) *CrashSchedule
NewCrashSchedule builds a crash schedule driven by seed and parameterised by cfg. When cfg.Enabled is false the returned schedule is inert: [ShouldCrash] always returns false and consumes no draws, so a run that does not opt into crashes is byte-identical to one built before crash support existed.
The seed passed here must be the crash sub-seed (derived via [crashSeedMix]), never the master workload seed, so that enabling crashes does not shift the workload draw stream.
func (*CrashSchedule) Enabled ¶
func (c *CrashSchedule) Enabled() bool
Enabled reports whether crash injection is active for this schedule.
func (*CrashSchedule) LastCrashTick ¶
func (c *CrashSchedule) LastCrashTick() int64
LastCrashTick returns the tick of the most recent crash this schedule fired, or a negative sentinel before the first crash. It is exposed for reports and tests asserting on crash timing.
func (*CrashSchedule) ShouldCrash ¶
func (c *CrashSchedule) ShouldCrash(tick int64) bool
ShouldCrash reports whether a crash should occur at the given tick. It returns false without drawing when crashes are disabled or when the tick is still inside the post-restart stability window (so the draw stream position depends only on the eligible ticks, keeping the crash sequence a pure function of the seed). On an eligible tick it draws exactly one Bool(crashProb); when that draw fires it records the tick as the most recent crash, opening a fresh stability window before the next eligible tick.
tick must be non-decreasing across calls (the simulator advances it monotonically); calling out of order would corrupt the stability-window bookkeeping.
type CrossReleaseDiffResult ¶
type CrossReleaseDiffResult struct {
// Tag is the prior release compared against.
Tag string
// Divergences lists every op-level difference, each classified. A run can
// agree overall while still carrying benign divergences here.
Divergences []CrossReleaseDivergence
// PriorNodes/PriorEdges/CurrentNodes/CurrentEdges are the end-state counts.
PriorNodes int64
PriorEdges int64
CurrentNodes int64
CurrentEdges int64
// PriorCommits / CurrentCommits are how many ops each side reported as
// committed. They are the commit axis's WITNESS: a run in which neither side
// committed anything compares false against false for every op and would
// agree vacuously — rmp #2729's failure mode, where 192 contended
// transactions produced zero commits and the oracle compared 0 with 0 and
// passed. A zero here is therefore itself a divergence.
PriorCommits int
CurrentCommits int
// Agreed reports whether no UNEXPECTED (non-benign) divergence occurred.
Agreed bool
// FinalCountsMatch reports whether the prior and current end-state counts
// matched.
FinalCountsMatch bool
}
CrossReleaseDiffResult is the outcome of a cross-release DIFFERENTIAL run: whether the prior and current releases agreed on every op (modulo benign, classified divergences) and the full list of classified divergences.
func RunCrossReleaseDifferential ¶
func RunCrossReleaseDifferential(ctx context.Context, repoRoot, tag string, seed uint64, ops int) (CrossReleaseDiffResult, error)
RunCrossReleaseDifferential replays the SAME deterministic op stream against a prior release (over its store, via the helper) and against the CURRENT in-process engine, then diffs the observable per-op results and the end-state. Each per-op difference is CLASSIFIED: a legitimately plan-dependent result (e.g. an unordered LIMIT) is benign; any other difference is an unexpected divergence that fails the comparison.
repoRoot is the GoGraph working-tree root. A build/worktree failure is returned as an error (clean environment-precondition skip); a behavioural divergence is carried in the result.
func (CrossReleaseDiffResult) String ¶
func (r CrossReleaseDiffResult) String() string
String renders the differential result.
type CrossReleaseDivergence ¶
type CrossReleaseDivergence struct {
// Op is the diverging op.
Op Op
// PriorRows / CurrentRows are the two canonical row signatures.
PriorRows string
CurrentRows string
// Reason explains the classification.
Reason string
// Index is the op index that diverged. A negative index marks a divergence
// about the run as a whole rather than about one op.
Index int
// Benign reports whether the divergence is an expected/benign class.
Benign bool
// PriorCommitted / CurrentCommitted are the two sides' COMMIT outcomes for
// this op. They are meaningful on every divergence, and they are the whole
// content of a divergence whose Reason names the commit axis.
PriorCommitted bool
CurrentCommitted bool
}
CrossReleaseDivergence classifies one op's prior-vs-current observable difference. Benign divergences (a query whose result is legitimately plan-dependent, or a deliberately-fixed-bug behaviour) are recorded as classified, NOT flagged as failures; an unexpected difference is a regression.
type CrossReleaseUpgradeResult ¶
type CrossReleaseUpgradeResult struct {
// DataCompatError is set when the current code FAILED-STOP opening the prior
// image (refused to recover it). This is the explicit, non-silent
// data-compatibility signal: a clear error rather than a silent mis-recovery.
DataCompatError error
// Tag is the prior release that wrote the image.
Tag string
// CountMismatch is set when the current code OPENED the image but recovered
// different node/edge counts than the prior release's own recovery — a genuine
// current-code data-compatibility regression.
CountMismatch string
// PriorLiveNodes / PriorLiveEdges are the prior release's LIVE engine counts
// after the write phase (before any reopen).
PriorLiveNodes int64
PriorLiveEdges int64
// PriorSelfNodes / PriorSelfEdges are the counts the prior release's OWN
// recovery rebuilds from the image — the durable truth the current code must
// reproduce.
PriorSelfNodes int64
PriorSelfEdges int64
// RecoveredNodes / RecoveredEdges are the CURRENT code's counts after reopen.
RecoveredNodes int64
RecoveredEdges int64
// ReplayedWALOps is how many WAL ops the current recovery replayed.
ReplayedWALOps int
// PriorWALFidelityGap is true when the prior release's OWN recovery already
// diverges from its live counts (its WAL does not round-trip in its own
// release). This is a PRIOR-release defect, recorded but NOT charged to the
// current code; the current/prior-self contract can still hold.
PriorWALFidelityGap bool
// HelperBuildFallbackErr is the error that forced the helper build to drop
// its checkpoint half at this tag, or nil when the checkpoint-bearing build
// succeeded. See [BuildPriorReleaseHelper].
HelperBuildFallbackErr error
// CheckpointErr is the prior release's own checkpoint failure, empty when the
// checkpoint succeeded or was never attempted.
CheckpointErr string
// SnapshotProvenanceGap is set when the image carries a snapshot directory on
// disk that the CURRENT recovery did not load, or loaded but could not parse.
// It is the fault this task exists to be able to detect at all.
SnapshotProvenanceGap string
// PriorSnapshot is what the current manifest reader saw in the prior
// release's snapshot directory, read straight off disk.
PriorSnapshot PriorSnapshotFacts
// SnapshotLabels / SnapshotProperties are how many label and property records
// the prior release's snapshot components fed back into the recovered graph.
SnapshotLabels int
SnapshotProperties int
// HelperCheckpointBuilt reports that the helper for this tag was built WITH
// its checkpoint half; false means the tag's checkpoint API did not match and
// the image is WAL-only.
HelperCheckpointBuilt bool
// CheckpointPublished reports that the prior release actually published the
// checkpoint (it built AND ran).
CheckpointPublished bool
// SnapshotOpened reports that the CURRENT recovery loaded a snapshot from the
// image. Compared against [PriorSnapshotFacts.Present], which is read from
// the filesystem, so "there was a snapshot and the reader ignored it" is a
// distinguishable state rather than an unfalsifiable false.
SnapshotOpened bool
// SnapshotOnlyRecovery reports the strongest provenance outcome: a non-empty
// graph recovered with a snapshot loaded and ZERO WAL ops replayed, so every
// recovered node came through the prior release's snapshot bytes and the WAL
// cannot account for any of it.
SnapshotOnlyRecovery bool
}
CrossReleaseUpgradeResult summarises a cross-release UPGRADE run: a PRIOR release wrote a durable store image, then BOTH the prior release (via its own recovery) and the CURRENT code reopened it. The cross-version data-compatibility contract is that the current code recovers the prior image IDENTICALLY to the prior release's own recovery of it — so a prior-release WAL that does not round-trip in its own release (a pre-existing prior defect) is surfaced as [PriorWALFidelityGap] rather than blamed on the current code.
func RunCrossReleaseUpgrade ¶
func RunCrossReleaseUpgrade(ctx context.Context, repoRoot, tag string, seed uint64, ops int) (CrossReleaseUpgradeResult, error)
RunCrossReleaseUpgrade performs a true CROSS-VERSION upgrade test: it builds a prior-release helper from tag, has the prior release WRITE a durable store image (running a deterministic op stream derived from seed), then reopens that SAME image with BOTH the prior release's own recovery and the CURRENT recovery code, and asserts the current code recovers the image IDENTICALLY to the prior release.
This is the genuine guard for the data-compatibility regression class the project hit (v0.2.0 -> v0.3.x adjlist recovery panic): if the current code cannot faithfully rebuild a prior release's image it must fail-stop with a clear error (carried in CrossReleaseUpgradeResult.DataCompatError) or recover a different graph than the prior release did (CountMismatch) — never silently lose or fabricate data. A prior-release WAL that does not even round-trip in its own release is flagged (PriorWALFidelityGap) but not charged to the current code, because the current code's job is to read the prior image faithfully, not to retroactively fix a prior release's persistence bug.
repoRoot is the GoGraph working-tree root. A build/worktree failure is returned as an error for the caller to treat as a clean environment-precondition skip; an honest data-compatibility fault is carried in the result, not the error.
func RunCrossReleaseUpgradeWithOptions ¶ added in v0.12.0
func RunCrossReleaseUpgradeWithOptions(ctx context.Context, repoRoot, tag string, seed uint64, ops int, opts XReleaseBuildOptions) (CrossReleaseUpgradeResult, error)
RunCrossReleaseUpgradeWithOptions is RunCrossReleaseUpgrade with explicit helper-build options.
XReleaseBuildOptions.ForceWALOnly drives the whole pipeline over a deliberately WAL-only image. That serves two distinct purposes, and both are load-bearing (rmp #2531):
- It is the snapshot oracle's negative control. Every tag in the harness's list now publishes a snapshot, so every ordinary arm reports SnapshotOpened=true; this arm is the one that must report false, which is what makes the true ones mean anything.
- It is the ONLY remaining exercise of a prior release's WAL-REPLAY path. Publishing a checkpoint truncates the WAL to the snapshot watermark, so recovery satisfies itself from the snapshot and never replays the full op history. A prior-release defect that lives in WAL replay — v0.2.0 has one — becomes invisible the moment the image gains a snapshot. Keeping this arm keeps that path covered rather than trading one blindness for another.
func (*CrossReleaseUpgradeResult) Parity ¶
func (r *CrossReleaseUpgradeResult) Parity() bool
Parity reports whether the current code reopened the prior image faithfully: no fail-stop, the current recovery matched the prior release's own recovery, and — when the image carries one — the prior release's snapshot directory was actually opened rather than skipped. A prior-release WAL fidelity gap does not, by itself, fail parity.
func (CrossReleaseUpgradeResult) String ¶
func (r CrossReleaseUpgradeResult) String() string
String renders the result for a test failure message.
type DBTeardownConfig ¶ added in v0.12.0
type DBTeardownConfig struct {
// Seed is the master seed for the SimDisk sub-stream.
Seed uint64
// Arm names the variant under test; it is carried into the evidence so the
// non-vacuity gate can apply the concurrency floor to the arm that claims
// concurrency. Empty defaults to [ArmDBTeardownConcurrentClosers].
Arm string
// Closers is how many goroutines call Close/CloseCtx simultaneously (< 1
// normalises to [dbTeardownClosers]). Two further SERIAL calls always follow
// them — a Close and then a CloseCtx — which is how the "Close after
// CloseCtx" and "CloseCtx after Close" orderings are pinned by the same
// identity clause.
Closers int
// CancelBeforeClose cancels the context handed to CloseCtx before any closer
// runs. It also forces EVERY concurrent closer onto CloseCtx: with the
// io.Closer Close() in the mix a background context could win the sync.Once
// and the arm would silently stop testing cancellation.
CancelBeforeClose bool
// FinalCheckpoint wires [store.WithFinalCheckpoint], so step 1 of the
// teardown exists at all and the context has something to bound.
//
// It belongs to the cancelled-context arm and to no other, which was MEASURED
// rather than assumed: the final checkpoint folds the whole WAL suffix into a
// fresh snapshot and truncates the WAL to nothing, so with it enabled the
// reopen replayed 0 WAL ops and every acknowledged key came back from the
// snapshot. An arm whose subject is the WAL CLOSE therefore leaves it off, so
// its durability clause is answered by the WAL the close secured.
FinalCheckpoint bool
// FaultOnClose arms a one-shot fsync fault on the next Sync, which is the
// WAL close's own fsync, so the teardown produces a NON-NIL error whose
// identity is a discriminating observation. It is mutually exclusive with
// InFlightCommit (both claim the next Sync ordinal) and is normally combined
// with FinalCheckpoint disabled, so the fault cannot be eaten by a
// checkpoint's fsync.
FaultOnClose bool
// InFlightCommit parks one commit inside its WAL fsync (a [SyncGate]) and
// keeps it there while the closers run, so the teardown genuinely races an
// in-flight writer rather than an idle store.
InFlightCommit bool
// SkipCheckpointerHandoff is the SENSITIVITY seam: the DB is built WITHOUT
// [store.WithCheckpointer], so nothing joins the checkpoint goroutine. It
// reproduces the leak the composed teardown exists to prevent, which is how
// the join clause is proved to fire. The run stops the loop itself
// afterwards, so the seam leaves no goroutine behind.
SkipCheckpointerHandoff bool
// Probe, when non-nil, is called after every closer has returned and BEFORE
// the run stops anything the DB did not stop itself. It is the seam that
// lets a test observe the goroutine population at the instant the teardown
// claims the checkpoint loop is joined. It must not touch the store.
Probe func()
}
DBTeardownConfig parameterises one teardown run. The zero value is not meaningful: RunDBTeardown normalises Arm and Closers, but the arm-selecting flags are deliberately explicit.
type DBTeardownEvidence ¶ added in v0.12.0
type DBTeardownEvidence struct {
// Arm is the variant that produced this evidence.
Arm string
// CloseErr renders the value every caller was expected to observe (the first
// closer's), and CloseErrNil records whether it was nil.
CloseErr string
// PostCloseTriggerErr is what a checkpoint request returned after the
// teardown, and PostCloseCommitErr what a commit attempt returned.
PostCloseTriggerErr string
PostCloseCommitErr string
// InFlightCommitErr is the error of the commit that was parked inside its
// fsync while the closers ran (empty when it was acknowledged).
InFlightCommitErr string
// AckedKeys are the keys of the commits acknowledged BEFORE the teardown,
// and MissingAckedKeys those the reopen could not find. MissingAckedKeys
// non-empty is a durability defect.
AckedKeys []string
MissingAckedKeys []string
// Closers is the number of concurrent callers, SerialClosers the number of
// serial calls made after them (always 2: a Close then a CloseCtx).
Closers int
SerialClosers int
// DistinctCloseErrs is how many distinct error VALUES the callers observed.
// It must be 1: the teardown publishes one result.
DistinctCloseErrs int
// WriterClosedClosers counts callers whose error carried
// [wal.ErrWriterClosed] — the signature of a second WAL close reaching a
// caller.
WriterClosedClosers int
// TeardownBodyRuns is how many times the quiesce callback (and therefore the
// WAL close inside it) was invoked. The sync.Once must make it exactly 1.
TeardownBodyRuns int64
// CloseCalls, CloseErrorMetric, FinalCheckpointErrorMetric and
// TriggerCtxErrors are the DB's and the checkpointer's OWN instrumentation,
// read independently of everything above: how many calls reached CloseCtx,
// how often the teardown recorded an error (inside the Once), how often the
// final checkpoint's error was classified as a genuine failure rather than a
// benign cancellation, and how often TriggerCtx returned an error.
CloseCalls uint64
CloseErrorMetric uint64
FinalCheckpointErrorMetric uint64
TriggerCtxErrors uint64
// Counters is every metric the recording sink saw across the teardown. It is
// what the arm's [ScenarioCounterDecl] is adjudicated against, so the
// coverage precondition reads the SAME observation the clauses above do.
Counters MetricsObservation
// CheckpointsBefore / CheckpointsAfter bracket the teardown, so the witness
// can report whether step 1 actually folded under a cancelled context.
CheckpointsBefore uint64
CheckpointsAfter uint64
// WALBytes is the size of the durable WAL image at reopen and
// SnapshotPublished whether a snapshot manifest sits beside it. Together they
// are the shape floor: an assertion about recovery over an absent durable
// image would be satisfied by definition.
WALBytes int
// AckedCommits is how many commits were acknowledged before the teardown, and
// AckedAfterCheckpoint how many of those landed after the last checkpoint and
// therefore live only in the WAL suffix.
AckedCommits int
AckedAfterCheckpoint int
// RecoveredWALOps is how many WAL ops the reopen's recovery actually
// replayed. It is the independent proof that the durability clause was
// answered by the WAL and not by the snapshot underneath it.
RecoveredWALOps int
// CtxCancelled records that the context handed to CloseCtx was already
// cancelled, and CloseErrIsCtx that the close returned that cancellation.
CtxCancelled bool
CloseErrIsCtx bool
// CloseErrNil, CloseErrIsSimFault report the published value's shape.
CloseErrNil bool
CloseErrIsSimFault bool
// FaultArmed records that a one-shot fsync fault was armed on the WAL close.
FaultArmed bool
// FinalCheckpoint records that step 1 was wired at all.
FinalCheckpoint bool
// CheckpointerOwned records that the DB was given the checkpointer to stop
// (false only on the sensitivity seam).
CheckpointerOwned bool
// LoopAliveBeforeClose is the liveness probe taken BEFORE the teardown: a
// checkpoint request succeeded, so there was a goroutine to join.
LoopAliveBeforeClose bool
// LoopStoppedAfterClose is the join verdict: after the teardown a checkpoint
// request returned [checkpoint.ErrCheckpointerStopped]. A WAL-closed error
// instead means the loop was still alive and touched a closed WAL — the exact
// failure store.DB exists to prevent.
LoopStoppedAfterClose bool
// PostCloseCommitRefused is whether a commit attempted after the teardown was
// refused with [wal.ErrWriterClosed], which is how "the WAL is closed" is
// observed from outside.
PostCloseCommitRefused bool
// PostCloseKeyRecovered is whether the key of that refused commit was found
// in the reopened graph — an unacknowledged write that became durable.
PostCloseKeyRecovered bool
// InFlightGated / GateFired record that the boundary arm parked a commit
// inside its fsync and that the rendezvous actually happened.
InFlightGated bool
GateFired bool
// CloseBlockedOnInFlight is the ordering observation: no closer had returned
// while the commit was still parked.
CloseBlockedOnInFlight bool
// SnapshotPublished is whether a snapshot manifest exists on the disk.
SnapshotPublished bool
// ReopenClean is whether the reopen's recovery found no genuine corruption.
ReopenClean bool
}
DBTeardownEvidence is what one teardown run OBSERVED. It carries measurements and no verdict, so the adjudicators below are pure functions of it and can be falsified by a doctored value rather than by hoping a real run misbehaves.
func RunDBTeardown ¶ added in v0.12.0
func RunDBTeardown(ctx context.Context, cfg DBTeardownConfig) (DBTeardownEvidence, error)
RunDBTeardown drives one teardown variant end to end: acknowledge a durable prefix, optionally park a commit inside its fsync, run cfg.Closers concurrent Close/CloseCtx callers plus two serial ones, probe what the teardown left behind, and reopen the SimDisk image through real recovery.
It installs the recording metrics backend for the duration (the same global sink MetricsOracle and the group-commit oracle use) and restores the no-op default before returning, so it must run SERIALLY: the caller must not run concurrent metrics-emitting work.
func (*DBTeardownEvidence) FinalCheckpointFolded ¶ added in v0.12.0
func (e *DBTeardownEvidence) FinalCheckpointFolded() bool
FinalCheckpointFolded reports whether step 1 actually took a checkpoint during the teardown. It is a WITNESS: under a cancelled context either outcome is legal (see the file comment), so it is logged and never adjudicated.
func (*DBTeardownEvidence) String ¶ added in v0.12.0
func (e *DBTeardownEvidence) String() string
String renders the evidence for a failure message or a test log.
type DiffResult ¶
type DiffResult struct {
// DivergedOp is the op at the first divergence (zero value when agreed).
DivergedOp Op
// SignatureA / SignatureB are the diverging observable signatures of variant
// A and B at DivergedAt (empty when agreed).
SignatureA string
SignatureB string
// VariantA / VariantB are the variant names, for the report.
VariantA string
VariantB string
// Reason is a human-readable description of the divergence (empty when
// agreed): an end-state mismatch or a per-op result mismatch.
Reason string
// DivergedAt is the 0-based op index of the first divergence (-1 when the
// variants agreed).
DivergedAt int
// Agreed reports whether the two variants produced identical observable
// output for every op and an identical end-state.
Agreed bool
}
DiffResult is the outcome of a differential run: whether the two variants agreed, and on a divergence the first op index, the op, and the two (canonicalised) observable signatures that differed.
func DifferentialTrace ¶
func DifferentialTrace(ctx context.Context, trace Trace, a, b *EngineVariant) (DiffResult, error)
DifferentialTrace replays the SAME recorded Trace against two engine variants and compares their observable outputs op-by-op, reporting the FIRST divergence. The observable output of an op is its canonicalised result-row multiset (for reads) plus the running engine node/edge counts (for writes), and the comparison also asserts the two variants reach an identical end-state.
Because the engine guarantees the default and toggled plans are result-equivalent (DisableHashJoin / DisableRangeIndexSeek exist precisely for this proof), a clean trace must replay to identical output on both.
It spawns no goroutines and is a pure function of trace + the two variants.
func DifferentialTraceInjectB ¶
func DifferentialTraceInjectB(ctx context.Context, trace Trace, a, b *EngineVariant, injectAt int) (DiffResult, error)
DifferentialTraceInjectB replays the trace against both variants but injects a deterministic lost-write fault into variant B at op index injectAt (a write op). It exists for the test that proves the differential CATCHES a behavioural divergence: variant B drops one write, so its end-state diverges from variant A and the first comparison after the drop fails. injectAt < 0 injects nothing (equivalent to DifferentialTrace).
func (DiffResult) String ¶
func (r DiffResult) String() string
String renders a differential result. On a divergence it names the variants, the first diverging op, and the two signatures so the regression is actionable.
type DiskConfig ¶ added in v0.6.0
type DiskConfig struct {
// CapacityBytes, when > 0, is the total byte budget across all files in the
// SimDisk-backed store. A WAL append or checkpoint write that would breach it
// returns an ENOSPC error on the real durability path.
CapacityBytes int64
// ENOSPCOnSync selects where the out-of-space condition surfaces: false
// (eager, at the growing Write) or true (delayed, at Sync). See [SimDisk].
ENOSPCOnSync bool
// FaultRate is the probability (clamped to [0,1]) that any individual Sync on
// the SimDisk-backed durable store fails with [ErrSimFault] and that a freshly
// written sector is marked faulted (a torn write). It is threaded into the
// disk the durable-mode [New] path drives; the zero value (the default every
// scenario carries) disables it, keeping the disk fault-free and every
// existing scenario byte-identical. It takes effect only on the durable path
// — the one a non-zero [DiskConfig.CapacityBytes], [Config.Crash] or
// [Config.Checkpoint] selects — because the plain in-memory engine path never
// touches the disk.
FaultRate float64
}
DiskConfig bounds the simulated disk so the harness can drive the engine through a disk-full (ENOSPC) condition. The zero value (CapacityBytes == 0) leaves the disk unbounded.
type EdgeByName ¶ added in v0.6.0
EdgeByName is a KNOWS edge identified by its endpoint Person names plus its modelled properties, the form the edge-property checker probes the engine with. EID is the instance discriminator for parallel-edge scenarios (0 for edges created by templates that predate instance modelling); a non-zero EID lets the checker pin the probe to exactly one of several parallel edges.
type EdgePropsWriter ¶ added in v0.6.0
type EdgePropsWriter struct {
// contains filtered or unexported fields
}
EdgePropsWriter is the edge-property coverage actor: it grows a Person population and links pairs with KNOWS edges carrying a unique `eid`, an ISO-8601 `since` string, and a float `weight`, then mutates individual edge INSTANCES with the full relationship write surface — standalone SET (r.weight), REMOVE (r.since), the SET-to-null removal (SET r.since = null), the instance-only-key SET (r.note), the whole-entity replace in its plain and WITH-projected forms (SET r = {…}, rmp #2502), the copy-from-instance assignment (SET r2 = r1, rmp #2503), and DELETE r — including over parallel-edge pairs, where one instance is touched and its twin must survive with its own property map. Every op pins its target instance by eid, so the oracle predicts exactly which instance changes.
Concurrency contract ¶
EdgePropsWriter is NOT safe for concurrent use; it is invoked from the single simulation goroutine.
func (*EdgePropsWriter) Name ¶ added in v0.6.0
func (*EdgePropsWriter) Name() string
Name returns the actor's identifier.
func (*EdgePropsWriter) NextOp ¶ added in v0.6.0
func (w *EdgePropsWriter) NextOp(seed *Seed, oracle *GraphOracle) Op
NextOp returns the next seed-chosen operation: a fresh Person, a KNOWS instance between a random pair, a PARALLEL twin of an existing instance, or a standalone SET / REMOVE / SET-to-null / DELETE r pinned to one existing instance. Every family that lacks a viable target falls back to a Person CREATE, so the op stream stays a pure function of (seed state, oracle state).
type EdgeState ¶
EdgeState is the oracle's record of a single directed edge between two oracle node ids, carrying its relationship label and properties. EID is the edge instance's unique discriminator for scenarios that model parallel edges (mirroring the edge's own eid property); it is 0 for every edge created by a template that predates instance modelling.
type Engine ¶
type Engine interface {
// Run executes a Cypher query with string-keyed parameters and returns a
// Result the caller must Close.
Run(ctx context.Context, query string, params map[string]any) (Result, error)
// NodeCount returns the number of live nodes in the engine.
NodeCount() (int64, error)
// EdgeCount returns the number of live edges in the engine.
EdgeCount() (int64, error)
}
Engine is the minimal surface the checker drives. The simulator supplies a thin adapter over the real cypher.Engine (see EngineAdapter).
Concurrency contract ¶
Implementations need only be safe for single-goroutine use; the simulator never calls them concurrently.
type EngineAdapter ¶
type EngineAdapter struct {
// contains filtered or unexported fields
}
EngineAdapter wraps the real github.com/FlavioCFOliveira/GoGraph/cypher.Engine so it satisfies the simulator's minimal Engine interface. It converts the simulator's string-keyed parameter maps into the engine's map[string]expr.Value and projects the engine's rich *cypher.Result onto the checker's narrow Result view.
Concurrency contract ¶
EngineAdapter is NOT safe for concurrent use; the simulator drives it from a single goroutine.
func NewEngineAdapter ¶
func NewEngineAdapter(eng *cypher.Engine) *EngineAdapter
NewEngineAdapter wraps eng. eng must be non-nil.
func (*EngineAdapter) ConstraintNames ¶ added in v0.14.2
func (a *EngineAdapter) ConstraintNames() []string
ConstraintNames returns the name of every schema constraint currently registered on the wrapped engine (cypher.Engine.Constraints), in the engine's deterministic order. It is the other half of the surface the DDL counters oracle derives from (rmp #2822); a constraint's backing index is NOT in it, which is what keeps a CREATE CONSTRAINT from looking like an index effect.
func (*EngineAdapter) CountSnapshot ¶ added in v0.12.0
func (a *EngineAdapter) CountSnapshot() count.Snapshot
CountSnapshot returns a point-in-time copy of the wrapped engine's relationship count-store cells and dirty markings (cypher.Engine.CountSnapshot). The count-store oracle (rmp #2494) reads it as the observed side of its cell-by-cell parity check; the keys are the interned label/relationship-type ids of the graph THIS engine holds, so a caller must resolve them through that same graph's registry — a recovered engine re-interns from scratch and its ids are unrelated to the crashed one's.
func (*EngineAdapter) CountStoreCells ¶ added in v0.12.0
func (a *EngineAdapter) CountStoreCells() int
CountStoreCells reports how many distinct live count-store cells the wrapped engine holds (cypher.Engine.CountStoreCells). The count-store oracle reads it as the footprint the boundedness clause bounds by observed schema cardinality rather than by |V| or |E|.
func (*EngineAdapter) EdgeCount ¶
func (a *EngineAdapter) EdgeCount() (int64, error)
EdgeCount returns the live edge count by running a whole-graph relationship count query through the real engine.
func (*EngineAdapter) Explain ¶ added in v0.12.0
Explain converts params and returns the engine's physical-plan rendering for query without executing it (cypher.Engine.Explain). The access-path parity checker reads the chosen access path (seek vs scan) from this rendering.
func (*EngineAdapter) ListIndexes ¶ added in v0.14.2
func (a *EngineAdapter) ListIndexes() []string
ListIndexes returns the names of every secondary index currently registered on the wrapped engine (cypher.Engine.ListIndexes), in unspecified order. It is the live in-memory registry, INTERNAL indexes included: the "__uniq__" index backing a UNIQUE constraint and the "_btree_num" numeric companion of a user index both appear. The DDL counters oracle (rmp #2822) reads it as one half of the schema surface it derives a statement's expected effect from, and filters those two kinds out itself ([isInternalIndexName]).
func (*EngineAdapter) NodeCount ¶
func (a *EngineAdapter) NodeCount() (int64, error)
NodeCount returns the live node count by running a whole-graph count query through the real engine, so it exercises the same execution path the workload uses.
func (*EngineAdapter) Profile ¶ added in v0.12.0
func (a *EngineAdapter) Profile(ctx context.Context, query string, params map[string]any) (string, error)
Profile converts params, executes the read-only query, and returns the physical plan annotated with per-operator rows, db-hits, and time (cypher.Engine.Profile). The access-path parity checker uses it to assert that a data-touching probe reports non-zero db-hits.
func (*EngineAdapter) Run ¶
func (a *EngineAdapter) Run(ctx context.Context, query string, params map[string]any) (Result, error)
Run converts params and executes a read-only query, returning a Result over the engine's result. The returned Result must be closed by the caller. It routes through the engine's read path (cypher.Engine.Run); use EngineAdapter.RunWrite for statements that mutate the graph.
func (*EngineAdapter) RunWrite ¶
func (a *EngineAdapter) RunWrite(ctx context.Context, query string, params map[string]any) (Result, error)
RunWrite converts params and executes a mutating query through the engine's autocommit write path (cypher.Engine.RunInTx), which the engine requires for CREATE / MERGE / SET / DELETE statements. The returned Result must be closed by the caller.
func (*EngineAdapter) StatsTrackedPairs ¶ added in v0.12.0
func (a *EngineAdapter) StatsTrackedPairs() int
StatsTrackedPairs reports how many distinct (label, property) pairs the wrapped engine currently holds planner statistics for (cypher.Engine.StatsTrackedPairs): 0 until the first completed db.stats.refresh() of the engine's lifetime, and 0 again on a recovered engine until its next refresh. The statistics-regime checker (rmp #2456) reads it as the something-was-seen observable that a rebuild really published statistics.
type EngineVariant ¶
type EngineVariant struct {
// Name is a short label for the variant, used in divergence reports.
Name string
// Options configures the engine. The zero value selects the engine's
// defaults (NewEngineWithOptions fills them in).
Options cypher.EngineOptions
}
EngineVariant names and builds one side of a differential comparison: a fresh engine over a fresh directed simple graph, configured by Options. Two variants that the engine guarantees are result-equivalent (e.g. the default planner vs the same planner with a physical optimisation disabled) MUST produce identical observable output on the same trace; any divergence is a regression.
The build is a factory so each differential run gets an isolated engine — the two variants never share graph state.
type ExecMode ¶
type ExecMode int
ExecMode selects which harness a Scenario drives. The deterministic modes are bit-reproducible from a seed and are the only modes trace recording, replay, and shrinking apply to; the concurrent and liveness modes use real goroutines whose interleaving is not seed-controlled and are convergence/leak-guarded rather than bit-replayable (see the package note on the hybrid determinism model).
const ( // ModeDeterministic is the single-goroutine, tick-driven engine-API safety // loop ([Simulator.Run]). It is fully bit-reproducible from a seed and is the // mode trace recording, scripted replay, and shrinking operate on. ModeDeterministic ExecMode = iota // ModeConcurrent drives N real client goroutines over the Bolt wire // ([RunConcurrent]). Interleaving is non-deterministic; correctness is the // eventual-consistency oracle plus goleak/no-panic. ModeConcurrent // ModeLiveness drives the two-phase safety->liveness flow ([RunLiveness]), // asserting convergence within a bounded budget plus a deadlock watchdog. ModeLiveness // ModeBulkVsOnline drives a concurrent bulk store-load alongside // transactional online writes (see [runBulkVsOnline]). ModeBulkVsOnline )
Execution modes.
func (ExecMode) Reproducible ¶
Reproducible reports whether a scenario in this mode is bit-reproducible from its seed and therefore eligible for trace recording, scripted replay, and shrinking. Only ModeDeterministic qualifies.
type FluentQueryConfig ¶ added in v0.12.0
type FluentQueryConfig struct {
// Seed is the master seed. Every sub-stream (probes, churn, prologue,
// workload, crash schedule, SimDisk) derives from it, so the whole run —
// including [FluentQueryEvidence.Digest] — is a pure function of this value.
Seed uint64
// MaxTicks bounds the deterministic loop.
MaxTicks int
// PrologueNodes is how many Persons (plus a KNOWS path over them) are
// created through the modelled templates before the loop. It is what makes
// the non-vacuity gates on a non-empty label and one-hop answer STRUCTURAL.
PrologueNodes int
// BatteryEvery and ChurnEvery are the in-loop cadences in ticks.
BatteryEvery int
ChurnEvery int
// Crash is the crash/recovery schedule. It is a field rather than a constant
// so a test can DISABLE the schedule and thereby guarantee that the forced
// crash arm ([fluentQueryForceCrash]) is the one that runs: with the schedule
// on, whether it fires inside a small budget is seed-dependent, and an arm
// that only some seeds reach is an arm no test can pin.
Crash CrashConfig
}
FluentQueryConfig parameterises a fluent-query run. The zero value is not usable; DefaultFluentQueryConfig fills in the short-layer budgets and [FluentQueryConfig.normalise] repairs any field a caller left at zero, so a test can override one field without restating the rest.
func DefaultFluentQueryConfig ¶ added in v0.12.0
func DefaultFluentQueryConfig(seed uint64) FluentQueryConfig
DefaultFluentQueryConfig returns the short-layer configuration for seed.
type FluentQueryEvidence ¶ added in v0.12.0
type FluentQueryEvidence struct {
// Batteries is how many times the full probe battery ran, and
// BatteriesAfterRecovery how many of those ran immediately after a crash
// recovery.
Batteries int
BatteriesAfterRecovery int
// ChurnCycles is how many delete-then-recreate pairs the churn phase
// committed.
ChurnCycles int
// MaxLiveNames / MaxOutTargets are the largest label and one-hop answers the
// model held at a battery, so "the probes were not comparing empty sets" is
// a measured fact.
MaxLiveNames int
MaxOutTargets int
// MaxTombstonedSlots is the largest number of TOMBSTONED ids
// [graph.Mapper.Walk] yielded at a battery, and MaxTombstoneCount the largest
// [lpg.Graph.TombstoneCount]. Both are DETERMINISTIC (the Mapper never
// forgets a slot and a tombstone is never cleared on this workload — Cypher
// CREATE cannot reuse a mapper key), and together they are the proof that
// query.pruneTombstones was load-bearing on the no-predicate seed path.
MaxTombstonedSlots int
MaxTombstoneCount int
// MaxTombstonedInLabelIndexObserved is the largest number of tombstoned ids
// the RAW :Person label bitmap carried at a battery. It is TELEMETRY, not a
// gate, and it is excluded from [FluentQueryEvidence.ReproducibleSummary]:
// the label-bitmap removal is deferred and applied by lpg's background
// vacuum, so this value depends on when that goroutine woke. MEASURED: the
// same seed in the same process gave 3 and 2. See the file header.
MaxTombstonedInLabelIndexObserved int
// StringRangeNonEmpty / StringRangeEmpty and IntRangeNonEmpty /
// IntRangeEmpty count how often each range probe's window matched something
// and nothing. Both directions are driven by CONSTRUCTION (one window is a
// live value, the other is out of the value space), so both must be
// positive.
StringRangeNonEmpty int
StringRangeEmpty int
IntRangeNonEmpty int
IntRangeEmpty int
// SeekEligible / SeekIneligible count the batteries at which every condition
// of query.trySeekProperty / query.trySeekRange held for the hash and the
// string btree respectively. An ineligible battery means the seek arm
// silently degraded to a scan and the seek-vs-scan clause compared a path
// with itself.
HashSeekEligible int
HashSeekIneligible int
BTreeSeekEligible int
BTreeSeekIneligible int
// GhostFixtures is how many times the constructed ghost-arc fixture ran, and
// GhostArcsSeen the total number of raw arcs into a tombstoned target it
// constructed. The fixture is the only place Out()'s ghost-arc prune branch
// is reachable (DETACH DELETE strips arcs on the live graph), so a zero here
// means that branch was never exercised.
GhostFixtures int
GhostArcsSeen int
// MixedKindProbes counts the mixed-kind probes — FLOAT64 bounds over an
// INT64-valued property — and MixedKindNonEmpty how many of them had a
// NON-EMPTY model answer. Since rmp #2600 this probe is a full three-way
// clause, not telemetry, so the second counter is the one that matters: two
// empty sets agree, and a probe that only ever compared empty sets would
// make the clause silent rather than satisfied. Both are gated in
// [FluentQueryProbes.Finish].
//
// MixedKindLastSeek and MixedKindLastOracle are the last probe's seek-arm
// and model cardinalities, kept so the run REPORTS the number the clause
// adjudicated instead of only whether it held.
MixedKindProbes int
MixedKindNonEmpty int
MixedKindLastSeek uint64
MixedKindLastOracle int
// EqMixed* are the same four quantities for the EQUALITY side of the same
// unification (rmp #2601): a FLOAT64 expected value in a [query.WithProperty]
// over an INT64-valued property. They are counted separately from the
// MixedKind* range counters on purpose — the two predicates take different
// index arms (a numeric equality is btree-served as a degenerate range, never
// hash-served) and one being exercised says nothing about the other, so a
// shared counter could not gate them independently.
//
// EqMixedNonEmpty is again the counter that matters: two empty answers agree
// whatever the comparison does, so a probe that only ever compared empty sets
// would make both the eq-mixed clause and the
// equality-vs-degenerate-range clause silent rather than satisfied.
EqMixedProbes int
EqMixedNonEmpty int
EqMixedLastSeek uint64
EqMixedLastOracle int
// NumericSeekEligible / NumericSeekIneligible count the batteries at which
// every condition of query.trySeekRange — and, since rmp #2601, of
// query.trySeekProperty for a numeric value — held for the internal numeric
// companion btree over (Person, age). An ineligible battery means every
// numeric arm degraded to a scan and the range-int / range-mixed / eq-mixed
// seek-vs-scan clauses compared one path with itself.
NumericSeekEligible int
NumericSeekIneligible int
// CSRRawArcs / CSRLiveArcs are the arc counts of the LAST battery's two CSR
// builds, and CSRGenerationsDiffered counts how many batteries — over the
// whole run, not just the last — saw the two builds disagree.
//
// The counter exists because "the two generations are equal on the live
// graph" is a claim about EVERY battery, and the last battery's pair cannot
// support it. It is expected to stay 0: DETACH DELETE strips the deleted
// node's arcs, so the tombstone-agnostic build has no ghost to keep. That is
// precisely why [fluentQueryGhostFixture] exists, and a non-zero value here
// would mean the live graph had started producing ghost arcs — worth knowing,
// which is why it is recorded rather than assumed.
CSRRawArcs uint64
CSRLiveArcs uint64
CSRGenerationsDiffered int
// Digest folds every probe's (tick, clause, cardinalities) triple. It is the
// scenario's reproducibility claim: same seed, same digest. It folds no
// NodeID and no mapper key, both of which come from a process-global counter
// and are not a function of the seed.
Digest uint64
}
FluentQueryEvidence is what the run MEASURED, handed back so a test asserts on numbers rather than on the mere absence of a violation, and so the report prints what actually happened.
Following the shape TxnOversizeEvidence and [indexDiversityEvidence] use, the checker itself IS the record: the non-vacuity gates in FluentQueryProbes.Finish read these very fields, so a test asserting on them cannot drift from what the gates enforce.
func (*FluentQueryEvidence) ReproducibleSummary ¶ added in v0.12.0
func (e *FluentQueryEvidence) ReproducibleSummary() string
ReproducibleSummary renders exactly the fields that are a pure function of the seed. It exists so the determinism test compares what the scenario CLAIMS is reproducible instead of the whole record: one field — FluentQueryEvidence.MaxTombstonedInLabelIndexObserved — depends on when lpg's background vacuum swept the deferred label-index removals, and asserting equality on it would be asserting a scheduler outcome.
func (*FluentQueryEvidence) String ¶ added in v0.12.0
func (e *FluentQueryEvidence) String() string
String renders the evidence for a report and for the run's own output.
type FluentQueryProbes ¶ added in v0.12.0
type FluentQueryProbes struct {
// contains filtered or unexported fields
}
FluentQueryProbes is the stateful three-way differential checker: it runs the probe battery on demand and accumulates the evidence its terminal non-vacuity gate reads.
Concurrency contract ¶
FluentQueryProbes is NOT safe for concurrent use. It draws from a Seed and issues engine reads that need a quiescent view of the graph, so it must be driven from the single simulation goroutine — the same contract CheckSearch carries and for the same reason.
func NewFluentQueryProbes ¶ added in v0.12.0
func NewFluentQueryProbes(seed *Seed) *FluentQueryProbes
NewFluentQueryProbes returns a probe battery drawing its windows from seed. The seed must be derived from the run seed and must NOT be the workload's, so the workload's op stream stays byte-identical.
func (*FluentQueryProbes) Check ¶ added in v0.12.0
func (p *FluentQueryProbes) Check( ctx context.Context, tick int64, g *lpg.Graph[string, float64], eng *EngineAdapter, o *GraphOracle, perturb fqPerturb, ) ([]Violation, error)
Check runs the whole probe battery once, at a quiescent instant, and returns every clause that failed (nil when all hold).
It must be called from the single simulation goroutine: it builds two CSR snapshots of the live adjacency and issues engine reads, both of which need a consistent, quiescent view — the same contract CheckSearch carries.
perturb is [fqPerturbNone] for every real run; the other values exist so a test can prove each clause family fires (see [fqPerturb]).
scatter the shared model/substrate/CSR setup across helpers that each need all three, and the probe list is the readable form of what the scenario covers.
one battery of independent probes; splitting it would
func (*FluentQueryProbes) Evidence ¶ added in v0.12.0
func (p *FluentQueryProbes) Evidence() *FluentQueryEvidence
Evidence returns the accumulating record. The pointer is owned by the probes and is live for the whole run.
func (*FluentQueryProbes) Finish ¶ added in v0.12.0
func (p *FluentQueryProbes) Finish(tick int64) []Violation
Finish is the terminal non-vacuity gate: it fails the run when the battery never reached the state that makes its clauses capable of failing.
Every gate here is STRUCTURAL — guaranteed by the prologue, by the churn phase, or by a window drawn outside the value space — rather than a rate the scheduler or the workload's draws might not deliver. That is the lesson #2587/#2596 taught this sprint: a threshold on a count nobody controls is a flake waiting to happen, and a coverage clause may only fail a run whose precondition was constructed.
type GenerationSwapConfig ¶ added in v0.12.0
type GenerationSwapConfig struct {
// Seed is the master seed. The plan is a pure function of it; see the
// determinism section in this file's header for exactly what that covers.
Seed uint64
// DrainTimeout is the deadline handed to PublishWithDrain in the
// drain-timeout arm. The arm's VERDICT does not depend on this value —
// the publisher holds a reference for the whole call, so the drain
// cannot complete at any timeout — only the arm's duration does.
DrainTimeout time.Duration
// JoinDeadline bounds every goroutine join. It is a HANG DETECTOR, not a
// performance threshold: a correct run joins in milliseconds, and a run
// that reaches this deadline is reported as a wedge rather than being
// waited on forever.
JoinDeadline time.Duration
// Readers is the number of concurrent reader goroutines.
Readers int
// MinOpsPerReader is the number of acquisitions a reader performs before
// it is allowed to notice that the publisher has finished.
MinOpsPerReader int
// MaxOpsPerReader caps a reader's acquisitions so the run's cost is
// bounded regardless of how the scheduler behaves.
MaxOpsPerReader int
}
GenerationSwapConfig parameterises one generation-swap run.
Concurrency contract ¶
A GenerationSwapConfig is a plain value read once by RunGenerationSwap before any goroutine spawns. It is not safe to mutate concurrently with a run, and a run mutates nothing in it.
func DefaultGenerationSwapConfig ¶ added in v0.12.0
func DefaultGenerationSwapConfig(seed uint64) GenerationSwapConfig
DefaultGenerationSwapConfig returns the short-layer configuration for a seed.
type GenerationSwapEvidence ¶ added in v0.12.0
type GenerationSwapEvidence struct {
// Findings are the clause firings observed in flight, unioned after the
// joins (whose happens-before edge publishes every worker's writes).
Findings []genSwapFinding
// PublishOps renders the plan's op for sequence 1..N. Seed-reproducible.
PublishOps []string
// ReaderSpans is per-reader telemetry. NOT seed-reproducible.
ReaderSpans []genSwapReaderSpan
// RefcountsAtRest is the refcount of every generation ever published,
// read after every worker was joined — the one point at which the value
// cannot move.
RefcountsAtRest []int64
CloseDuration time.Duration
DrainTimeout time.Duration
// Seed-reproducible plan facts.
Seed uint64
PlanDigest uint64
Generations int
Nodes int
DrainTimeoutAt int
// Config.
Readers int
// Interleaving-dependent counters (telemetry; the clauses that read them
// use structural minima only).
Acquires int64
NilAcquires int64
FullChecks int64
CheapChecks int64
// Publisher-side counts. Each is bounded by the plan, so a mismatch
// against the plan is itself a finding.
PlainPublishes int
DrainsCompleted int
DrainTimeouts int
HostageWasPrev int
Recycled int
GenerationsSeen int
DistinctGenerations int
// Terminal facts.
// Cancelled records that ctx cut the publish sequence short. A cancelled
// run carries no verdict; RunGenerationSwap returns a harness error for it.
Cancelled bool
PublisherFinished bool
ReadersJoined bool
CloseReturned bool
AcquireAfterCloseNil bool
CurrentAfterCloseNil bool
PublishAfterCloseErrClosed bool
DrainAfterCloseErrClosed bool
// Arm 2 — Close under live reader load.
Arm2Ran bool
Arm2ReadersJoined bool
Arm2CloseReturned bool
Arm2Readers int
Arm2PreCloseAcquires int64
Arm2NilAcquires int64
Arm2PublishOK int
Arm2PublishClosed int
// Arm2PublishBeforeClose is how many publishes ran SYNCHRONOUSLY before
// the Close goroutine spawned. Every one of them met an open publisher,
// so it is the structural floor for Arm2PublishOK.
Arm2PublishBeforeClose int
Arm2RefcountNonZero int
Arm2CapExhausted int
}
GenerationSwapEvidence is what one generation-swap run OBSERVED. It separates the seed-reproducible plan facts from the interleaving-dependent counters, because conflating them is how a DST report starts asserting against a number the scheduler chose.
Concurrency contract ¶
A GenerationSwapEvidence is populated by RunGenerationSwap on its own goroutine, after every worker has been joined, and is read-only thereafter. It is not safe to mutate concurrently.
func RunGenerationSwap ¶ added in v0.12.0
func RunGenerationSwap(ctx context.Context, cfg GenerationSwapConfig) (*GenerationSwapEvidence, error)
RunGenerationSwap performs one generation-swap run and returns what it observed, or a harness error.
Concurrency contract ¶
RunGenerationSwap spawns cfg.Readers reader goroutines plus one publisher goroutine per arm, every one with a lifecycle bounded by the plan and by ctx, and JOINS them all before returning — so no goroutine outlives the call and goleak has nothing to find. A join that reaches cfg.JoinDeadline is reported as a wedge; the run then STOPS rather than auditing refcounts a live goroutine could still be moving, and the leaked goroutine is left for goleak to report, loudly, as the defect it is.
func (*GenerationSwapEvidence) String ¶ added in v0.12.0
func (e *GenerationSwapEvidence) String() string
String renders the evidence for a report or a test log.
type GraphIOCSVArm ¶ added in v0.12.0
type GraphIOCSVArm struct {
Name string
Literal string
ImportErr string
Want map[string]int
Delimiter rune
Comment rune
Bytes int
Rows int
HasHeader bool
Sanitize bool
// ExpectRoundTrip is the arm's DECLARED outcome. It is false for exactly one
// arm — the sanitised one — because csv.Options.SanitizeFormulae documents
// that an apostrophe-prefixed cell no longer re-imports byte-identically.
// Declaring it makes that documented asymmetry an assertion instead of an
// unexplained failure.
ExpectRoundTrip bool
RoundTrips bool
}
GraphIOCSVArm is one point of the csv.Options space: a delimiter, a comment rune, a header flag and the formula-sanitisation flag, driven either by exporting the model (Literal == "") or by importing a hand-built document (Literal != "") that no exporter produces — the two-column weightless layout, a comment line, and a header row.
type GraphIOCancelObservation ¶ added in v0.12.0
type GraphIOCancelObservation struct {
Err error
Name string
Panicked string
Rows int
ControlRows int
Canceled bool
GraphNil bool
ControlEqual bool
}
GraphIOCancelObservation is one reader's mid-parse cancellation outcome, paired with the uncancelled control run over the same bytes.
type GraphIOCapObservation ¶ added in v0.12.0
type GraphIOCapObservation struct {
Err error
Name string
Panicked string
AllocBytes uint64
InputBytes int64
Matched bool
Ran bool
// Overran records that the probe's endless input hit its safety ceiling —
// which means the cap under test did NOT stop the reader.
Overran bool
}
GraphIOCapObservation is one cap probe's outcome.
type GraphIOGuardDecl ¶ added in v0.12.0
type GraphIOGuardDecl struct {
// Sentinel is the exported error the probe must match with errors.Is. A
// non-nil error of the wrong identity is a failure, not a pass.
Sentinel error
Name string
// Side is "reader" or "writer".
Side string
// Unreachable, when non-empty, is why no probe can provoke this cap. A
// declaration that carries it is not exempt from adjudication: the reason
// itself is asserted, so a change that made the cap reachable — or that
// made the stated reason false — fails the run.
Unreachable string
// AllocBoundNote states what the bound below actually proves, because it
// differs by probe class and a single number would otherwise imply more
// than it establishes.
AllocBoundNote string
// AllocBoundBytes is the ceiling on heap allocated while provoking the cap.
AllocBoundBytes uint64
}
GraphIOGuardDecl declares one defensive cap in graph/io: the sentinel it raises, WHICH SIDE of the codec it lives on, whether the simulator can provoke it at all, and the heap ceiling the provocation must respect.
The side matters and was mis-stated in the audit this file closes. Three of the caps — ErrPropertyValueTooLarge, ErrPropertyNestingTooDeep and ErrInvalidXMLChar — are raised by the ENCODERS, so no mutated export can ever reach them; they are provoked by handing the writer a hostile graph instead.
func GraphIOGuardDecls ¶ added in v0.12.0
func GraphIOGuardDecls() []GraphIOGuardDecl
GraphIOGuardDecls returns the declared cap surface of graph/io, as verified in source rather than from the audit list. Adding a sentinel to graph/io without adding it here is caught by the pinned-name assertion in the tests.
type GraphIOGuardResult ¶ added in v0.12.0
type GraphIOGuardResult struct {
Caps []GraphIOCapObservation
Cancels []GraphIOCancelObservation
// ListDepthBytes[i] is the encoded size of a list property nested i+1 deep.
// It is the evidence for the one cap declared unreachable: the wire re-
// escapes every level, so the series must roughly double, and a guard that
// only fires at depth 64 therefore needs an input no machine can hold.
ListDepthBytes []int
// DeepestRoundTrip is the deepest nesting that still round-trips through the
// reader, proving the depth guard is not firing early.
DeepestRoundTrip int
}
GraphIOGuardResult is one run of the whole guard battery.
func RunGraphIOGuards ¶ added in v0.12.0
func RunGraphIOGuards(ctx context.Context) (GraphIOGuardResult, error)
RunGraphIOGuards drives the crafted half of the surface: every defensive cap in graph/io, the reachability evidence for the one that cannot be provoked, and every *Ctx reader cancelled mid-parse against an uncancelled control.
CONCURRENCY CONTRACT — it measures heap through a process-global counter (see measureProcessAlloc), so it must be driven from a serialised arm and never from a scenario the swarm can schedule alongside others. It is called from its own test only.
type GraphIOMutation ¶ added in v0.12.0
type GraphIOMutation struct {
Format string
Kind string
Err string
Panicked string
Offset int
Source int
Changed bool
Typed bool
Equal bool
}
GraphIOMutation is one seed-derived corruption of an export, fed back through the matching importer.
func (*GraphIOMutation) Effective ¶ added in v0.12.0
func (m *GraphIOMutation) Effective() bool
Effective reports whether the mutation actually altered what the importer produced. A mutation that leaves the re-imported graph equal to the model and raises no error proves nothing, so the non-vacuity gate counts these.
type GraphIOPropsObservation ¶ added in v0.12.0
type GraphIOPropsObservation struct {
Mismatch string
KindsOnWire []string
Bytes int
Rows int
PropertyRecords int
LabelRecords int
Equal bool
}
GraphIOPropsObservation records the JSONL property-graph round-trip: the exported size, how many "property" records the encoder emitted, which kind tags appeared on the wire, and whether the re-imported graph reproduced the model.
type GraphIOSurfaceResult ¶ added in v0.12.0
type GraphIOSurfaceResult struct {
ModelTriples map[string]int
CSVTriples map[string]int
JSONLTriples map[string]int
ModelNodes []string
CSVNodes []string
JSONLNodes []string
CSVArms []GraphIOCSVArm
Mutations []GraphIOMutation
DOT dotDocument
Props GraphIOPropsObservation
Seed uint64
DOTBytes int
CSVBytes int
JSONLBytes int
// MutationAllocBytes is the heap allocated by the mutation sweep's REPLAY
// LOOP — its own exports sit outside the window — and MutationInputBytes
// the total input fed to that loop. The ratio is the bounded-allocation
// evidence for the readers, but ONLY when
// AllocMeasured is true: the figure comes from a process-global counter, so
// [RunGraphIOSurface] — which the swarm runs concurrently — does not measure
// it at all and leaves it zero (rmp #2553).
MutationAllocBytes uint64
MutationInputBytes int
// AllocMeasured reports whether MutationAllocBytes was actually measured,
// under the exclusivity measureProcessAlloc requires. A verdict over the
// ratio must refuse a result where this is false rather than read a zero as
// a pass.
AllocMeasured bool
// ExportStability maps an encoder to the number of repeat exports of the
// SAME graph that differed byte for byte from the first, out of
// graphIOStabilityRuns-1 repeats. Every encoder here must be
// byte-reproducible, and the verdict asserts it for all of them. Until
// rmp #2534 jsonl.WriteWithProps was exempt, because it emitted its
// property records in Go map order; it now emits them in ascending key order.
ExportStability map[string]int
}
GraphIOSurfaceResult is one seed's observation of the cross-format surface.
func RunGraphIOSurface ¶ added in v0.12.0
func RunGraphIOSurface(ctx context.Context, seed uint64) (GraphIOSurfaceResult, error)
RunGraphIOSurface drives the whole seed-dependent half of the graph/io completeness surface for one seed: the DOT / CSV / JSONL cross-format agreement, the JSONL property-graph round-trip, the csv.Options matrix, and the mutated-export sweep. It reports a harness error only when the harness itself could not proceed; every adjudication is left to CheckGraphIOSurface and CheckGraphIOSurfaceShape.
It performs NO allocation measurement, because it is reachable from the concurrently scheduled io-roundtrip-fault scenario; the bounded-allocation property is adjudicated by the serialised arm through [checkGraphIOMutationAlloc].
type GraphOracle ¶
type GraphOracle struct {
// contains filtered or unexported fields
}
GraphOracle is a correct-by-construction shadow model of what the graph must contain after a sequence of Phase-1 workload operations. It is deliberately minimal: it models only the five templates the workload emits and treats the Person name property as a logical key (the workload binds names uniquely and MERGE de-duplicates on it), which is what makes its predictions obviously correct without re-implementing the engine.
Concurrency contract ¶
GraphOracle is NOT safe for concurrent use; it is mutated and read from the single simulation goroutine.
func NewGraphOracle ¶
func NewGraphOracle() *GraphOracle
NewGraphOracle returns an empty oracle. Node ids start at 1 so zero can mean "no node".
func (*GraphOracle) ApplyCreate ¶
func (o *GraphOracle) ApplyCreate(cypher string, params map[string]any) OracleResult
ApplyCreate models the CREATE templates: a bare Person create ([tmplCreatePerson]) or a KNOWS edge between two existing Person nodes ([tmplCreateKnows]). It mutates the oracle to reflect the predicted committed state and returns the prediction.
func (*GraphOracle) ApplyDelete ¶
func (o *GraphOracle) ApplyDelete(cypher string, params map[string]any) OracleResult
ApplyDelete models [tmplDetachDelete]: DETACH DELETE removes the Person matched by name together with every incident edge. A miss is a committed zero-effect result.
func (*GraphOracle) ApplyMalformed ¶
func (o *GraphOracle) ApplyMalformed(cypher string, params map[string]any) OracleResult
ApplyMalformed models an intentionally ill-formed operation (OpMalformed from MalformedSender): the engine is expected to reject it with a typed error and apply no mutation, so the oracle records it as an expected-error no-op and changes no modelled state. Recording it keeps the operation history complete for replay/shrinking.
func (*GraphOracle) ApplyMatch ¶
func (o *GraphOracle) ApplyMatch(cypher string, params map[string]any) OracleResult
ApplyMatch models read-only and SET templates. Pure reads ([RETURN]/aggregate queries) never change state and commit trivially; the SET template ([tmplSetAge]) updates the matched node's age in place.
func (*GraphOracle) ApplyMerge ¶
func (o *GraphOracle) ApplyMerge(cypher string, params map[string]any) OracleResult
ApplyMerge models [tmplMergePerson]: MERGE by name creates the Person only when absent (setting created=true on the new one) and is a no-op otherwise.
func (*GraphOracle) BeginTx ¶ added in v0.12.0
func (o *GraphOracle) BeginTx() *OracleTx
BeginTx opens a per-transaction workspace over the oracle's committed state. The workspace captures its begin-snapshot NOW — statements observe that snapshot plus the transaction's own pending writes, never a commit that lands later (the engine's snapshot-isolation contract) — and publishes nothing until OracleTx.Commit.
func (*GraphOracle) DeletedKnowsInstances ¶ added in v0.12.0
func (o *GraphOracle) DeletedKnowsInstances() []KnowsInstance
DeletedKnowsInstances returns every KNOWS instance the model deleted through the standalone DELETE r template, in deletion order. The returned slice aliases the oracle's backing store and must not be mutated.
func (*GraphOracle) EdgeCount ¶
func (o *GraphOracle) EdgeCount() int
EdgeCount returns the number of edges the oracle currently models.
func (*GraphOracle) ExistenceOnEmail ¶ added in v0.6.0
func (o *GraphOracle) ExistenceOnEmail() bool
ExistenceOnEmail reports whether the oracle models an active NOT NULL (Acct, email) existence constraint.
func (*GraphOracle) HasEdge ¶
func (o *GraphOracle) HasEdge(src, dst uint64, label string) bool
HasEdge reports whether the oracle models a directed edge of the given label between src and dst.
func (*GraphOracle) HasKnowsByName ¶ added in v0.6.0
func (o *GraphOracle) HasKnowsByName(a, b string) bool
HasKnowsByName reports whether the oracle models a KNOWS edge between the Person nodes named a and b. The edge-property writer uses it to avoid creating a duplicate edge (a simple-graph no-op that would desync edge properties).
func (*GraphOracle) HasNode ¶
func (o *GraphOracle) HasNode(id uint64) bool
HasNode reports whether the oracle models a node with the given id.
func (*GraphOracle) HasPersonName ¶ added in v0.12.0
func (o *GraphOracle) HasPersonName(name string) bool
HasPersonName reports whether the committed model currently holds a Person node of the given name. The multi-session isolation checkers use it to adjudicate cross-transaction read-your-own-writes probes: a name the session committed is expected to be visible to its next transaction exactly while the committed model still holds it.
func (*GraphOracle) KnowsEdgesByName ¶ added in v0.6.0
func (o *GraphOracle) KnowsEdgesByName() []EdgeByName
KnowsEdgesByName returns every modelled KNOWS edge as (srcName, dstName, properties) triples in deterministic order, for the edge-property checker.
func (*GraphOracle) KnowsInstancesByEID ¶ added in v0.12.0
func (o *GraphOracle) KnowsInstancesByEID() []KnowsInstance
KnowsInstancesByEID returns every modelled KNOWS edge instance that carries a non-zero eid, ascending by eid. The deterministic order is load-bearing: the edge-properties writer indexes into this slice with seed-derived integers, so a map-range order would break reproducibility. The returned slice is freshly allocated and owned by the caller.
func (*GraphOracle) NodeCount ¶
func (o *GraphOracle) NodeCount() int
NodeCount returns the number of nodes the oracle currently models.
func (*GraphOracle) NodeNames ¶
func (o *GraphOracle) NodeNames() []string
NodeNames returns the Person names currently modelled, in ascending sorted order. The deterministic order is load-bearing: actors index into this slice with seed-derived integers, so a non-deterministic (map-range) order would make the op stream depend on Go's randomised map iteration and break reproducibility. The returned slice is freshly allocated and owned by the caller.
func (*GraphOracle) Ops ¶
func (o *GraphOracle) Ops() []OracleOp
Ops returns the recorded operation history (for replay and Phase-4 shrinking). The returned slice aliases the oracle's backing store and must not be mutated.
func (*GraphOracle) SetExistenceOnEmail ¶ added in v0.6.0
func (o *GraphOracle) SetExistenceOnEmail(active bool)
SetExistenceOnEmail declares (or clears) an active NOT NULL (Acct, email) existence constraint in the model, so the oracle predicts the engine REJECTS an email-omitting Acct CREATE. It is called by the existence scenario after it creates the constraint in the engine; the oracle keeps modelling it across a crash so the recovered engine must still enforce it (#1754).
func (*GraphOracle) SetUniqueOnName ¶ added in v0.6.0
func (o *GraphOracle) SetUniqueOnName(active bool)
SetUniqueOnName declares (or clears) an active UNIQUE constraint on (Person, name) in the model, so the oracle predicts the engine will REJECT a CREATE of a duplicate name. It is called by the constraint-enforcement scenario after it creates the constraint in the engine, and again after a crash/recovery to assert the constraint is still modelled as enforced.
func (*GraphOracle) String ¶
func (o *GraphOracle) String() string
String renders a compact summary of the oracle state for inclusion in a failure report.
func (*GraphOracle) TypedIDs ¶ added in v0.6.0
func (o *GraphOracle) TypedIDs() []int64
TypedIDs returns the ids of every modelled Typed node in ascending order (deterministic, for reproducible checker iteration).
func (*GraphOracle) TypedNode ¶ added in v0.6.0
func (o *GraphOracle) TypedNode(id int64) (map[string]any, bool)
TypedNode returns the modelled property map for the Typed node with the given id, and whether it exists. The returned map is the oracle's own and must not be mutated.
func (*GraphOracle) UniqueOnName ¶ added in v0.6.0
func (o *GraphOracle) UniqueOnName() bool
UniqueOnName reports whether the oracle models an active UNIQUE (Person, name) constraint.
type GroupCommitConfig ¶ added in v0.12.0
type GroupCommitConfig struct {
// Seed is the master seed for the workload and the disk sub-stream.
Seed uint64
// Connections is how many concurrent Bolt writer connections commit. The
// coverage gate requires at least [groupCommitMinCommitters]; the control arm
// deliberately sets 1 to prove the gate fires.
Connections int
// OpsPerConn is how many ops each connection issues (< 1 normalises to
// groupCommitDefaultOps).
OpsPerConn int
}
GroupCommitConfig parameterises a coalescing run.
type GroupCommitEvidence ¶ added in v0.12.0
type GroupCommitEvidence struct {
// Committers is the concurrency the run actually drove.
Committers int
// Coalesced is the follower count: SyncGroup calls that returned durable
// without an fsync of their own.
Coalesced uint64
// Leaders is the number of completed group fsync rounds.
Leaders uint64
// Errors is the number of SyncGroup calls that failed.
Errors uint64
// Acked is how many commits the Bolt clients saw acknowledged, which is the
// independent, engine-side count of durable commits — deliberately NOT
// derived from the same metrics the gate reads.
Acked int
}
GroupCommitEvidence is what a coalescing run OBSERVED. It holds measurements and no verdict, so a test can log what happened and the adjudicators can work on numbers rather than on a claim.
func RunGroupCommitCoalescing ¶ added in v0.12.0
func RunGroupCommitCoalescing(ctx context.Context, cfg GroupCommitConfig) (GroupCommitEvidence, error)
RunGroupCommitCoalescing drives cfg.Connections concurrent Bolt writer connections through a real WAL-backed store on a SimDisk and returns what the group-commit counters observed.
It installs the recording metrics backend for the duration (the same test-side sink MetricsOracle uses) and restores the no-op default before returning. Because that sink is GLOBAL it must run SERIALLY: the caller must not run concurrent metrics-emitting work.
func (GroupCommitEvidence) FollowerRate ¶ added in v0.12.0
func (e GroupCommitEvidence) FollowerRate() float64
FollowerRate is the fraction of observed rounds that took the follower fast path. It is reported for evidence, never asserted against a threshold: the rate is a scheduling outcome and pinning it would be flaky.
func (GroupCommitEvidence) Rounds ¶ added in v0.12.0
func (e GroupCommitEvidence) Rounds() uint64
Rounds is the total number of SyncGroup calls the run observed: every call either coalesced onto another round, led one, or failed.
func (GroupCommitEvidence) String ¶ added in v0.12.0
func (e GroupCommitEvidence) String() string
String renders the evidence for a failure message or a test log.
type GroupCommitFailAllResult ¶ added in v0.12.0
type GroupCommitFailAllResult struct {
// Members is how many committers were in the group.
Members int
// Acked is how many members' SyncGroup returned nil. It MUST be zero: the
// shared fsync failed, so no member may believe it committed.
Acked int
// Failed is how many members received an error.
Failed int
// DurabilityClass is how many of those errors satisfy
// errors.Is(err, wal.ErrDurabilityFailed) — the class a member needs to tell
// a durability fail-stop from a conflict of its own.
DurabilityClass int
// Leaders is store.wal.SyncGroup.leader over the arm: completed leader
// rounds. A failed round completes none, so a genuine single-group fail-all
// leaves this at zero.
Leaders uint64
// PoisonedRounds is wal.Stats.SyncFailed: how many group rounds ended in a
// poison. Exactly ONE, for a group of many members, is the direct statement
// that they shared a single leader's fsync rather than each paying its own.
PoisonedRounds uint64
// FsyncAttempts is the number of Sync calls the SimDisk saw after arming,
// counted independently of the WAL's own counters. It is expected to be
// groupCommitFailAllFsyncs — see that constant for the decomposition, which
// was MEASURED rather than assumed.
FsyncAttempts int64
// GateFired reports that the rendezvous actually held the leader inside its
// fsync. Without it the members would have serialised and there would have
// been no group to fail.
GateFired bool
// PriorDurable / GroupDurable are what recovery found afterwards: the frames
// of the commit acknowledged BEFORE the group (which must survive) and of the
// group itself (which must be gone).
PriorDurable, GroupDurable int
}
GroupCommitFailAllResult is what the fail-all arm observed: the per-member outcome of one commit group whose shared fsync failed.
func RunGroupCommitFailAll ¶ added in v0.12.0
func RunGroupCommitFailAll(ctx context.Context, seed uint64) (GroupCommitFailAllResult, error)
RunGroupCommitFailAll constructs one genuine multi-member commit group whose shared fsync fails, and reports what every member saw.
The protocol is deterministic by construction:
- One frame is appended and synced successfully — the PRIOR commit, which recovery must still find afterwards.
- Every member appends its frame, so all of them are buffered before any SyncGroup runs and a single leader's flush covers them all.
- Member 0 calls SyncGroup and becomes the leader. Its fsync is both GATED (so it stays inside the call) and armed to FAIL.
- Once the gate reports the leader is parked inside the fsync, members 1..n-1 call SyncGroup and find a leader active.
- The gate is released, the leader's fsync returns the fault, the writer poisons and discards the whole un-synced suffix, and every member fails.
Step 3 is why SimDisk.ArmSyncGateAt exists: against an in-memory disk the leader's fsync window is otherwise far too short for a follower to reach it.
func (GroupCommitFailAllResult) String ¶ added in v0.12.0
func (r GroupCommitFailAllResult) String() string
String renders the result for a failure message or a test log.
type HelperOpResult ¶
HelperOpResult is the prior release's observable outcome for one op: whether it committed and a canonical, order-independent signature of its result rows.
type HelperRunResult ¶
type HelperRunResult struct {
// CheckpointErr is the prior release's own checkpoint failure message, empty
// when it succeeded or was never attempted.
CheckpointErr string
Ops []HelperOpResult
Nodes int64
Edges int64
// Checkpoint reports that the image at dir carries a SNAPSHOT DIRECTORY the
// prior release published, not merely a WAL (rmp #2477).
Checkpoint bool
}
HelperRunResult is the full outcome of driving an op stream through the prior release: the per-op results in order, the prior engine's final counts, and whether the prior release published a checkpoint over the image.
type HonestReader ¶
type HonestReader struct{}
HonestReader emits valid read-only operations: projections, a relationship join, a filtered aggregate, and a bounded variable-length path query. It never mutates the graph, so its operations always commit with no effect on the oracle state.
func (HonestReader) NextOp ¶
func (HonestReader) NextOp(seed *Seed, _ *GraphOracle) Op
NextOp picks one read template at random and binds its parameters from the seed.
type HonestWriter ¶
type HonestWriter struct{}
HonestWriter emits valid mutating operations: it creates Person nodes, links existing ones with KNOWS edges, updates ages, merges by name, and detaches and deletes. Every edge, SET, and DELETE references a node the oracle already knows about, so the writer never emits a statement that the engine would reject on well-formedness grounds.
func (HonestWriter) NextOp ¶
func (w HonestWriter) NextOp(seed *Seed, oracle *GraphOracle) Op
NextOp chooses a mutating operation. When the oracle is empty it can only create (there is nothing to reference yet); otherwise it picks among create, link, update, merge, and delete with fixed seed-driven weights.
type IndexIntersectProbes ¶ added in v0.12.0
type IndexIntersectProbes struct {
// contains filtered or unexported fields
}
IndexIntersectProbes is the intersect-planner probe set: a seed-drawn two-predicate conjunction over the two BTREE-indexed properties of :Person, in both its unbounded-above ([boundStringRange.RangeCountFrom]) and its bounded ([boundStringRange.RangeCount]) string spelling, each result-verified against an independent full-scan reference and each required to compose in the physical plan — plus a single-property control that must seek through the SAME operator WITHOUT composing, so "the marker is present" cannot pass for "an index was used".
It is stateful so IndexIntersectProbes.Finish can assert non-vacuity over the whole run: a run in which nothing ever composed, or in which every composed arm returned zero rows, proved nothing about the intersection path.
Concurrency contract ¶
IndexIntersectProbes is NOT safe for concurrent use; the simulator drives it from the single simulation goroutine.
func NewIndexIntersectProbes ¶ added in v0.12.0
func NewIndexIntersectProbes(seed *Seed, bulk int) *IndexIntersectProbes
NewIndexIntersectProbes draws the two conjunct windows from seed.
bulk is the size of the scenario's bulk load, used only to reject a fixture too small for the planner's population floor (cypher.rangeSeekMinLabelPopulation is 1024): below it every conjunct is declined whatever its selectivity, so every arm would assert a composition that cannot happen. A caller passing less than 1100 is a programmer error and panics here rather than producing a scenario that fails for the wrong reason.
func (*IndexIntersectProbes) Check ¶ added in v0.12.0
func (k *IndexIntersectProbes) Check(tick int64, engine PlanEngine) []Violation
Check runs every arm through engine and returns one violation per divergence:
- a composed arm whose id-multiset differs from the independent full-scan reference is a ViolationACIDConsistency: the intersected read disagrees with the base data;
- a parameterised spelling that differs from its literal twin is a ViolationACIDConsistency (the rmp #2414 family, on results);
- a composed arm whose plan does not carry [intersectComposedMarker] is a ViolationOracleDeviation: the shape the probe exists to drive was not planned, so the arm proved nothing about the intersection path;
- a composed arm whose plan does not carry the expected string bound is a ViolationOracleDeviation: the composition happened but through the other budgeted count, so the branch this arm targets was not the one that ran;
- a composed arm that lost its residual Filter is a ViolationACIDConsistency: each part is only a superset of its conjunct (#F-EXEC1), so the exact predicate must still be re-applied per row;
- the control arm composing, or not reaching the range-scan operator at all, is a ViolationOracleDeviation: without it a composed marker could not be distinguished from "any index was used".
A query that errors is a ViolationOracleDeviation. Every message renders sorted ids, so it is deterministic.
func (*IndexIntersectProbes) Counts ¶ added in v0.12.0
func (k *IndexIntersectProbes) Counts() (composed, withRows, soloSeeks int)
Counts reports what the run observed: how many composed plans were seen, how many composed arms returned at least one row, and how many single-property controls seeked WITHOUT composing. It is the same state IndexIntersectProbes.Finish adjudicates, exposed so a test logs and asserts the measured numbers instead of inferring them from the gate's silence.
func (*IndexIntersectProbes) Finish ¶ added in v0.12.0
func (k *IndexIntersectProbes) Finish(tick int64) []Violation
Finish asserts non-vacuity over the whole run and must be called once, after the terminal IndexIntersectProbes.Check:
- at least one arm must have composed, or the intersection path never ran;
- at least one composed arm must have returned a row, or every multiset comparison was between empty sets;
- at least one control arm must have seeked without composing, or nothing established that the composed marker discriminates a composition.
Each is reported as a ViolationOracleDeviation: the engine is not wrong, the RUN failed to exercise what it claims to cover.
type IndexSeekResults ¶ added in v0.12.0
type IndexSeekResults struct {
// contains filtered or unexported fields
}
IndexSeekResults is the seek-result diversity checker: a fixed, seed-drawn set of range, prefix, and IN-shaped probes over the index-diversity scenario's indexed properties, each result-verified against an independent full-scan reference in both literal and parameterised spellings. It is stateful so IndexSeekResults.Finish can assert non-vacuity over the whole run: at least one probe arm must have returned at least one row at least once, otherwise every comparison was between empty sets and proved nothing.
Concurrency contract ¶
IndexSeekResults is NOT safe for concurrent use; the simulator drives it from the single simulation goroutine.
func NewIndexSeekResults ¶ added in v0.12.0
func NewIndexSeekResults(seed *Seed, bulk int) *IndexSeekResults
NewIndexSeekResults draws the probe windows from seed. bulk is the exclusive upper bound of the bulk "p<i>" name space the name probes may draw from (the index-diversity scenario passes [indexDiversityBulk]; the churn loop never deletes bulk names, so every drawn name stays resolvable for the whole run). bulk must be at least 20; a smaller fixture is a programmer error and panics in the seed draw.
func (*IndexSeekResults) Check ¶ added in v0.12.0
func (k *IndexSeekResults) Check(tick int64, engine Engine) []Violation
Check runs every probe arm through engine and returns a violation for each divergence found:
- a probe arm (literal spelling) whose id-multiset differs from the independent full-scan reference is a ViolationACIDConsistency — the index-served answer disagrees with the base data;
- a parameterised spelling whose id-multiset differs from its literal twin is a ViolationACIDConsistency (the rmp #2414 family, on results);
- the IN-list and UNWIND spellings of the same predicate disagreeing with each other is a ViolationACIDConsistency (per rmp #2183 they may plan differently — seek-set vs label scan — but must answer identically);
- count arms (bounded range, half-open range, prefix) must equal the reference cardinality in both spellings, driving the dedicated range-count paths.
Probe failures (a query erroring) are ViolationOracleDeviation. Result ids are sorted before rendering so messages are deterministic.
func (*IndexSeekResults) Finish ¶ added in v0.12.0
func (k *IndexSeekResults) Finish(tick int64) []Violation
Finish asserts non-vacuity over the whole run: at least one probe arm must have returned at least one row in at least one IndexSeekResults.Check invocation. A run in which every comparison was between empty multisets proved nothing about the index read paths and is reported as a ViolationOracleDeviation rather than passing silently. Call it once, after the terminal Check.
type IndexSpec ¶
type IndexSpec struct {
// Label is the node label the index is declared on (e.g. "Person").
Label string
// Property is the property key the index covers (e.g. "name").
Property string
// Numeric declares the indexed property as integer-valued, so the
// consistency check groups the full scan by integer value and binds the
// index-seek probe with an integer parameter. The default (false) treats the
// property as string-valued. It lets the checker cover numeric (btree) indexes
// alongside string ones; the seek-vs-scan invariant is identical for either.
Numeric bool
}
IndexSpec declares one secondary index the simulator created during a run, by the (Label, Property) it covers. The index-consistency checker cross-checks each declared spec against the engine's base data. Specs are declared by the scenario (the simulator does not introspect (label, property) from the engine index manager, which exposes only opaque names), so the registry of specs is the authoritative set the checker walks.
type InvariantChecker ¶
type InvariantChecker struct {
// contains filtered or unexported fields
}
InvariantChecker compares the engine against the oracle after operations and accumulates any Violation it finds. It samples a bounded, seed-driven subset of oracle state per call so its cost stays bounded on large graphs.
Concurrency contract ¶
InvariantChecker is NOT safe for concurrent use; it is driven from the single simulation goroutine.
func NewInvariantChecker ¶
func NewInvariantChecker(seed *Seed) *InvariantChecker
NewInvariantChecker returns a checker whose sampling draws from seed.
func (*InvariantChecker) Check ¶
func (c *InvariantChecker) Check(tick int64, oracle *GraphOracle, engine Engine) []Violation
Check verifies the engine against the oracle at the given tick and returns any newly-found violations (also accumulated internally). It performs:
- node- and edge-count parity (oracle vs engine);
- sampled oracle-node existence in the engine (no missing nodes);
- sampled oracle-edge existence in the engine (no ghost or missing edges).
Each check that fails appends a typed Violation; a clean pass returns nil.
func (*InvariantChecker) CheckDurability ¶
func (c *InvariantChecker) CheckDurability(tick int64, oracle *GraphOracle, engine Engine) []Violation
CheckDurability verifies ACID Durability at a crash-recovery boundary: every operation the engine ACKed as committed before the crash (which the oracle models exactly, because [Simulator.applyToOracle] advances the oracle only on a committed write) must be present in the recovered engine, and nothing that was never committed may have leaked in as partial state. Unlike InvariantChecker.Check it scans the FULL oracle node and edge set, not a bounded sample, because a single dropped committed op is a durability violation that sampling could miss.
It performs:
- exact node- and edge-count parity (a recovered count below the oracle's means a committed op was lost — a Durability breach; a count above means uncommitted state leaked in — an Atomicity breach at the crash boundary);
- full-scan oracle-node presence (every committed node survived recovery);
- full-scan oracle-edge presence (every committed edge survived recovery).
Count mismatches are tagged ViolationACIDDurability; a missing node or edge is tagged ViolationACIDDurability (the committed datum did not survive). Each failing check appends a typed Violation; a clean pass returns nil.
func (*InvariantChecker) ChecksRun ¶
func (c *InvariantChecker) ChecksRun() int
ChecksRun reports how many times InvariantChecker.Check has executed since construction. It exposes the realised invariant-check cadence so callers can confirm, for a given CheckEvery, that the expected number of checks ran (including the simulator's terminal check).
func (*InvariantChecker) HasViolations ¶
func (c *InvariantChecker) HasViolations() bool
HasViolations reports whether any violation has been recorded.
func (*InvariantChecker) Violations ¶
func (c *InvariantChecker) Violations() []Violation
Violations returns all recorded violations. The returned slice aliases the checker's backing store and must not be mutated.
type KnowsInstance ¶ added in v0.12.0
KnowsInstance identifies one modelled KNOWS edge instance by its unique eid and its endpoint Person names — the coordinates the edge-properties writer targets ops with and the checker probes the engine with.
type LabelIndexScopedConfig ¶ added in v0.12.0
type LabelIndexScopedConfig struct {
// Seed drives the base bands, the sweep order and the membership probes.
Seed uint64
// Epochs is the number of relationship-sweep epochs. 0 uses liEpochs.
Epochs int
// UnionDraws is the number of Union subsets per shape. 0 uses liUnionDraws.
UnionDraws int
// Perturb is the deliberate corruption to apply, threaded as a parameter so
// no package-level variable carries it and two concurrent runs cannot see
// each other's.
Perturb liPerturb
}
LabelIndexScopedConfig parameterises one run.
func DefaultLabelIndexScopedConfig ¶ added in v0.12.0
func DefaultLabelIndexScopedConfig(seed uint64) LabelIndexScopedConfig
DefaultLabelIndexScopedConfig returns the configuration the catalogue runs.
type LabelIndexScopedEvidence ¶ added in v0.12.0
type LabelIndexScopedEvidence struct {
Cells []liCellEvidence
Corrupt []liCorruptEvidence
Scopes []liScopeEvidence
Boundary liBoundaryEvidence
Phantom liPhantomEvidence
// FirstMismatch localises the first membership disagreement of the whole run,
// so a failing report names an id rather than only a count.
FirstMismatch string
// Sweep aggregates.
Epochs int
RangeOps int
SingleOps int
Compares int
Mismatches int
FinalLabels int
FinalMembers int
PeakMembers int
// EmptiedLabels counts the isolated labels a RemoveRange drove from non-empty
// to empty, which is the path RemoveRange's godoc promises deletes the map
// entry.
EmptiedLabels int
// PromotedAfterAdd counts the epochs in which the accumulating label took
// individual Adds BEFORE its first AddRange of the epoch. It is a property of
// the OP STREAM, not an observation of the tier: which tier a set sits on is
// deliberately unobservable through this index's public surface (the whole
// point of the byte-identity claim), so this counter says what was DRIVEN and
// claims nothing about what the set became.
PromotedAfterAdd int
// Union arm.
UnionDraws int
UnionMismatches int
UnionMultiLabel int
UnionUnknownLabel int
UnionDuplicate int
UnionEmptyDraws int
// Round-trip arm.
RangeTier []liTierRow
DenseSmall liDenseSmallEvidence
RoundTrips int
RTContentMismatch int
// RTByteMismatch counts fixpoint failures: the image produced by the SECOND
// serialize must equal the one produced by the third. The FIRST cycle is
// deliberately not asserted — see [liDriveRoundTrip].
RTByteMismatch int
// FirstCycleStable records whether the very first round trip left the bytes
// alone. It is a MEASUREMENT, not a claim: a run-container label whose
// cardinality is at most smallSetMax is re-tiered on the way back in and
// legitimately changes size. [liDriveDenseSmall] pins that case directly.
FirstCycleStable bool
SerializeReruns int
SerializeUnstable int
TierChecks int
TierMismatch int
ImageBytes int
// ImageBytes2 is the size of the image the second serialize produced.
ImageBytes2 int
// ImageLabelCount is the labelCount the final image declares, and
// ModelLiveLabels how many labels the model says carry a member. The excess
// is PhantomExcess: entries the index holds for labels with nothing in them.
ImageLabelCount uint32
ModelLiveLabels int
PhantomExcess int
// Digest is an order-sensitive hash of every reproducible fact.
Digest uint64
Perturb liPerturb
}
LabelIndexScopedEvidence is everything one run measured.
func (*LabelIndexScopedEvidence) ReproducibleSummary ¶ added in v0.12.0
func (e *LabelIndexScopedEvidence) ReproducibleSummary() string
ReproducibleSummary renders the facts a determinism comparison uses. Every measurement this scenario takes is a pure function of the seed, so this is the digest plus the aggregates that would localise a divergence.
func (*LabelIndexScopedEvidence) String ¶ added in v0.12.0
func (e *LabelIndexScopedEvidence) String() string
String renders the evidence for a report and a log line.
type LegacyCSRVerdict ¶ added in v0.12.0
type LegacyCSRVerdict struct {
// Reason describes the outcome in the terms of the guard that produced it.
Reason string
// Accepted reports that neither guard refused.
Accepted bool
// WidthGuardRefused reports the ApplyCSRToGraph guard: that build compared
// the header width against csrWeightSize[W](), which returned only 0, 1, 2, 4
// or 8, and returned ErrCorrupted on any mismatch.
WidthGuardRefused bool
// ExtentGuardRefused reports the readCSRLimited guard: that build computed
// the dense weights extent as width*nEdges and rejected it against the
// manifest-recorded file size.
ExtentGuardRefused bool
}
LegacyCSRVerdict is what a csr.bin reader from BEFORE rmp #2526 would have done with a header, and — when it refuses — which of that release's two independent guards refused it.
func LegacyCSRReaderVerdict ¶ added in v0.12.0
func LegacyCSRReaderVerdict(h CSRHeaderFacts, fileSize int64, nativeWidth uint8) LegacyCSRVerdict
LegacyCSRReaderVerdict applies the PRE-rmp-#2526 reader's decision rule to a header. nativeWidth is the width that build's csrWeightSize[W]() returned for the weight type it was compiled with — 8 for the float64 stores every release of this module has shipped, 4 for float32, and so on.
The rule is restated from the two guards that release contained, and both are evaluated so a refusal reports whether it was over-determined:
- The EXTENT guard, in readCSRLimited. The dense layout implies exactly width*nEdges weight bytes, which cannot exceed the file the manifest describes. The rmp #2526 sentinel is 0xFF, so a new artefact claims 255 bytes per edge and blows this bound on any file with edges.
- The WIDTH guard, in ApplyCSRToGraph. The header width had to equal the compiled-in native width exactly. 0xFF is not a width csrWeightSize could return for any W, so it can never match.
A zero fileSize means "unknown", which disables the extent guard alone — the width guard still decides. That mirrors the real reader, whose extent bound is only as tight as the manifest size it was given.
type LivenessConfig ¶
type LivenessConfig struct {
// Seed controls the honest convergence workload.
Seed uint64
// Connections / OpsPerConn bound the honest convergence workload (same
// bounded-resource discipline as the concurrent harness).
Connections int
OpsPerConn int
// ConvergeBudget is the maximum simulated time the harness waits for the
// pending() predicate to reach null. Exceeding it is a liveness failure
// (the system did not converge). Measured on the injected clock.
ConvergeBudget time.Duration
// PollStep is the simulated interval between pending() samples. Values <= 0
// default to ConvergeBudget/100 (at least 1ms).
PollStep time.Duration
// NoProgressGrace is the number of consecutive non-improving samples the
// watchdog tolerates before declaring a resonance (no progress despite
// pending work). Values <= 0 default to 8.
NoProgressGrace int
}
LivenessConfig parameterises the liveness phase. It runs AFTER the safety phase, with all fault injection healed/disabled: only honest actors run, and the harness asserts the system CONVERGES (drains to quiescence) within a bounded tick budget. A watchdog catches the resonance class — pending work that never makes progress (deadlock/livelock).
type LivenessOutcome ¶
type LivenessOutcome struct {
Seed uint64
Converged bool
Resonance bool
Ticks int
FinalPending PendingState
}
LivenessOutcome is the result of the liveness phase. Converged is true when the system reached quiescence within the budget. When false, FinalPending and Resonance describe why: Resonance == true means the watchdog detected no-progress-despite-pending (deadlock/livelock); Resonance == false means the budget simply elapsed while still making progress (under-provisioned budget).
func RunLiveness ¶
func RunLiveness(ctx context.Context, srv *SimServer, clk clock.Clock, cfg LivenessConfig) (LivenessOutcome, error)
RunLiveness runs the LIVENESS phase against srv: it executes a bounded honest convergence workload (no faults) and then polls the pending() predicate on the injected clock until the system is quiescent or the budget elapses, with a watchdog for the resonance (no-progress) class.
The honest workload is run to completion first (every connection goroutine is joined), so at the polling stage the only residual pending work is structural (an oracle divergence the engine never reconciled, or a leaked goroutine). A healthy system is quiescent on the first poll; a system with a real liveness bug (a stuck writer, a leaked stream goroutine, a permanent oracle divergence) never reaches quiescence and the budget/watchdog fires.
Determinism ¶
The convergence workload follows the concurrent (non-bit-reproducible) model, but the polling and watchdog are driven by the injected clock.Clock, so the liveness decision (converged vs budget-exceeded vs resonance) is deterministic under a clock.Fake for a given pending() trajectory.
func (LivenessOutcome) Report ¶
func (o LivenessOutcome) Report() string
Report renders a VOPR-style liveness failure report with the seed and the pending-work dump, mirroring the safety phase's SimReport.String. It returns the empty string for a converged outcome.
type MVCCContentionConfig ¶ added in v0.12.0
type MVCCContentionConfig struct {
// Seed is the master seed; scheduling and per-session choices derive from it.
Seed uint64
// Ticks is the number of scheduler steps (one session step per tick).
Ticks int
// Sessions is the number of concurrently colliding sessions (< 2
// normalises to 2 — below that nothing contends).
Sessions int
// Counters is the size of the SHARED counter key space (< 1 normalises to
// 2). Fewer counters means more collisions.
Counters int
// OnStep is the observation hook, as in [MVCCSessionsConfig.OnStep].
OnStep func(tick, session int, what string)
// OnQuiesce, when set, is called once at the DRAIN POINT: every open
// transaction has been rolled back and no write is in flight, but the store
// is still open and the adjudication has not yet run.
//
// It exists for the MVCC-substrate telemetry oracle (rmp #2470), which must
// read [lpg.MVCCStats] at a point where in-flight commits returning to zero
// is a fair question. It is deliberately NOT allowed to influence the run:
// it is called after the last transaction and its observations go to the
// caller, never into [MVCCContentionResult], because substrate telemetry is
// scheduling-dependent and the result is compared byte for byte by the
// determinism gate.
OnQuiesce func(st *SimStore)
}
MVCCContentionConfig parameterises a deterministic contended-counter run.
type MVCCContentionResult ¶ added in v0.12.0
type MVCCContentionResult struct {
// Violations holds the adjudication findings; empty on a clean run.
Violations []Violation
// AckedIncrements is the number of COMMIT-acknowledged increments per
// counter — the value each counter must hold at the end.
AckedIncrements []int
Seed uint64
Sessions int
Counters int
// TxCommitted / TxConflicted count finished write transactions by outcome.
TxCommitted int
TxConflicted int
// TypedConflicts counts refusals that matched cypher.ErrSerializationConflict
// via errors.Is. Every refusal must be typed, so on a clean run
// TypedConflicts == TxConflicted; an untyped refusal is a hard error, not a
// counter.
TypedConflicts int
}
MVCCContentionResult summarises a contended-counter run; a pure function of the seed (the determinism gate compares results byte for byte).
func RunMVCCContention ¶ added in v0.12.0
func RunMVCCContention(ctx context.Context, cfg MVCCContentionConfig) (*MVCCContentionResult, error)
RunMVCCContention executes the deterministic contended-counter scenario and returns its result. Determinism contract as RunMVCCSessions: same config, identical result.
Concurrency contract ¶
Runs entirely on the calling goroutine; "concurrent" transactions are interleaved deterministically, never parallel.
func (*MVCCContentionResult) Clean ¶ added in v0.12.0
func (r *MVCCContentionResult) Clean() bool
Clean reports whether the run finished with no violations.
type MVCCSessionsConfig ¶ added in v0.12.0
type MVCCSessionsConfig struct {
// Seed is the master seed; scheduling, per-session op streams, transaction
// shapes, and terminal choices are all derived from it.
Seed uint64
// Ticks is the number of scheduler steps. Each tick advances exactly one
// session by one step (BEGIN, one statement, or COMMIT/ROLLBACK).
Ticks int
// Sessions is the number of logical sessions (each a [cypher.Session] with
// its own transaction lifecycle). Values < 2 are normalised to 2 — below
// that no transactions can overlap and the mode loses its purpose.
Sessions int
// MinTxOps and MaxTxOps bound the statements per transaction, drawn per
// transaction from the session's sub-seed. Non-positive values are
// normalised to 1..4.
MinTxOps, MaxTxOps int
// ReadTxWeight is the probability (clamped to [0,1]) that a new transaction
// is read-only ([cypher.Session.BeginReadTx]).
ReadTxWeight float64
// RollbackWeight is the probability (clamped to [0,1]) that a finished
// write transaction is rolled back instead of committed, so the abort path
// is exercised alongside the commit path.
RollbackWeight float64
// CheckEvery is the oracle/engine parity-check cadence in ticks; values
// <= 0 are normalised to 1. Parity holds at every tick boundary because
// the oracle folds a workspace exactly when the engine acknowledges the
// matching COMMIT.
CheckEvery int
// Crash opts into deterministic crash injection (rmp #2438): at
// seed-scheduled ticks the SimDisk takes a HOST crash (every byte and every
// dirent no successful fsync covered is lost — [SimDisk.CrashHost]), the
// store reopens through real WAL recovery, every open
// transaction dies unacknowledged, and the recovered state is adjudicated
// at TRANSACTION granularity against the folded oracle. The zero value
// disables crashes — the safe default, byte-identical to a pre-crash run.
//
// Crashes land BETWEEN scheduler steps (the mode is single-goroutine), so
// they always land with transactions OPEN mid-flight; intra-commit crash
// points remain the internal/crashpoint battery's domain.
Crash CrashConfig
// OnStep, when non-nil, is called synchronously after every scheduler step
// with the tick, the session advanced, and a one-line description of what
// happened. Like [Config.OnOp] it is an observation hook: it must not
// mutate state or draw randomness, or reproducibility breaks.
OnStep func(tick, session int, what string)
}
MVCCSessionsConfig parameterises a deterministic multi-session run. Every field is bounded and the whole run is a pure function of Seed.
type MVCCSessionsResult ¶ added in v0.12.0
type MVCCSessionsResult struct {
// Violations holds the first invariant violations detected, if any; empty
// on a clean run.
Violations []Violation
// FoldErrors records every oracle fold refusal: the engine acknowledged a
// COMMIT whose decided effects no longer fold over the committed model.
// Any entry is an isolation finding (or a harness bug) — never noise.
FoldErrors []string
Seed uint64
Sessions int
Ticks int
// Statements counts the statements executed inside transactions (BEGIN and
// COMMIT/ROLLBACK excluded).
Statements int
// TxCommitted / TxRolledBack / TxConflicted / TxReadOnly count finished
// transactions by outcome. A conflicted transaction is one the engine
// refused with [mvcc.ErrSerializationConflict] — at a statement or at
// COMMIT — and whose workspace was discarded.
TxCommitted int
TxRolledBack int
TxConflicted int
TxReadOnly int
// StabilityProbes counts adjudicated count reads inside read-only
// transactions (snapshot stability, rmp #2436); StabilityInterleaved
// counts the subset with at least one commit folded since the
// transaction's BEGIN — the probes that prove stability holds even when
// other sessions commit in between. A run with StabilityInterleaved == 0
// never exercised the interesting case.
StabilityProbes, StabilityInterleaved int
// RYOWProbes counts in-transaction read-your-own-writes probes (after
// each write statement, plus the write-view count adjudications);
// RYOWCrossTx counts the session-level probes at BEGIN for a name the
// session committed earlier.
RYOWProbes, RYOWCrossTx int
// PairsCommitted counts committed invariant-bearing pairs (two nodes
// created by two statements of one transaction); PairProbes counts the
// atomic-visibility probes adjudicated against them.
PairsCommitted, PairProbes int
// TxDoomed counts write transactions observed under the doomed-tx
// contract (rmp #2354): an own write read back wrong — a refused void
// write — and the transaction then ended in a serialization conflict, as
// the contract requires. A suspect that instead COMMITs cleanly is a
// Violation, never counted here.
TxDoomed int
// Crashes counts injected crash+recovery cycles (rmp #2438); TxCrashed
// counts transactions that were OPEN when a crash landed and therefore
// died unacknowledged — their effects must be absent after recovery.
// ReplayedOps totals the WAL operations recovery replayed.
Crashes int
TxCrashed int
// NodeIDsCompared counts the surviving nodes the NodeID stability oracle
// compared across the run's recoveries (WAL v2 step 3, nodeid_stability.go).
NodeIDsCompared int
ReplayedOps int
// OverlapTicks counts ticks at which >= 2 transactions were open at once;
// WriteOverlapTicks counts ticks with >= 2 WRITE transactions open. A run
// that never overlaps proves nothing about MVCC — the gates assert these
// are nonzero.
OverlapTicks int
WriteOverlapTicks int
// MaxOpenTx is the high-water mark of simultaneously open transactions.
MaxOpenTx int
// KnowsEdges is the number of KNOWS edges the committed model holds when
// the run returns — every edge some transaction created and committed,
// minus those a committed DETACH DELETE removed. It is the mode's
// relationship-coverage observable: a run with KnowsEdges == 0 never
// exercised the edge-write path at all, so a generator change that
// silently stopped drawing CREATE ... KNOWS would show up here rather
// than passing unnoticed.
KnowsEdges int
}
MVCCSessionsResult summarises a deterministic multi-session run. All fields are pure functions of the config seed, so two runs with the same config are directly comparable (the determinism gate compares them byte for byte).
func RunMVCCSessions ¶ added in v0.12.0
func RunMVCCSessions(ctx context.Context, cfg MVCCSessionsConfig) (*MVCCSessionsResult, error)
RunMVCCSessions executes a deterministic multi-session transaction run over a WAL-backed SimDisk store and returns its result. The whole run — schedule, statements, outcomes — is a pure function of cfg.Seed: two calls with the same config produce identical results (the determinism gate relies on it).
Durability: the engine is the real persistence stack (OpenSimStore, WAL append+sync on the SimDisk), never the bare in-memory engine, so every acknowledged COMMIT in this mode is a durable commit (rmp #2435).
Concurrency contract ¶
RunMVCCSessions runs entirely on the calling goroutine and spawns none; "concurrent" transactions are interleaved deterministically, never parallel.
func (*MVCCSessionsResult) Clean ¶ added in v0.12.0
func (r *MVCCSessionsResult) Clean() bool
Clean reports whether the run finished with no violations and no fold refusals.
type MVCCSubstrateConfig ¶ added in v0.12.0
type MVCCSubstrateConfig struct {
// Seed is the master seed; the committed data is a pure function of it.
Seed uint64
// Objects is how many distinct nodes the churn spreads over (< 1 normalises
// to 8). Fewer objects concentrate the versions onto deeper chains.
Objects int
// Rounds is how many committed writes the churn performs (< 1 normalises to
// mvccSubstrateShortRounds). It must be large enough to carry the
// reclamation debt past the vacuum's wake threshold, or the run proves
// nothing about reclamation — which [checkMVCCSubstrateNonVacuity] enforces
// rather than assumes.
Rounds int
// SampleEvery is how often, in rounds, the substrate is sampled (< 1
// normalises to mvccSubstrateSampleEvery). Sampling DURING churn is what
// catches a chain-depth distribution at all: the histogram describes the
// last complete sweep and reads empty once everything has been released.
SampleEvery int
// Checkpoints, when positive, publishes that many checkpoints spread through
// the churn, exercising the commit-quiescence boundary under live traffic.
// Requires a full-stack store, which this scenario always opens.
Checkpoints int
}
MVCCSubstrateConfig parameterises a substrate-telemetry run.
type MVCCSubstrateResult ¶ added in v0.12.0
type MVCCSubstrateResult struct {
// Violations holds the adjudication findings; empty on a clean run.
Violations []Violation
// Summary is the folded evidence rendered for a log line.
Summary string
// Commits, Aborts and Conflicts are the write outcomes over the run.
Commits, Aborts, Conflicts uint64
// MaxVersionRecords is the high-water mark of live version records, against
// the two bounds the substrate publishes for it.
MaxVersionRecords int64
Bound, Ceiling int64
// CrossedBound records that the run put the substrate under reclamation
// pressure: a reading at or above the vacuum's wake threshold, or a sweep
// having run. See [mvccSubstrateEvidence.reclamationPressured] for why the
// first witness alone aliases.
CrossedBound bool
// WatermarkFrom and WatermarkTo bracket the reclamation watermark's travel.
WatermarkFrom, WatermarkTo uint64
// Sweeps and ReclaimedRecords are the vacuum's progress.
Sweeps uint64
ReclaimedRecords int64
// DeepestChain is the greatest retained chain depth any sweep reported, and
// ChainSamples how many readings carried a non-empty distribution.
DeepestChain uint64
ChainSamples int
// DeepestUnpinnedChain is the greatest retained depth observed with NO
// snapshot registered, over UnpinnedChainSamples such readings. It, and not
// DeepestChain, is what the depth bound is adjudicated on.
DeepestUnpinnedChain uint64
UnpinnedChainSamples int
// PeakActiveSnapshots is the most snapshots registered at any reading.
PeakActiveSnapshots int
// CheckpointsRun is how many checkpoints the run published, each of which
// took the commit-quiescence boundary.
CheckpointsRun int
}
MVCCSubstrateResult summarises a substrate-telemetry run.
The counts are reported so a caller can assert the run was not vacuous without re-deriving them, and so a green run logs what it actually measured rather than merely that it passed.
func RunMVCCSubstrateChurn ¶ added in v0.12.0
func RunMVCCSubstrateChurn(ctx context.Context, cfg MVCCSubstrateConfig) (*MVCCSubstrateResult, error)
RunMVCCSubstrateChurn drives version churn past the vacuum's wake threshold and adjudicates the MVCC substrate's own telemetry against it.
It samples throughout the churn rather than only at the end, because two of the four properties are not observable at rest: the chain-depth histogram is published by a sweep and reset by the next one, and the vacuum exits once it has nothing left to free. It finishes at a genuine quiescent point — every transaction committed, nothing in flight — which is the only point at which in-flight commits returning to zero is a fair question.
Concurrency contract ¶
Runs entirely on the calling goroutine. The vacuum sweeps on its own goroutine, which is what makes the telemetry scheduling-dependent; see the file comment for why the oracle is nevertheless stable.
func (*MVCCSubstrateResult) Clean ¶ added in v0.12.0
func (r *MVCCSubstrateResult) Clean() bool
Clean reports whether the run finished with no violations.
type MalformedSender ¶
type MalformedSender struct{}
MalformedSender is a bad actor: it emits intentionally ill-formed operations — invalid Cypher syntax, missing parameters, wrong parameter types, type-mismatched predicates, and oversized-but-bounded inputs — to assert that the engine rejects each with a typed error WITHOUT panicking, corrupting state, or applying any partial mutation. Every operation it emits is modelled by the oracle as a no-op (OpMalformed), so a clean run sees the engine error and the modelled state stay in lock-step (unchanged) after each one.
Concurrency contract ¶
MalformedSender is NOT safe for concurrent use; it is invoked from the single simulation goroutine.
func (MalformedSender) Name ¶
func (MalformedSender) Name() string
Name returns the actor's identifier.
func (MalformedSender) NextOp ¶
func (m MalformedSender) NextOp(seed *Seed, _ *GraphOracle) Op
NextOp returns one malformed operation, chosen by a single seed draw across the malformed families. Each family is constructed to be rejected by the engine for a distinct reason, so the workload exercises several rejection paths (parser, parameter binding, type checking, input caps) rather than one.
type ManifestFraming ¶ added in v0.12.0
type ManifestFraming struct {
// DeclaredIntegrity is the value of the document's `integrity` key, empty
// when the key is absent. rmp #2520's writer always sets it; a manifest from
// before it never has it.
DeclaredIntegrity string
// PayloadEnd is the offset one past the closing brace of the JSON value.
PayloadEnd int64
// TrailerBytes is how many bytes follow the JSON value.
TrailerBytes int
// TrailerMagicPresent reports that the last 16 bytes open with rmp #2520's
// trailer magic. The magic starts with a NUL precisely so it can never be
// mistaken for the JSON whitespace a legacy manifest ends with.
TrailerMagicPresent bool
}
ManifestFraming is a manifest.json's rmp #2520 framing, decoded from the raw bytes independently of store/snapshot.LoadManifest.
It answers the byte-level question the loader answers semantically: is the integrity trailer physically there, and does the document declare one? The loader's own IntegrityVerified is the semantic answer; a fixture assertion that used only that would be trusting the component under test to describe its own input.
func InspectManifestBytes ¶ added in v0.12.0
func InspectManifestBytes(b []byte) ManifestFraming
InspectManifestBytes decodes the framing of a manifest.json from its raw bytes. A malformed document yields a zero PayloadEnd and no trailer, which the caller reads as "not a framed manifest" — the same conclusion the loader draws before it refuses.
func (ManifestFraming) Framed ¶ added in v0.12.0
func (f ManifestFraming) Framed() bool
Framed reports that this manifest carries the rmp #2520 integrity trailer: bytes beyond the JSON value AND the trailer magic at the very end. Whitespace after the closing brace — every manifest ends with the newline the JSON encoder appends — is not a trailer, which is exactly the rule store/snapshot's splitManifestTrailer applies.
type MergeRelWriter ¶ added in v0.8.0
type MergeRelWriter struct{}
MergeRelWriter builds a Person population and repeatedly MERGEs KNOWS edges between existing Persons with a hit counter. Because every KNOWS edge in this scenario is born through MERGE (never a bare CREATE), r.n is always initialised before it is incremented, so the counter invariant is well-defined. It never deletes, so the edge set only grows or re-hits.
Concurrency contract ¶
MergeRelWriter is NOT safe for concurrent use; it is invoked from the single simulation goroutine.
func (MergeRelWriter) Name ¶ added in v0.8.0
func (MergeRelWriter) Name() string
Name returns the actor's identifier.
func (MergeRelWriter) NextOp ¶ added in v0.8.0
func (MergeRelWriter) NextOp(seed *Seed, oracle *GraphOracle) Op
NextOp creates a Person when there are fewer than two to link (or one draw in three, to keep the population growing), otherwise MERGEs a KNOWS edge between two seed-chosen existing Persons. The op is a pure function of (seed, oracle).
type MetricsObservation ¶ added in v0.12.0
type MetricsObservation struct {
// Counters maps counter name to the total delta the sink accumulated.
Counters map[string]uint64
// Latencies maps latency name to how many observations arrived under it.
Latencies map[string]uint64
}
MetricsObservation is the immutable set of metric names a recording sink saw across one bracketed run: counter totals and per-name latency-observation counts. It is what a ScenarioCounterDecl is adjudicated against.
Concurrency contract ¶
A MetricsObservation is produced by copying out of the sink under its mutex and is not mutated afterwards; it is safe to read from many goroutines.
func ObserveMetrics ¶ added in v0.12.0
func ObserveMetrics(fn func() error) (MetricsObservation, error)
ObserveMetrics installs the recording sink, runs fn, restores the no-op default and returns everything the sink saw. It is the bracket for a scenario that does NOT install a sink of its own; a scenario that does (see RunDBTeardown) publishes its own MetricsObservation on its evidence instead, and must not be wrapped in this.
Because the metrics sink is GLOBAL (see MetricsOracle) the bracket must run SERIALLY: the caller must not run concurrent metrics-emitting work, and must not call t.Parallel in a test that uses it.
func (MetricsObservation) Counter ¶ added in v0.12.0
func (o MetricsObservation) Counter(name string) uint64
Counter returns a counter's total (0 when it never moved).
func (MetricsObservation) LatencyCount ¶ added in v0.12.0
func (o MetricsObservation) LatencyCount(name string) uint64
LatencyCount returns how many latency observations a name received.
func (MetricsObservation) Names ¶ added in v0.12.0
func (o MetricsObservation) Names() []string
Names returns every metric name observed — counters and latency names alike — in sorted order. It is what the metrics-blind assertion reads, and what a witness log prints.
func (MetricsObservation) String ¶ added in v0.12.0
func (o MetricsObservation) String() string
String renders the observed counters (latency names are omitted: they are per-invocation and would swamp a failure message).
type MetricsOracle ¶
type MetricsOracle struct {
// contains filtered or unexported fields
}
MetricsOracle reads the engine's exported metrics and the goroutine count around a run and certifies that the observed deltas match an oracle's accounting (committed writes, observed errors) and the reliability bounds (goroutine baseline restored). It installs a test-side recording metrics.Backend; because that backend is the global metrics sink, an oracle must be used SERIALLY (the caller must not run two concurrent oracles or any other metrics-emitting work in parallel) — NewMetricsOracle documents this.
Concurrency contract ¶
A MetricsOracle is NOT safe for concurrent use and must be the only metrics consumer active for the duration of its MetricsOracle.Snapshot bracket.
func NewMetricsOracle ¶
func NewMetricsOracle() *MetricsOracle
NewMetricsOracle installs a recording backend as the global metrics sink and returns an oracle over it. The caller MUST call MetricsOracle.Restore when done (typically deferred) to put the previous backend back. Because the metrics sink is global, the oracle and the work it brackets must run serially — install it, run the workload, snapshot, restore, all on one goroutine with no concurrent metrics-emitting work.
func (*MetricsOracle) Check ¶
func (o *MetricsOracle) Check(before, after MetricsSnapshot, expectedWrites, expectedWriteErrors uint64, goroutineSlack int) MetricsOracleResult
Check certifies a before/after pair against the oracle's accounting and the reliability bounds, returning the verdict. expectedWrites is how many write-path statements the workload executed; expectedWriteErrors is how many of those the engine should have rejected (the oracle's count of expected failures). goroutineSlack is the tolerance on the goroutine delta — a healthy run returns to its baseline, but a small positive slack absorbs the runtime's own bookkeeping goroutines that a single -count run may leave parked.
How callers choose goroutineSlack (rmp #2592) ¶
It is a parameter and not a constant because the right value is a property of the RUN, not of the quantity. Two things create room for the runtime to park a helper, and they are independent:
- CONCURRENCY. A swarm of N workers needs slack that scales with N. The callers in phase5_integration_test.go pass 2, 4 and 8 for progressively wider swarms; metrics_oracle_test.go passes 4.
- DURATION. A minutes-long run has more opportunity than a short one whatever its concurrency. phase4_long_running_soak_test.go is single-goroutine but drives 2,000,000 ticks, and passes 8 on that ground alone; bolt/server's connection-churn soak independently settled on the same value for the same reason.
A run that is neither concurrent nor long takes ZERO, which is what the single-scenario path at [MetricsOracle.RunWithMetricsOracle] does.
Note what this delta is and is not. runtime.NumGoroutine counts goroutines the RUNTIME owns as well as the harness's, so it cannot distinguish a leak from a parked GC worker. Where a caller also runs goleak, THAT is the instrument that certifies "leaks no goroutine" — it identifies a leak by its stack — and this delta is a coarse cross-check that additionally catches a leak which has already exited by teardown.
func (*MetricsOracle) CheckGoroutineBaseline ¶
func (o *MetricsOracle) CheckGoroutineBaseline(before, after MetricsSnapshot, goroutineSlack int) MetricsOracleResult
CheckGoroutineBaseline certifies ONLY the reliability bound — that the live goroutine count returned to its baseline (within slack) — between the before and after snapshots. It is the metrics-oracle check that applies to the CONCURRENT swarm: per-run write/error counts cannot be attributed to one run when many workers share the global metrics sink, but a goroutine leak across the whole swarm IS observable and is the bound that matters there. A clean swarm spawns its workers and joins them all, so the count must return to baseline.
func (*MetricsOracle) Restore ¶
func (o *MetricsOracle) Restore()
Restore reinstalls the no-op default metrics backend. It is safe to call more than once.
func (*MetricsOracle) Snapshot ¶
func (o *MetricsOracle) Snapshot() MetricsSnapshot
Snapshot reads the current exported-metric values and the live goroutine count into a MetricsSnapshot.
type MetricsOracleResult ¶
type MetricsOracleResult struct {
// Discrepancies lists every mismatch found (empty when the metrics are
// consistent with the oracle and the reliability bounds hold).
Discrepancies []string
// Before / After are the snapshots bracketing the run.
Before MetricsSnapshot
After MetricsSnapshot
// ExpectedWrites / ExpectedWriteErrors are the oracle's accounting of how
// many write statements ran and how many the engine should have rejected.
ExpectedWrites uint64
ExpectedWriteErrors uint64
}
MetricsOracleResult is the verdict of a metrics-oracle check: the before/after snapshots, the accounting the oracle expected, and any discrepancy.
func RunWithMetricsOracle ¶
func RunWithMetricsOracle(ctx context.Context, seed uint64, ops int, wlFactory func(*Seed) *Workload) (MetricsOracleResult, error)
RunWithMetricsOracle drives a deterministic write workload for the seed against a fresh in-memory engine while a MetricsOracle brackets it, and returns the verdict. It is the wired metrics-oracle check used by the swarm and the integration tests: after the run, the engine's exported RunInTx observation count and error counter must match the oracle's own count of write statements and rejections, and the goroutine count must return to its baseline.
It must run SERIALLY (it installs the global metrics backend); the caller must not run concurrent metrics-emitting work. It restores the no-op backend before returning.
func (*MetricsOracleResult) Consistent ¶
func (r *MetricsOracleResult) Consistent() bool
Consistent reports whether the metrics matched the oracle and the reliability bounds (no discrepancy).
func (*MetricsOracleResult) String ¶
func (r *MetricsOracleResult) String() string
String renders the result: the deltas and every discrepancy.
type MetricsRunStats ¶
type MetricsRunStats struct {
// Writes is the number of write-path statements issued.
Writes uint64
// WriteErrors is the number of write statements the engine rejected (the
// per-op execute reported not-committed because RunWrite returned an error).
WriteErrors uint64
}
MetricsRunStats is the oracle-side accounting of a metrics-bracketed run: how many write statements were issued and how many the engine rejected, derived from the run's own outcomes (the per-op committed flag). It is the ground truth the metric deltas are checked against.
type MetricsSnapshot ¶
type MetricsSnapshot struct {
// RunInTxCount / RunCount are the per-invocation latency-observation counts
// for the write and read paths.
RunInTxCount uint64
RunCount uint64
// RunInTxErrors / RunErrors are the engine error counters.
RunInTxErrors uint64
RunErrors uint64
// Goroutines is the live goroutine count at the snapshot instant.
Goroutines int
}
MetricsSnapshot is an immutable read of the engine's exported metrics plus the goroutine count at one instant. The oracle takes one before and one after a run and asserts the deltas match the oracle's accounting and the reliability bounds.
type NodeState ¶
NodeState is the oracle's record of a single node: its synthetic oracle id, labels, and properties. It mirrors what the engine must hold, not how the engine stores it.
type NullSemanticsWriter ¶ added in v0.12.0
type NullSemanticsWriter struct {
// contains filtered or unexported fields
}
NullSemanticsWriter builds the graph the null-semantics battery reads: it creates Person nodes of three deterministic, seed-chosen shapes — AGELESS (name+city, ~1/3 of creates, via [tmplCreatePersonNoAge]), CITYLESS (name+age, via [tmplCreatePerson]), and full (name+age+city, via [tmplCreatePersonCity]) — plus KNOWS edges between existing Persons, so the NULL-padding OPTIONAL MATCH probe sees both matched and NULL rows. It avoids MERGE and SET so the oracle's age-present vs age-absent bookkeeping is unambiguous.
Concurrency contract ¶
NullSemanticsWriter is NOT safe for concurrent use; it is invoked from the single simulation goroutine.
func (*NullSemanticsWriter) Name ¶ added in v0.12.0
func (*NullSemanticsWriter) Name() string
Name returns the actor's identifier.
func (*NullSemanticsWriter) NextOp ¶ added in v0.12.0
func (w *NullSemanticsWriter) NextOp(seed *Seed, oracle *GraphOracle) Op
NextOp returns a KNOWS edge between two existing Persons or a fresh Person CREATE in one of the three property shapes, all seed-derived.
type Op ¶
Op is a single Cypher operation an actor emits: the query text, its bound parameters (string-keyed, value kinds limited to those toExprParams supports), and its kind.
func GenerateCrossReleaseOps ¶
GenerateCrossReleaseOps produces a deterministic write-biased op stream from seed for the cross-release harness. It is the SAME workload the in-process upgrade harness drives (so the two are directly comparable), captured as a flat slice the harness can serialise to the prior-release helper AND replay in-process. Params are normalised through a JSON round-trip so the prior helper (which receives them as JSON) and the current side bind byte-identical parameter values.
type OpKind ¶
type OpKind string
OpKind classifies an operation so the simulator can route it to the engine's read or write path and the oracle to the matching Apply method.
const ( OpCreate OpKind = "OpCreate" OpMatch OpKind = "OpMatch" OpMerge OpKind = "OpMerge" OpDelete OpKind = "OpDelete" OpUpdate OpKind = "OpUpdate" // OpMalformed is an intentionally ill-formed operation emitted by // [MalformedSender]. The engine must reject it with a typed error without // panicking, corrupting state, or applying any partial mutation; the oracle // models it as a no-op (it never changes modelled state). OpMalformed OpKind = "OpMalformed" )
Operation kinds.
func (OpKind) IsWrite ¶
IsWrite reports whether an operation of this kind mutates the graph and must therefore run through the engine's write (RunInTx) path. Malformed operations are routed through the write path too: it is the stricter, atomicity-bearing path, so proving a malformed statement is rejected there (with a full rollback and no partial application) is the stronger guarantee. A malformed read-shaped statement run through the write path still simply errors.
type OracleOp ¶
type OracleOp struct {
Params map[string]any
Cypher string
Expected OracleResult
Tick int64
}
OracleOp is one entry in the oracle's operation history: the tick at which it ran, the Cypher and parameters issued, and the predicted result. The history is retained so a future phase can shrink a failing trace to a minimal reproducer.
type OracleResult ¶
OracleResult is the oracle's prediction for one operation: whether it commits, how many nodes and edges it creates, and (when it predicts a failure) the reason. The simulator records it for comparison with the engine outcome and for replay/shrinking.
type OracleSnapshot ¶
OracleSnapshot is an immutable summary of the oracle state at the moment a simulation failed, captured for the report. It deliberately holds only aggregate counts and the operation history length, not the full node/edge maps, so a report stays compact; the seed plus the failing tick are enough to replay the full state.
type OracleTx ¶ added in v0.12.0
type OracleTx struct {
// contains filtered or unexported fields
}
OracleTx is a per-transaction workspace over a GraphOracle, so the simulator can adjudicate runs where several transactions are in flight at once (the MVCC multi-session mode). Obtain one with GraphOracle.BeginTx; finish it with exactly one OracleTx.Commit or OracleTx.Abort.
Model ¶
An OracleTx mirrors what the engine gives an explicit WRITE transaction under MVCC: SNAPSHOT ISOLATION. Each statement reads the committed state AS OF BEGIN plus the transaction's own pending writes — a commit that lands after BEGIN is invisible for the transaction's whole lifetime — and nothing the transaction wrote is visible to anyone else until Commit publishes it as one step. (Established empirically against the store-backed engine by TestProbe_WriteTxSnapshotAtBegin: a post-BEGIN commit reads as absent, count=0. The engine's own BeginTx godoc agrees: concurrent readers "observe the state before it began until it commits".) The workspace therefore captures a SNAPSHOT of the visible node set at GraphOracle.BeginTx, keeps an OVERLAY (pending creates, deletes, property updates, pending edges) over that snapshot, decides every statement's effect at statement time against snapshot+overlay, and folds the DECIDED effects into the parent only at OracleTx.Commit — never re-evaluating them, exactly as the engine publishes rather than re-runs a transaction at COMMIT.
The workspace models the transactional workload's template set only, and it does NOT model schema constraints (GraphOracle.SetUniqueOnName and friends stay autocommit-scenario concerns): the transactional workload binds globally-unique names for CREATE, so the only cross-transaction name collision it can produce is a MERGE race, which Commit's validation refuses (see below).
Adjudication boundary ¶
The workspace applies decided effects; it does not decide whether the engine was ENTITLED to commit. When the engine refuses a transaction with a serialization conflict, the driver calls OracleTx.Abort and the workspace vanishes without trace. When the engine acknowledges a commit, the driver calls OracleTx.Commit; a decided effect that no longer folds cleanly (its target vanished from — or its MERGE-decided create appeared in — the committed state since the statement ran) is reported as an error and the parent is left UNCHANGED: the schedule let two transactions collide where the engine should have conflicted, and the checker above this layer owns that verdict.
Concurrency contract ¶
OracleTx is NOT safe for concurrent use. Like the GraphOracle it overlays, it is created, mutated, and folded from the single simulation goroutine; "concurrent transactions" in the deterministic mode are interleaved on that one goroutine, never parallel.
func (*OracleTx) Abort ¶ added in v0.12.0
func (t *OracleTx) Abort()
Abort discards the workspace: no pending create, delete, update, or edge reaches the parent, and the transaction-local history is dropped. It is idempotent and safe after Commit (where it is a no-op).
func (*OracleTx) AgeOf ¶ added in v0.12.0
AgeOf returns the age this transaction would observe for the named Person, and whether the node is visible: a pending in-tx update wins over the begin-snapshot value.
func (*OracleTx) ApplyCreate ¶ added in v0.12.0
func (t *OracleTx) ApplyCreate(cypher string, params map[string]any) OracleResult
ApplyCreate models [tmplCreatePerson] inside the transaction: the node lands in the overlay, invisible to every other session until Commit.
func (*OracleTx) ApplyCreateKnows ¶ added in v0.12.0
func (t *OracleTx) ApplyCreateKnows(params map[string]any) OracleResult
ApplyCreateKnows models [tmplCreateKnows] inside the transaction: the edge is decided against the endpoints visible to the transaction (begin-snapshot plus own pending nodes) and folded at Commit. Like the parent model, a missing endpoint is a committed zero-effect result and a duplicate (a,b) is idempotent.
func (*OracleTx) ApplyDelete ¶ added in v0.12.0
func (t *OracleTx) ApplyDelete(cypher string, params map[string]any) OracleResult
ApplyDelete models [tmplDetachDelete] inside the transaction: an own-pending node is cancelled outright (with its pending edges); a committed node is marked for deletion at Commit. A miss is a committed zero-effect result.
func (*OracleTx) ApplyMatch ¶ added in v0.12.0
func (t *OracleTx) ApplyMatch(cypher string, params map[string]any) OracleResult
ApplyMatch models [tmplSetAge] and pure reads inside the transaction. A SET on a node visible to the transaction is decided now (against the begin-snapshot plus own writes) and folded at Commit; a miss is a committed zero-effect result; and a SET that stores the value the transaction already observes is not a write at all (see below).
Value-preserving writes are not writes (rmp #2717) ¶
The engine records NO version for a property write whose value equals the one already stored: graph/lpg/property.go, setNodePropertyInfo, whose delta push is guarded by `case !propValuesDefinitelyEqual(prev, value)`. The conflict test in front of it is unconditional (the rmp #2324 fix), so a write that reaches the guard is one whose stored head IS visible to the writing transaction — the value it leaves behind is byte-identical to the value that was there, and the transaction therefore neither conflicts with a concurrent writer of that node nor makes a concurrent writer conflict.
The workspace must model it the same way. Recording it as a pending update makes OracleTx.Commit refuse to fold a transaction whose target a concurrently committed DETACH DELETE removed — a commit the engine was entitled to acknowledge, because the resulting state is exactly what BOTH serial orders produce (the SET changes nothing, so "SET then DELETE" and "DELETE then SET-misses" both end with the node gone). Measured on seeds 22 (crash arm), 500 and 572 (no-crash arm) of the MVCC sessions mode: in all three the SET re-asserted the age the node was CREATEd with.
Recording it was also wrong in the APPLY direction: the fold wrote the re-asserted value over a newer one a concurrent transaction had committed in the meantime. The engine keeps the newer value (the value-preserving write pushed no version, so the other writer never conflicted), so the model was silently drifting away from it.
The strict refusal is kept for a value-CHANGING SET, which is a real write: there the engine does refuse one of the two transactions with mvcc.ErrSerializationConflict, so a clean commit of both would be a genuine isolation finding.
func (*OracleTx) ApplyMerge ¶ added in v0.12.0
func (t *OracleTx) ApplyMerge(cypher string, params map[string]any) OracleResult
ApplyMerge models [tmplMergePerson] inside the transaction: MERGE by name is a no-op when the name is visible (committed or own-pending) and a pending create otherwise.
func (*OracleTx) Commit ¶ added in v0.12.0
Commit folds the transaction's decided effects into the parent atomically: it VALIDATES every effect first and applies only when all of them fold cleanly, so a failed Commit leaves the parent byte-identical (all-or-nothing, like the engine's own publish step).
A validation failure means the schedule let this transaction's decided effects collide with a commit that landed after they were decided — a collision the engine is expected to refuse with a serialization conflict, in which case the driver must call OracleTx.Abort instead. Commit returning an error is therefore a checker-level finding, not a state transition: the workspace stays unfinished so the caller can still Abort it.
func (*OracleTx) CreatedNames ¶ added in v0.12.0
CreatedNames returns, in ascending sorted order, the names this transaction has pending-created (CREATE or MERGE-create) and not since cancelled. The isolation checkers use it to pick a committed name for the session's cross-transaction read-your-own-writes probe.
func (*OracleTx) HasPerson ¶ added in v0.12.0
HasPerson reports whether a Person of the given name is visible to this transaction: its own pending writes overlaid on the begin-snapshot.
func (*OracleTx) NodeCount ¶ added in v0.12.0
NodeCount returns the number of nodes visible to this transaction.
func (*OracleTx) NodeNames ¶ added in v0.12.0
NodeNames returns the Person names visible to this transaction in ascending sorted order — the begin-snapshot minus pending deletes, plus pending creates. The deterministic order is load-bearing for the same reason as GraphOracle.NodeNames: actors index into it with seed-derived integers.
func (*OracleTx) Ops ¶ added in v0.12.0
Ops returns the transaction-local operation history (folded into the parent at Commit, discarded at Abort). The returned slice aliases the workspace's backing store and must not be mutated.
func (*OracleTx) PendingKnows ¶ added in v0.12.0
PendingKnows reports whether this transaction has already decided a KNOWS edge between a and b. The multi-session workload combines it with the parent's present-state GraphOracle.HasKnowsByName to avoid re-creating an existing edge — the engine's parallel-edge check runs against the PRESENT adjacency, so on the simulator's non-multigraph store a duplicate CREATE is refused with a typed error rather than deduplicated.
type OrderingStats ¶ added in v0.12.0
type OrderingStats struct {
// contains filtered or unexported fields
}
OrderingStats accumulates what the ordering probes actually OBSERVED over a run, so the terminal gate ([checkOrderingNonVacuity]) can prove the sensitive arms were exercised instead of passing on a graph too small or too uniform to distinguish a working sort from a broken one.
A nil *OrderingStats is a valid receiver for every note method, so a one-shot (test) call may pass nil when it does not care about the run-level record.
Concurrency contract ¶
OrderingStats is NOT safe for concurrent use; the simulator drives the battery from a single goroutine.
type OverloadActor ¶
type OverloadActor struct {
// Tag, when non-empty, is stamped into every node an
// [OverloadLargeCreateTx] creates, as the "run" property, so the caller can
// tell the nodes ITS OWN heavy write committed from every other node
// carrying the same label.
//
// It is load-bearing wherever the heavy write is adjudicated, not
// decorative. Several connections issue heavy writes against one engine at
// the same time, so an untagged count of the label answers for all of their
// populations at once and can therefore adjudicate none of them — the same
// cross-writer confusion the contended counter namespace of rmp #2729
// exists to prevent, one level down.
//
// Empty leaves the statement untagged, which is what a single-connection
// exercise with the engine to itself wants.
Tag string
}
OverloadActor issues legitimately heavy work over the real Bolt wire and classifies the engine's response. It asserts the engine enforces its declared bounds (a typed FAILURE or a bounded, fully-streamed success) and degrades gracefully — never OOM, panic, deadlock, or drop an acknowledged write.
Scope: reads AND the one heavy write ¶
Three of the four families are reads; OverloadLargeCreateTx is the heavy WRITE, and it is the family that answers the write half of the graceful-degradation mandate. It runs on the concurrent overload role when the caller's mix sets ConcurrentMix.OverloadHeavyWrites — which the catalogue's "overload" scenario does — and the nodes it commits are adjudicated against its acknowledgement and then removed (see overloadHeavyWriteOp). Before rmp #2736 the concurrent role mapped the family away unconditionally, so the only heavy write was issued by this package's own unit exercise and never on a production path.
Concurrency contract ¶
OverloadActor holds no mutable state: Tag is set by the caller at construction and never written afterwards, so each OverloadActor.Run call drives only the one connection it is given. It is safe to use from many goroutines (the concurrent harness does), each with its own value and its own connection.
func (OverloadActor) PickFamily ¶
func (OverloadActor) PickFamily(seed *Seed) OverloadFamily
PickFamily chooses an overload family from the seed (one int draw).
func (OverloadActor) Run ¶
func (a OverloadActor) Run(c *WireClient, family OverloadFamily) (OverloadOutcome, error)
Run executes one heavy operation of the given family over c and returns the classified outcome. The connection must already be Connected. It never returns an error for an expected engine bound (that is a BoundedError outcome); it returns an error only for a harness/transport failure.
type OverloadFamily ¶
type OverloadFamily int
OverloadFamily identifies one class of legitimately-heavy operation the OverloadActor issues. Each is well-formed Cypher that pushes a resource dimension (transaction size, list size, traversal breadth/depth, result-set size) toward or past the engine's declared bound.
const ( // OverloadHugeUnwind unwinds a very large literal range, producing many rows. OverloadHugeUnwind OverloadFamily = iota // OverloadLargeCreateTx creates many nodes in a single autocommit transaction, // pushing the per-transaction op count toward DefaultMaxTxnOps. OverloadLargeCreateTx // OverloadLargeResultSet matches a Cartesian product to materialise a large // result set, exercising the engine's MaxResultRows cap. OverloadLargeResultSet // OverloadDeepVLE runs a deep/wide variable-length expansion over the seeded // graph, exercising traversal bounds. OverloadDeepVLE )
Overload families.
func (OverloadFamily) String ¶
func (f OverloadFamily) String() string
String renders an OverloadFamily for reports.
type OverloadOutcome ¶
type OverloadOutcome struct {
FailureMsg string // populated when BoundedError
Family OverloadFamily
Rows int // rows actually streamed back (always bounded)
Succeeded bool // the op completed and drained cleanly
BoundedError bool // the engine refused it with a typed FAILURE (a declared bound)
}
OverloadOutcome records the result of one heavy operation. Exactly one of Succeeded or BoundedError is the acceptable result: the engine either served the work within its limits or refused it with a typed limit/bound error. A hang, panic, OOM, or dropped acknowledged write is a violation (a hang is caught by the caller's deadline; the others by goleak/no-panic and the durability re-read).
func (OverloadOutcome) Acceptable ¶
func (o OverloadOutcome) Acceptable() bool
Acceptable reports whether the outcome honours the graceful-degradation contract: success within limits OR a typed bound error.
type OverloadReader ¶ added in v0.6.0
type OverloadReader struct{}
OverloadReader is the deterministic, engine-API counterpart of the wire-only OverloadActor: it emits over-budget READ statements (see [overloadReadTemplates]) to drive the engine's bounded-resource / graceful- degradation contract under the mem-pressure scenario. With the engine's logical budgets clamped low, each over-budget read is refused with a typed resource-exhausted error during the result drain — never a panic, never a partial result — and, being a read, changes no modelled state. The oracle records it as a no-op read, so engine and oracle stay in lock-step.
Concurrency contract ¶
OverloadReader is NOT safe for concurrent use; it is invoked from the single simulation goroutine.
func (OverloadReader) Name ¶ added in v0.6.0
func (OverloadReader) Name() string
Name returns the actor's identifier.
func (OverloadReader) NextOp ¶ added in v0.6.0
func (OverloadReader) NextOp(seed *Seed, _ *GraphOracle) Op
NextOp returns one over-budget read, chosen by a single seed draw so the op stream stays a pure function of the seed.
type PageRankRankerConfig ¶ added in v0.12.0
type PageRankRankerConfig struct {
// Plan is the window sequence. Nil derives it from Seed via [prPlan].
Plan []prWindow
// CrossRegimeWorkers are the worker counts the cross-regime arm sweeps. Nil
// uses [pageRankerCrossRegimeWorkers].
CrossRegimeWorkers []int
// Seed drives the fixture and the drawn dampings.
Seed uint64
// Perturb is the deliberate corruption to apply, threaded as a parameter so
// no package-level variable carries it.
Perturb prPerturb
}
PageRankRankerConfig parameterises one run of the scenario.
func DefaultPageRankRankerConfig ¶ added in v0.12.0
func DefaultPageRankRankerConfig(seed uint64) PageRankRankerConfig
DefaultPageRankRankerConfig returns the configuration the catalogue entry runs.
type PageRankRankerEvidence ¶ added in v0.12.0
type PageRankRankerEvidence struct {
Windows []prWindowEvidence
CrossRegime []prCrossRegime
// The fixture.
Nodes int
Edges int
Live int
Threshold int
Hub int
HubInDeg int
// Sequence aggregates.
SerialWindows int
ParallelWindows int
FirstParallel int
DistinctBuffers int
BufferRepeats int
AliasArmed int
ConvergedRuns int
CappedRuns int
DistinctIters int
MaxEmptyRanges int
// The reference arm.
RefWindow int
RefMaxDev float64
RefEpsilon float64
RefIndex int
// TransposeFloor is the byte floor the first parallel window is held to, and
// FirstParallelAlloc the delta it measured.
TransposeFloor uint64
FirstParallelAlloc uint64
SecondParallelAlloc uint64
// Digest is an order-sensitive hash of every reproducible fact above.
Digest uint64
Perturb prPerturb
}
PageRankRankerEvidence is what one run of the scenario measured. It is the object both the terminal gate and the short-layer test read, so "the run passed" and "the run exercised something" are separate questions with separate answers.
func (*PageRankRankerEvidence) ReproducibleSummary ¶ added in v0.12.0
func (e *PageRankRankerEvidence) ReproducibleSummary() string
ReproducibleSummary renders only the facts that are a pure function of the seed, so two runs of one seed can be compared without the timing-dependent counters (the label-lookup position within its band, and the process-global allocation deltas) making a deterministic harness look non-deterministic.
func (*PageRankRankerEvidence) String ¶ added in v0.12.0
func (e *PageRankRankerEvidence) String() string
String renders the evidence for a report and a log line.
type ParityProbe ¶ added in v0.12.0
type ParityProbe struct {
// Shape names the predicate shape for violation messages
// (e.g. "equality", "range", "starts-with", "in-list").
Shape string
// Literal is the query with the value(s) inlined. It must project a single
// integer column — id(n) in the scenario probes — so results can be
// compared as multisets.
Literal string
// Param is the identical query with the value(s) replaced by parameters.
Param string
// Params binds Param's parameters to exactly the inlined values.
Params map[string]any
// MustSeek asserts the LITERAL arm resolves through an index seek
// (NodeByIndexSeek / NodeByIndexSeekSet / NodeByIndexRangeScan). Set it for
// shapes the engine is known to seek on the probed data, so a pair that
// agrees on two scans is reported instead of passing vacuously.
MustSeek bool
}
ParityProbe is one predicate shape written twice — once with the value inlined, once parameterised — over the same indexed property. Literal and Param must be the SAME predicate: the checker asserts their plans and their result multisets agree.
type PatternShapesWriter ¶ added in v0.12.0
type PatternShapesWriter struct {
// contains filtered or unexported fields
}
PatternShapesWriter builds the two-relationship-type Person graph the pattern-shapes battery reads: fresh Person nodes, seed-drawn KNOWS and FOLLOWS edges (endpoint draws are independent, so self-loops occur), and a deterministic motif planted every [patternMotifEvery] ops — a directed KNOWS triangle a→b→c→a, a mutual KNOWS pair a⇄b, a FOLLOWS a→b on top of the KNOWS a→b (a both-types pair), and a KNOWS self-loop c→c. The motifs guarantee every reference shape (triangle, mutual pair, both-types union, relationship-uniqueness exclusion) is exercised non-vacuously.
The scenario opens the engine as a directed MULTIGRAPH, because a simple graph admits at most one edge per ordered node pair REGARDLESS of type — a KNOWS+FOLLOWS both-types pair is impossible there. The writer, however, never re-CREATEs an existing (src,dst,label) edge (it consults the oracle first, the edge-properties guard precedent), so the multigraph never holds parallel instances and edge identity remains exactly (src,dst,label): every adjacency-derived reference count below is exact without per-instance modelling (parallel-instance multiplicity is the edge-properties scenario's concern). Self-loops are deliberately ALLOWED and planted: openCypher matches a self-loop once for an undirected relationship pattern, and a self-loop is the only single-instance shape where the relationship-isomorphism exclusion (r1<>r2) actually removes a binding — planting them is what makes the uniqueness probe sensitive to a Cyphermorphism regression.
Concurrency contract ¶
PatternShapesWriter is NOT safe for concurrent use; it is invoked from the single simulation goroutine.
func (*PatternShapesWriter) Name ¶ added in v0.12.0
func (*PatternShapesWriter) Name() string
Name returns the actor's identifier.
func (*PatternShapesWriter) NextOp ¶ added in v0.12.0
func (w *PatternShapesWriter) NextOp(seed *Seed, oracle *GraphOracle) Op
NextOp drains the pending motif queue first; otherwise it returns a fresh Person CREATE or a seed-drawn KNOWS/FOLLOWS edge between existing Persons (endpoints drawn independently, so occasional random self-loops occur). A drawn pair whose (src,dst,label) edge the oracle already models falls back to a Person CREATE instead, so the multigraph never holds a parallel instance and edge identity stays (src,dst,label)-exact.
type PendingState ¶
type PendingState struct {
InFlightOps int // operations not yet acknowledged (0 at quiescence)
UngatedStreams int // result streams opened but not drained (0 at quiescence)
OracleDivergent bool // expected engine node count != observed engine node count
ExpectedNodes int64 // for the divergence report (baseline + acked creates)
EngineNodeCount int64
}
PendingState is the liveness predicate's snapshot of outstanding work. The system is QUIESCENT when every field is at its converged value: no in-flight operations, no ungated (undrained) streams, and the oracle equal to the engine. Goroutine-leak detection is delegated to goleak in test teardown (process-global goroutine counts are too noisy to gate convergence on), so it is deliberately NOT a pending() term.
func (PendingState) Magnitude ¶
func (p PendingState) Magnitude() int
Magnitude is a scalar measure of outstanding work the watchdog tracks for progress: a strictly decreasing magnitude is progress, a flat non-zero magnitude across the grace window is resonance.
func (PendingState) Pending ¶
func (p PendingState) Pending() bool
Pending reports whether any work is still outstanding. False means quiescent.
func (PendingState) String ¶
func (p PendingState) String() string
String renders a PendingState for the liveness report's pending-work dump.
type PlanBaseline ¶ added in v0.12.0
type PlanBaseline struct {
// contains filtered or unexported fields
}
PlanBaseline is the plan-stability oracle's captured state: the physical Explain rendering of both arms of a fixed probe set at one instant. A later CheckPlanStability against the same engine (or its crash-recovered successor) must reproduce every rendering byte-identically — a plan-cache rebuild that changes an access path is a violation even when both arms of a pair change together, which the parity check alone would miss.
Concurrency contract ¶
PlanBaseline is immutable after CapturePlanBaseline returns and is safe to read from one goroutine at a time; the simulator drives it from the single simulation goroutine.
func CapturePlanBaseline ¶ added in v0.12.0
func CapturePlanBaseline(engine PlanEngine, probes ...ParityProbe) (*PlanBaseline, error)
CapturePlanBaseline renders both arms of every probe through engine.Explain and freezes the result. Capture failures are returned as an error (the scenario cannot assert stability against a baseline it failed to take).
type PlanEngine ¶ added in v0.12.0
type PlanEngine interface {
Engine
// Explain returns the engine's physical-plan rendering for query without
// executing it.
Explain(query string, params map[string]any) (string, error)
// Profile executes the (read-only) query and returns the physical plan
// annotated with per-operator rows, db-hits, and time.
Profile(ctx context.Context, query string, params map[string]any) (string, error)
}
PlanEngine is the slice of the engine surface the access-path parity checker needs beyond Engine: the physical-plan rendering (Explain) and the profiled execution (Profile). The simulator's EngineAdapter satisfies it.
Concurrency contract ¶
Implementations need only be safe for single-goroutine use; the simulator never calls them concurrently.
type PriorReleaseHelper ¶
type PriorReleaseHelper struct {
// Tag is the git tag the helper was built from (e.g. "v0.3.0").
Tag string
// BinPath is the absolute path of the built helper binary.
BinPath string
// BuildFallbackErr is the error the FIRST (checkpoint-bearing) build
// returned, when the harness had to fall back to the WAL-only helper. It is
// nil when the checkpoint-bearing build succeeded. Callers report it so a
// silent loss of checkpoint coverage at some tag is visible in the test log
// rather than inferred from CheckpointSupported alone.
BuildFallbackErr error
// CheckpointSupported reports that the tag built WITH
// [xreleaseHelperCheckpointRel] staged, so [PriorReleaseHelper.WriteImage]
// will publish a snapshot directory alongside the WAL. False means the tag's
// checkpoint API is not the one the staged file uses and the helper was
// rebuilt without it.
CheckpointSupported bool
// contains filtered or unexported fields
}
PriorReleaseHelper is a built prior-release helper binary plus the worktree it was compiled in. Close removes both, deterministically. It is the cross-release equivalent of [subproc]: instead of re-execing the current test binary, it spawns a binary built from a PRIOR git tag's source so the harness can observe genuine cross-version behaviour.
Concurrency contract ¶
A PriorReleaseHelper is not safe for concurrent use across its own methods, but PriorReleaseHelper.WriteImage is a pure spawn-and-wait and may be called from one goroutine at a time. Close is idempotent.
func BuildPriorReleaseHelper ¶
func BuildPriorReleaseHelper(ctx context.Context, repoRoot, tag string) (*PriorReleaseHelper, error)
BuildPriorReleaseHelper checks out tag into a temporary git worktree, copies the current helper sources into it, and builds the helper binary against that tag's packages. The returned helper's Close removes the worktree and the temporary build root.
repoRoot must be the absolute path of the GoGraph repository working tree (the directory holding .git). The build runs `go build` inside the worktree so the binary links the tag's store/txn/wal/cypher code.
The build is two-stage (rmp #2477) ¶
Both [xreleaseHelperMainRel] and [xreleaseHelperCheckpointRel] are staged and the build is attempted. If THAT build fails, the checkpoint file alone is removed and the build is retried with the WAL-only helper. The distinction is recorded in PriorReleaseHelper.CheckpointSupported and, for the failing case, PriorReleaseHelper.BuildFallbackErr.
The fallback exists because the two files carry different compatibility risk. main.go is pinned to the API stable across v0.2.0..HEAD; checkpoint.go reaches the younger store/checkpoint API. Putting both in one file would make any checkpoint-API drift at some tag fail the WHOLE build, which the caller reports as an environment-precondition SKIP — silently deleting that release from cross-release coverage entirely. Two stages degrade one capability instead of losing the tag.
An error from this function is an ENVIRONMENT-PRECONDITION failure (the tag is not present, git worktree is unavailable, or the tag's tree does not build with the current toolchain EVEN WITHOUT the checkpoint file): callers gate on it as a clean skip, exactly like an optional external tool being absent, NOT as a test failure.
func BuildPriorReleaseHelperWithOptions ¶ added in v0.12.0
func BuildPriorReleaseHelperWithOptions(ctx context.Context, repoRoot, tag string, opts XReleaseBuildOptions) (*PriorReleaseHelper, error)
BuildPriorReleaseHelperWithOptions is BuildPriorReleaseHelper with explicit build options. See XReleaseBuildOptions.
func (*PriorReleaseHelper) Close ¶
func (h *PriorReleaseHelper) Close() error
Close removes the worktree and temporary build artefacts. It is idempotent.
func (*PriorReleaseHelper) SelfRecoverCounts ¶
func (h *PriorReleaseHelper) SelfRecoverCounts(ctx context.Context, dir string) (nodes, edges int64, err error)
SelfRecoverCounts reopens a dir this helper previously wrote using the PRIOR release's OWN recovery and returns the node/edge counts it recovers. It is the durable truth of what the prior release wrote, as the prior release itself reads it back — the reference the current code's recovery must reproduce. It discriminates a prior-release WAL that does not round-trip in its own release (a prior defect) from one the current code mis-reads (a current regression): the cross-version contract is current-recovery == prior-self-recovery.
func (*PriorReleaseHelper) WriteImage ¶
func (h *PriorReleaseHelper) WriteImage(ctx context.Context, dir string, ops []Op) (HelperRunResult, error)
WriteImage drives ops through the prior-release helper, which opens a WAL-backed store under dir, runs each op, publishes a checkpoint when the tag supports it, and closes (flush+fsync) so dir holds a durable store image written ENTIRELY by the prior release. It returns the prior release's per-op results, final counts, and whether the image carries a snapshot directory.
dir must be an existing, empty directory the current process owns; after this returns, the current code can reopen dir via recovery.Open to perform the cross-version upgrade check. When HelperRunResult.Checkpoint is true that reopen parses the PRIOR RELEASE'S snapshot bytes — manifest.json, csr.bin and the component files — which is the surface the WAL-only image never reached.
type PriorSnapshotFacts ¶ added in v0.12.0
type PriorSnapshotFacts struct {
// ManifestErr is the error the CURRENT manifest reader returned for the
// prior release's manifest.json, or nil when it loaded. A non-nil value with
// Present true is a cross-version manifest-format fault.
ManifestErr error
// Integrity is the manifest's declared integrity scheme, empty for a manifest
// written before the CRC32C trailer existed (rmp #2520).
Integrity string
// Files lists the component file names the manifest declares, in manifest
// order.
Files []string
// ManifestVersion is the on-disk schema version the prior release stamped.
ManifestVersion int
// Present reports that <dir>/snapshot/manifest.json exists on disk.
Present bool
// IntegrityVerified reports that the current reader VERIFIED a checksum
// trailer over these bytes. False for every manifest written before rmp
// #2520 — which is the expected, documented outcome for a prior release and
// is asserted as such, not tolerated.
IntegrityVerified bool
}
PriorSnapshotFacts is what the snapshot directory a PRIOR RELEASE published looks like from the CURRENT reader's point of view, read straight off disk and independently of what recovery reports (rmp #2477).
It is read independently on purpose. "The current code opened the prior release's snapshot" is otherwise a single-source claim — recovery's own SnapshotHit — and a reader that silently ignored the directory would report exactly the same false as one that was never given a directory at all. Reading the filesystem separately makes those two states distinguishable.
func InspectPriorSnapshotDir ¶ added in v0.12.0
func InspectPriorSnapshotDir(dir string) PriorSnapshotFacts
InspectPriorSnapshotDir reads <dir>/snapshot/manifest.json with the CURRENT manifest reader and reports what it found. A missing directory is not an error: it yields Present false, which is the WAL-only image shape.
type RefreshExpectation ¶ added in v0.12.0
type RefreshExpectation int
RefreshExpectation states what a StatsRegime.CheckRefresh call may expect from the procedure's ok column, given what the caller knows about the engine's rate-limiter state.
const ( // ExpectRebuild asserts ok=true: the limiter is provably fresh (the first // call on a newly built or newly recovered engine), so a refusal means the // limiter leaked across an engine lifetime. ExpectRebuild RefreshExpectation = iota // ExpectRefusal asserts ok=false: the call is issued back-to-back with a // completed rebuild, inside the rate-limit window, so a rebuild means the // amplification-vector guard did not engage. ExpectRefusal // ExpectEither accepts both outcomes: a mid-run, seed-chosen probe whose // distance from the previous rebuild is wall-clock dependent. The row // shape, the detail/ok agreement, the tracked-pairs observable, and result // identity across the call are still asserted in full. ExpectEither )
type Registry ¶
type Registry struct {
// contains filtered or unexported fields
}
Registry maps scenario names to scenarios and lists them in a stable order. It holds no global mutable state: a Registry is built explicitly with NewRegistry (or DefaultRegistry for the standard catalogue) and is read-only after construction.
Concurrency contract ¶
A Registry is immutable after NewRegistry returns and safe for concurrent reads.
func DefaultRegistry ¶
DefaultRegistry builds the standard Phase-4 scenario catalogue. It holds no global mutable state — every call returns a freshly-built Registry — so two callers never share scenario state. The returned registry lists every scenario named by the Scenario* constants.
The tick/connection budgets here are the SHORT-layer defaults: small enough that the catalogue runs well under the per-package 60s short-test ceiling. The long-running scenario carries a larger budget and is exercised only under the soak layer (see the integration tests).
func NewRegistry ¶
NewRegistry builds a registry from the given scenarios. It returns an error if two scenarios share a name (a programmer error) so the catalogue cannot silently shadow one scenario with another.
func (*Registry) Lookup ¶
Lookup returns the scenario registered under name and whether it was found.
type RequiredCounter ¶ added in v0.12.0
type RequiredCounter struct {
// Name is the exact metric name, as OBSERVED by driving the scenario with a
// recording sink installed — never copied from an inventory.
Name string
// Why states what running the name proves about the fault, in one line.
Why string
// Discriminator is how the declaration stays falsifiable when Uniqueness is
// [CounterSharedWithOtherPaths]: the second signal, or the control arm, that
// separates this fault from the other paths that move the same name. It is
// REQUIRED for a shared counter and the shape gate rejects a declaration that
// omits it.
Discriminator string
// Min is the floor the counter must reach. It is a structural number (one per
// window, one per component) wherever the scenario fixes one, not the value a
// single run happened to produce.
Min uint64
// Uniqueness is how many paths emit Name; see [CounterUniqueness].
Uniqueness CounterUniqueness
}
RequiredCounter is one counter a fault scenario declares its faults must move, with the minimum it must reach and the reason the path emits it.
func (RequiredCounter) String ¶ added in v0.12.0
func (r RequiredCounter) String() string
String renders a required counter for a violation message.
type Result ¶
type Result interface {
// Next advances to the next row and reports whether one is available.
Next() bool
// ScalarInt returns the integer value of the first column of the current
// row. It is only valid after a successful Next.
ScalarInt() (int64, bool)
// IntAt returns the integer value of column i of the current row. It is only
// valid after a successful Next.
IntAt(i int) (int64, bool)
// StringAt returns the string value of column i of the current row. It is
// only valid after a successful Next.
StringAt(i int) (string, bool)
// RowCount reports how many rows the result has produced so far via Next.
RowCount() int
// Err returns any error accumulated during iteration.
Err() error
// Close releases the result.
Close() error
}
Result is the minimal row-iterator the checker needs from a query. It is a thin projection of the engine's real result type, exposing only forward iteration and a single scalar read, which is all the count and existence probes require.
Concurrency contract ¶
A Result is single-use and not safe for concurrent use; drive it from one goroutine and Close it when done.
type Scenario ¶
type Scenario struct {
// Mix is the per-connection role mix for the concurrent/liveness modes. When
// nil, the harness default is used.
Mix *ConcurrentMix
// Workload is the deterministic-mode actor mix factory. When nil,
// [DefaultWorkload] is used. It is a factory (not a built Workload) so each
// run gets a fresh, seed-parameterised mix.
Workload func(*Seed) *Workload
// Checkpoint configures deterministic in-loop checkpointing (ModeDeterministic):
// the store is opened in full-stack mode (WAL + snapshot) and the loop
// publishes a real snapshot + truncates the WAL prefix on the configured
// cadence, so a subsequent crash recovers via the full snapshot+WAL path. The
// zero value disables it. See [CheckpointConfig].
Checkpoint CheckpointConfig
// Name is the stable catalogue key (kebab-case).
Name string
// Description is a one-line human summary of what the scenario stresses,
// printed by the catalogue listing.
Description string
// Checks selects the extra invariant checks.
Checks CheckSelection
// EngineOpts configures the in-memory engine the deterministic loop drives.
// The mem-pressure scenario clamps the logical-resource budgets here; every
// other scenario leaves it zero (byte-identical to the default engine).
EngineOpts cypher.EngineOptions
// Crash configures deterministic crash/recovery injection (ModeDeterministic).
Crash CrashConfig
// Disk, when its CapacityBytes > 0, bounds the SimDisk-backed store to a
// finite size so the run drives the engine through a disk-full (ENOSPC)
// condition (ModeDeterministic). See [DiskConfig].
Disk DiskConfig
// ClampsGOMAXPROCS declares that this scenario WRITES process-global
// `GOMAXPROCS`. [Swarm] runs such a scenario alone, because the clamp would
// otherwise decide the regime of whatever its neighbours drew; every other
// scenario runs under a shared hold. See [gomaxprocsMu] (rmp #2613).
ClampsGOMAXPROCS bool
// Multigraph opens the engine's graph as a directed multigraph
// (ModeDeterministic), so repeated CREATEs between the same endpoints add
// parallel edge instances. Only a scenario whose oracle models edges per
// instance sets it (edge-properties, rmp #2449). See [Config.Multigraph].
Multigraph bool
// SearchEvery is the in-loop cadence (in ticks) for the search battery
// (ModeDeterministic): [runDeterministic] sets it on the simulator. 0 disables
// periodic search checks; the terminal search check still runs when
// Checks.Search is set. See [Simulator.searchEvery] and [CheckSearch].
SearchEvery int
// CheckEvery is the invariant-check cadence in ticks (ModeDeterministic). A
// value <= 0 checks every tick. A long run sets it higher so the per-tick
// full-graph parity probes do not dominate a millions-of-ops workload (the
// scenario's value there is heap/goroutine stability, not per-tick parity).
CheckEvery int
// MaxTicks bounds the deterministic safety loop (ModeDeterministic).
MaxTicks int
// Connections / OpsPerConn bound the concurrent and liveness modes.
Connections int
OpsPerConn int
// DefaultSeed is the seed used when a caller does not supply one.
DefaultSeed uint64
// ConvergeBudget bounds the liveness convergence phase (ModeLiveness).
ConvergeBudget time.Duration
// Mode selects the harness (see [ExecMode]).
Mode ExecMode
// contains filtered or unexported fields
}
Scenario is a named, self-contained simulation configuration: a default seed, a workload mix, a fault/crash schedule, a tick/time budget, the execution mode, and which extra checks to run. A scenario is a pure config — it carries no mutable run state — so the same (scenario, seed) always describes the same run. The registry maps names to scenarios; Scenario.Run executes one.
Concurrency contract ¶
A Scenario value is immutable after construction and safe to read from many goroutines. Scenario.Run itself drives a single run; the concurrent modes it dispatches to spawn and join their own goroutines internally.
func (*Scenario) DeterministicConfig ¶
DeterministicConfig builds the Config for a deterministic-mode run from the scenario plus a resolved seed. It is exported-internal so trace recording and shrinking can build the identical config.
func (*Scenario) Run ¶
Run executes the scenario once with the given seed and returns a report (nil means the scenario passed) or an error for a harness/transport failure that is not itself an invariant violation. It dispatches on Scenario.Mode; a scenario with a custom run override delegates to it.
For the deterministic mode the run is fully reproducible from seed and the returned report (on failure) carries enough to replay and shrink. For the concurrent and liveness modes the run is convergence/leak-guarded and the report, when non-nil, describes the inconsistency found at quiescence.
type ScenarioCounterDecl ¶ added in v0.12.0
type ScenarioCounterDecl struct {
// Scenario names the scenario, and the arm when the scenario has more than
// one fault shape (for example "db-teardown[fault-on-close]").
Scenario string
// BlindReason, when non-empty, declares that NOTHING in the module emits a
// counter for this scenario's fault. Required must then be empty. The claim is
// not taken on trust: BlindPrefix names the metric namespace whose silence is
// ASSERTED, so wiring a counter into that namespace fails the declaration and
// forces it to be updated rather than leaving a stale blindness on record.
BlindReason string
// BlindPrefix is the namespace asserted to stay silent. It is meaningful only
// alongside BlindReason.
BlindPrefix string
// Required is every counter the scenario's faults must move.
Required []RequiredCounter
}
ScenarioCounterDecl is one fault scenario's required-counters declaration: the counters its faults must move, or an explicit statement that the module emits NO counter capable of witnessing them.
A declaration is adjudicated by ScenarioCounterDecl.Check against a MetricsObservation taken across the scenario run. Its own well-formedness is adjudicated separately by CheckCounterDeclShape: an empty or self- contradictory declaration would otherwise pass by saying nothing, which is the vacuity this file exists to prevent.
func ScenarioCounterDecls ¶ added in v0.12.0
func ScenarioCounterDecls() []ScenarioCounterDecl
ScenarioCounterDecls returns every per-scenario required-counters declaration, in a fixed order.
Every name below was obtained by DRIVING the scenario with a recording sink installed and reading what arrived, across the scenario's own spread of seeds; the floors are the structural counts the scenario fixes (one per publish window, one per corrupted component) rather than the value one run produced. Nothing here was copied out of docs/metrics.md.
func (ScenarioCounterDecl) Check ¶ added in v0.12.0
func (d ScenarioCounterDecl) Check(obs MetricsObservation) []Violation
Check adjudicates the declaration against what one scenario run emitted. Every declared counter that did not reach its floor is a violation, and so is a metrics-blind declaration whose namespace turned out to be noisy.
A violation here is a COVERAGE failure: it says the run did not demonstrate the fault path was entered, so the scenario's own verdict rests on nothing. It is therefore reported separately from that verdict.
func (ScenarioCounterDecl) String ¶ added in v0.12.0
func (d ScenarioCounterDecl) String() string
String renders the declaration for a test log.
type ScenarioSelector ¶
type ScenarioSelector interface {
// Select returns the scenario name for the run identified by runIndex. The
// default scenario is supplied so a selector can fall back to it.
Select(runIndex int, defaultScenario string) string
}
ScenarioSelector chooses the scenario name a given swarm run should execute. The coverage tracker implements it to steer runs toward under-covered paths; a nil selector means "always run the configured scenario". Implementations must be safe for concurrent use.
type SchemaChangeFamily ¶
type SchemaChangeFamily int
SchemaChangeFamily identifies one DDL operation the SchemaChanger issues.
const ( // SchemaCreateIndex creates a hash index on (:Person).name. It is an // idempotent op family: IF NOT EXISTS makes a re-create a clean no-op, so // its contract is SUCCESS, never a tolerated failure (rmp #2455). SchemaCreateIndex SchemaChangeFamily = iota // SchemaDropIndex drops the (:Person).name index (idempotent via IF EXISTS). SchemaDropIndex // SchemaCreateConstraint creates a UNIQUE constraint on (:Account).email, // alternating (seed-chosen) between the legacy ON ... ASSERT grammar and // the modern FOR ... REQUIRE grammar — both parsed by // cypher/ir/ddl_parser.go into the same IR. Like SchemaCreateIndex it is // idempotent via IF NOT EXISTS, so its contract is SUCCESS: a re-create is // absorbed as a no-op, never a tolerated "already exists" failure. SchemaCreateConstraint // SchemaDropConstraint drops the (:Account).email UNIQUE constraint // (idempotent via IF EXISTS). SchemaDropConstraint )
Schema-change families.
func (SchemaChangeFamily) String ¶
func (f SchemaChangeFamily) String() string
String renders a SchemaChangeFamily for reports.
type SchemaChangeOutcome ¶
type SchemaChangeOutcome struct {
// Counters is the statement's effect report as the terminal SUCCESS carried
// it on the wire, decoded by [wireDDLCounters]: nil when the SUCCESS carried
// no `stats` map (the server's report of a statement that changed nothing).
// It is set only when Succeeded.
Counters *exec.QueryCounters
FailureMsg string
Family SchemaChangeFamily
Succeeded bool
Failed bool
}
SchemaChangeOutcome records one DDL attempt. A DDL either succeeds or returns a typed FAILURE (e.g. a transient conflict under contention); both are acceptable. A panic, leak, or torn index/lost constraint is a violation, checked structurally after the run rather than per-attempt.
func RunSchemaChurn ¶
func RunSchemaChurn(ctx context.Context, srv *SimServer, seed *Seed, rounds int) ([]SchemaChangeOutcome, error)
RunSchemaChurn drives a SchemaChanger through rounds DDL statements over a single connection, returning the per-round outcomes. Every statement goes through SchemaChanger.RunChecked, so a wire-reported counter that disagrees with the engine's registries ends the churn with an error (rmp #2829); the random family draw makes absorbed forms (a drop of an absent object, a re-create of a present one) part of the stream. It stops early on ctx cancellation. It is the unit the concurrent integration test runs alongside honest writers.
func (SchemaChangeOutcome) Acceptable ¶
func (o SchemaChangeOutcome) Acceptable() bool
Acceptable reports whether the DDL completed cleanly (success or typed FAILURE) without wedging the connection.
func (SchemaChangeOutcome) MeetsContract ¶ added in v0.12.0
func (o SchemaChangeOutcome) MeetsContract() bool
MeetsContract reports whether the outcome satisfies the family's contract. The idempotent IF NOT EXISTS create families (SchemaCreateIndex, SchemaCreateConstraint) must SUCCEED — a re-create is a clean no-op by the engine's IF NOT EXISTS contract (cypher: TestCreateConstraint_IfNotExists_ Idempotent), so a typed FAILURE there is a contract breach, not an acceptable bounded outcome (rmp #2455). The drop families keep the tolerant contract (success or typed FAILURE under contention).
type SchemaChanger ¶
type SchemaChanger struct{}
SchemaChanger issues DDL (CREATE/DROP INDEX, CREATE/DROP CONSTRAINT) over the real Bolt wire, concurrently with honest writers and readers, to exercise index and constraint maintenance under races. Every statement is idempotent (IF [NOT] EXISTS) so a create/drop race never produces a spurious error; the invariants the harness asserts after the churn are that the index stays consistent with its base data and that a UNIQUE constraint, when present, stays enforced.
SchemaChanger runs in the CONCURRENT mode (one goroutine), so its DDL interleaves non-deterministically with concurrent writes; correctness is the structural invariants at quiescence, not bit-replay.
Concurrency contract ¶
SchemaChanger is stateless; each SchemaChanger.Run call drives one connection it owns and may run on its own goroutine.
func (SchemaChanger) PickFamily ¶
func (SchemaChanger) PickFamily(seed *Seed) SchemaChangeFamily
PickFamily chooses a DDL family from the seed (one int draw).
func (SchemaChanger) PickModernForm ¶ added in v0.12.0
func (SchemaChanger) PickModernForm(seed *Seed) bool
PickModernForm chooses (one int draw) whether the next constraint DDL uses the modern FOR ... REQUIRE grammar (true) or the legacy ON ... ASSERT grammar (false), so the churn exercises both parse paths under the same seed-deterministic stream (rmp #2455). Only SchemaCreateConstraint consults the flag; the other families have a single grammar.
func (SchemaChanger) Run ¶
func (a SchemaChanger) Run(c *WireClient, family SchemaChangeFamily, modern bool) (SchemaChangeOutcome, error)
Run issues one DDL statement of the given family over c and returns the classified outcome. modern selects the constraint grammar for SchemaCreateConstraint (see SchemaChanger.PickModernForm); the other families ignore it. The connection must already be Connected.
func (SchemaChanger) RunChecked ¶ added in v0.16.0
func (a SchemaChanger) RunChecked(srv *SimServer, c *WireClient, family SchemaChangeFamily, modern bool) (SchemaChangeOutcome, []Violation, error)
RunChecked is SchemaChanger.Run with the statement's wire-reported counters ADJUDICATED (rmp #2829): it snapshots srv's engine schema registries before and after the statement and holds the counters the terminal SUCCESS carried to their difference with CheckDDLCounters — the same expectation the in-process DDL route ([runDDLChecked], rmp #2822) applies. A counter that lies anywhere between the engine's operator and the Bolt `stats` encoding is returned as a violation. A typed FAILURE is not adjudicated, as on the in-process route: a failed statement applied no effect to report.
The registry difference is attributable to this statement only while no other DDL runs against srv's engine; data writes never change the registries. Every caller in this package issues DDL from a single connection.
type SchemaModel ¶ added in v0.12.0
type SchemaModel struct {
// contains filtered or unexported fields
}
SchemaModel is the harness's own model of every index and constraint the scenario has issued. Scenarios mutate it in lock-step with the DDL they run (AddIndex on CREATE INDEX, DropConstraint on DROP CONSTRAINT, ...) and CheckSchemaIntrospection holds the engine's introspection surfaces to it.
A UNIQUE constraint implies its hash backing index ("__uniq__<label>.<prop>"), which the engine lists in SHOW INDEXES / db.indexes(); the model derives that row automatically, so scenarios only declare what they issued.
Concurrency contract ¶
SchemaModel is NOT safe for concurrent use; it is driven from the single simulation goroutine.
func NewSchemaModel ¶ added in v0.12.0
func NewSchemaModel() *SchemaModel
NewSchemaModel returns an empty schema model.
func (*SchemaModel) AddIndex ¶ added in v0.12.0
func (m *SchemaModel) AddIndex(name, kind, label, prop string)
AddIndex records a user index the scenario created. kind is SchemaIndexHash or SchemaIndexBTree; label/prop are the indexed (label, property) pair recorded by the engine's index-def registry.
func (*SchemaModel) AddNotNullConstraint ¶ added in v0.12.0
func (m *SchemaModel) AddNotNullConstraint(name, label, prop string)
AddNotNullConstraint records a NOT NULL (existence) constraint the scenario created. Existence constraints have no backing index.
func (*SchemaModel) AddUniqueConstraint ¶ added in v0.12.0
func (m *SchemaModel) AddUniqueConstraint(name, label, prop string)
AddUniqueConstraint records a UNIQUE constraint the scenario created. Its hash backing index row is derived automatically.
func (*SchemaModel) DropConstraint ¶ added in v0.12.0
func (m *SchemaModel) DropConstraint(name string)
DropConstraint removes a constraint from the model (a DROP CONSTRAINT was issued). For a UNIQUE constraint the derived backing-index row disappears with it, matching the engine's by-name drop (#1556), which removes the "__uniq__" backing index alongside the constraint.
func (*SchemaModel) DropIndex ¶ added in v0.12.0
func (m *SchemaModel) DropIndex(name string)
DropIndex removes a user index from the model (a DROP INDEX was issued).
type SchemaMutationWriter ¶ added in v0.8.0
type SchemaMutationWriter struct{}
SchemaMutationWriter mutates the labels and properties of existing Person nodes: it removes a property (REMOVE n.tag), removes a label (REMOVE n:Vip), adds a label (SET n:Vip), merges a property map (SET n += $props), and replaces the property set (SET n = $props). Since rmp #2454 it also drives the FOREACH write path with two genuinely different bodies: a per-element CREATE over a bound list ([tmplForeachCreatePersons] — the one op family through which this writer creates nodes) and a per-element SET on an outer MATCH variable ([tmplForeachSetTag]).
Since rmp #2461 it additionally drives the four MERGE families the DST did not reach: a node MERGE with BOTH action branches ([tmplMergePersonCounter]), the whole-map ON CREATE assignment ([tmplMergePersonSetAll]), a whole-pattern MERGE that creates both endpoints and the relationship together ([tmplMergePairPattern]), and the map-parameter MERGE the engine must REJECT ([tmplMergeParamMap]). Since rmp #2510 it also drives the whole-entity action on a pattern's relationship variable ([tmplMergePairSetAll]), since rmp #2511 the two actions whose target is bound by the clause PRECEDING the MERGE — a node ([tmplMergePairOuter]) and a relationship ([tmplMergePairOuterRel]) — and since rmp #2512 the two MERGE forms behind a driving clause that binds NOTHING ([tmplMergeZeroDriverNode], [tmplMergeZeroDriverPair]), which must therefore change nothing at all. Since rmp #2515 it also drives the NODE-ONLY MERGE whose branch action targets a relationship bound by the preceding clause ([tmplMergeHandleOuterRelCreate], [tmplMergeHandleOuterRelMatch]) — the shape whose write was misdirected onto an unrelated node. That family REQUIRES the handle/id collision [seedMergeHandleCollision] constructs, so a workload that mixes this writer in without running that bootstrap first would drive its driving MATCH against nothing; only [schemaMutationWorkload] uses this writer, and [runSchemaMutationCfg] runs the bootstrap before its first tick. The first three create nodes, so the writer is no longer create-free; the last is the one statement it emits deliberately expecting an engine error, and it is modelled as an OpMalformed no-op for exactly that reason. Every other op targets a name the oracle already models, so it emits nothing the engine would reject on well-formedness grounds. The co-actor HonestWriter keeps the population churning and is still the only actor that deletes.
Concurrency contract ¶
SchemaMutationWriter is NOT safe for concurrent use; it is invoked from the single simulation goroutine.
func (SchemaMutationWriter) Name ¶ added in v0.8.0
func (SchemaMutationWriter) Name() string
Name returns the actor's identifier.
func (SchemaMutationWriter) NextOp ¶ added in v0.8.0
func (w SchemaMutationWriter) NextOp(seed *Seed, oracle *GraphOracle) Op
NextOp picks a mutation on a seed-chosen existing Person. When the graph is empty it creates a Person (nothing to mutate yet), so the op stream is a pure function of (seed state, oracle state).
type ScriptedResult ¶
type ScriptedResult struct {
Report *SimReport
NodeCount int64
EdgeCount int64
OracleN int
OracleE int
}
ScriptedResult is the outcome of a scripted replay: the report (nil when the replay found no violation) and the engine/oracle end-state counts, so a caller can assert two runs reach the identical end-state.
func ReplayTrace ¶
func ReplayTrace(ctx context.Context, trace Trace) (ScriptedResult, error)
ReplayTrace executes a recorded Trace against a FRESH plain in-memory engine, oracle, and checker — WITHOUT drawing from any seed — applying each op in order and checking invariants after every op exactly as the deterministic safety loop does. It is the foundation for both exact failure replay and shrinking: because the deterministic engine-API mode is a pure function of its op stream, replaying the stream reproduces the same end-state and re-triggers the same violation.
An op carrying an injected TraceFault is applied with that fault (e.g. FaultDropEngineWrite applies the write to the oracle but skips the engine, producing a deterministic divergence). A trace with no faults replays cleanly iff the original run did.
ReplayTrace spawns no goroutines and is a pure function of trace.
func (ScriptedResult) Violated ¶
func (r ScriptedResult) Violated() bool
Violated reports whether the scripted replay detected a violation.
type Seed ¶
type Seed struct {
// contains filtered or unexported fields
}
Seed is the single source of randomness for an entire simulation. Every probabilistic decision — which actor runs, which operation it emits, the parameter values it binds, and whether the disk injects a fault — draws from one Seed, so the complete simulation is a pure function of the seed value.
Seed wraps a deterministic PCG generator (math/rand/v2.PCG) seeded from the seed value alone; it never consults the operating system, a global generator, or the wall clock. The original value is retained so it can be reported and replayed.
Concurrency contract ¶
Seed is NOT safe for concurrent use. It backs the single-goroutine simulation loop and its draw order is load-bearing for reproducibility; concurrent draws would interleave non-deterministically and break replay.
func NewSeed ¶
NewSeed returns a Seed whose generator is initialised deterministically from val. The two PCG stream words are val and val^seedMix, so distinct seed values yield distinct generator states.
func (*Seed) Bool ¶
Bool returns true with probability p and false otherwise. p is clamped to [0.0, 1.0]: p <= 0 always returns false, p >= 1 always returns true. Each call consumes exactly one float64 draw, keeping the draw stream stable regardless of p.
func (*Seed) IntN ¶
IntN returns a uniform integer in [0, n). It panics if n <= 0, mirroring the contract of math/rand/v2.Rand.IntN.
func (*Seed) Pick ¶
Pick returns a uniformly-chosen element of items. It panics if items is empty, which signals a programmer error (a workload that offers no choices).
func (*Seed) Shuffle ¶
Shuffle returns a new slice holding the elements of items in a deterministically-shuffled order, using an in-place Fisher–Yates pass over the copy. The input slice is never mutated. The result is a function of the generator state alone, so the same seed produces the same permutation.
func (*Seed) Uint64N ¶
Uint64N returns a uniform unsigned integer in [0, n). It panics if n == 0, mirroring the contract of math/rand/v2.Rand.Uint64N.
type ShrinkConfig ¶
type ShrinkConfig struct {
// MaxIterations caps the scripted-replay attempts. A non-positive value uses
// [defaultMaxShrinkIterations].
MaxIterations int
}
ShrinkConfig parameterises trace shrinking. The zero value is valid and uses the defaults.
type ShrinkResult ¶
type ShrinkResult struct {
// Violation is a representative violation the minimal trace reproduces (the
// first one found on the final replay), for the report.
Violation Violation
// Minimal is the reduced trace that still reproduces the target violation.
Minimal Trace
// OriginalLen / MinimalLen are the op counts before and after shrinking.
OriginalLen int
MinimalLen int
// Iterations is the number of scripted-replay attempts performed.
Iterations int
}
ShrinkResult is the outcome of shrinking: the minimal trace found, the violation it still reproduces, and the work the shrinker did.
func ShrinkTrace ¶
func ShrinkTrace(ctx context.Context, trace Trace, cfg ShrinkConfig) (ShrinkResult, error)
ShrinkTrace reduces a failing trace to a (near-)minimal subsequence that still reproduces the SAME violation, via delta-debugging (ddmin). It first confirms the full trace fails under scripted replay and captures the target violation signature, then repeatedly partitions the op sequence into n chunks and tries (a) removing each chunk and (b) keeping each chunk's complement, accepting the smallest candidate whose replay still reproduces the target signature and increasing the granularity otherwise — the classic Zeller–Hildebrandt ddmin.
Determinism: every step is a scripted ReplayTrace (no seed draws, no goroutines, no wall clock), so shrinking the same failing trace always yields the same minimal trace. Boundedness: the search is capped at cfg.MaxIterations replay attempts; the best reduction found by then is returned.
Cross-op dependencies are preserved implicitly: a candidate that drops an op the violation depends on (e.g. the lost-write CREATE itself, or a node a later edge needs) replays to a DIFFERENT signature (or to no violation), so ddmin rejects it and keeps the op. No explicit reference repair is required because the "still reproduces the same violation" oracle is exact.
It returns an error only when the input trace does NOT reproduce a violation under scripted replay (there is nothing to shrink), or when ctx is cancelled.
func (*ShrinkResult) Ratio ¶
func (r *ShrinkResult) Ratio() float64
Ratio returns the reduction factor (original / minimal); 1 means no reduction. It is reported so a caller can assert an orders-of-magnitude shrink.
type SimConn ¶
type SimConn struct {
// contains filtered or unexported fields
}
SimConn is one end of an in-memory, bounded-buffer net.Conn pair. It carries the real Bolt wire bytes between the in-sim client harness and the genuine bolt/server, with no OS socket, so the server's actual handshake, framing, and message loop run unchanged.
A SimConn pair supports two usage modes that share one implementation:
- LOCK-STEP single-connection mode: the client writes a complete request and then blocks reading the server's full terminal response. Because the bounded buffer is far larger than any single request/response and exactly one logical exchange is in flight, the byte stream is fully deterministic and a given seed replays identically. Used for protocol round-trips and the BoltAbuser so violations reproduce exactly.
- CONCURRENT mode: one real goroutine drives each end. Interleaving across connections is non-deterministic, but each end's reads and writes are individually safe and the bounded buffer applies backpressure (a stalled reader parks the peer's writer), which is what the SlowConsumer and overload actors exercise.
Deadlines route through the injected clock.Clock; with clock.Real a SimConn behaves like an ordinary socket, and with a clock.Fake a deadline fires only when virtual time is advanced.
Concurrency contract ¶
A single SimConn end is safe for use by one reader goroutine and one writer goroutine concurrently (the full-duplex net.Conn contract). It is NOT safe to share one end across multiple readers or multiple writers.
func NewSimConnPair ¶
NewSimConnPair returns the two ends of a connected in-memory pipe. Bytes written to one end are read from the other. Both ends share the injected clock for deadline handling; pass clock.Real for ordinary timing or a clock.Fake for deterministic virtual deadlines. clk must be non-nil.
The conventional use is to hand the server end to the bolt/server (via SimListener) and drive the client end with the wire client harness.
func (*SimConn) Close ¶
Close implements net.Conn.Close. It closes both directions of this end so a blocked peer read returns io.EOF and a blocked peer write returns ErrSimConnClosed. It is idempotent.
func (*SimConn) CloseWithError ¶
CloseWithError closes this end abruptly, delivering err to the peer's blocked reads and writes instead of a clean io.EOF. It models a connection reset (an abrupt client disconnect mid-stream) so the harness can assert the server neither panics nor leaks a goroutine on a hard close. It is idempotent.
func (*SimConn) LocalAddr ¶
LocalAddr implements net.Conn.LocalAddr.
func (*SimConn) ReadBuffered ¶
ReadBuffered reports how many bytes the peer has written that this end has not yet read. It never exceeds [simConnBufferSize]; the bound holding under a stalled reader is the backpressure property the SlowConsumer asserts.
func (*SimConn) RemoteAddr ¶
RemoteAddr implements net.Conn.RemoteAddr.
func (*SimConn) SetDeadline ¶
SetDeadline implements net.Conn.SetDeadline, setting both the read and write deadlines to t. A zero t clears the deadlines.
func (*SimConn) SetReadDeadline ¶
SetReadDeadline implements net.Conn.SetReadDeadline.
func (*SimConn) SetWriteDeadline ¶
SetWriteDeadline implements net.Conn.SetWriteDeadline.
type SimDisk ¶
type SimDisk struct {
// contains filtered or unexported fields
}
Fidelity limits of the SimDisk crash model (rmp #2514).
The model is deliberately narrower than a real filesystem. The notes below record, per operation, whether a crash can lose an effect that no real filesystem loses — the class of infidelity that makes the simulator ACCUSE the engine of a defect it does not have — or, conversely, whether it keeps an effect a real crash could take away, which lets an engine defect pass. The first class is a harness bug; the second is missing coverage. Both are recorded so a scenario author knows what the harness does and does not model.
This is a summary of the limits that bear on CRASH OUTCOMES, written where the code is; it is not the complete audit of the type. In particular it does not cover the operations whose modelling gap is that they cannot FAIL (a Sync retry always succeeds after a fault, a write is never partial), which bound which engine error branches a scenario can reach rather than which durable images a crash can leave.
Rename FIXED. A crash used to revoke the new directory entry while
the old one had already been unlinked, losing BOTH names —
an outcome rename(2)'s atomicity forbids, and the one that
made the snapshot publish protocol look as if a crash
between its two renames could destroy every copy of the
graph. A crash now rolls the rename back instead, so the
outcome is always one of the two legal ones. See
[SimDisk.Crash].
Remove FIXED (rmp #2536). An unlink is metadata, in the same
RemoveAll durability class as the create and the rename above: until
the parent directory is fsync'd a crash may legally leave the
removed name in place. The model applied the unlink
immediately and [SimDisk.Crash] could never restore it, so
the harness only ever presented the EASIER input to
recovery — a stale "<dir>/snapshot.bak" surviving the publish
path's happy-path cleanup, and a stale "<dir>/snapshot.tmp"
surviving recovery's own best-effort staging cleanup, were
both unreachable from the removal side although recovery is
written to tolerate them. A removal now records an undo (see
[unlinkUndo]) in the SAME ordered log as a rename, so one
durable-prefix draw adjudicates both and no crash can keep an
unlink while discarding a rename issued before it.
Truncate NOT MODELLED, coverage gap (never a false accusation).
TruncatePath A shrink is applied immediately and a crash never restores
O_TRUNC the previous length, so the harness cannot present a WAL
whose prefix truncation was lost. Recovery would then have
to re-read frames a checkpoint had already folded, which is
exactly the idempotence it claims. The durable shadow added
for rmp #2535 leaves this unchanged on purpose — a shrink
lowers the durable image too ([simFile.truncateDurableTo]) —
and that one function is the seam rmp #2542 needs to make a
truncation itself survivable.
Write (data) FIXED (rmp #2535). Bytes written but never Sync'd used to
survive [SimDisk.Crash] intact, whereas a real crash loses
whatever never left the page cache — so every "the commit
was acked, therefore the bytes are recovered" assertion held
irrespective of any fsync, and deleting the WAL commit fsync
failed no scenario. Each file now carries a durable image
advanced ONLY by a Sync that returns nil, and the crash
primitive is split: [SimDisk.CrashHost] reverts to that
image (power failure), [SimDisk.CrashProcess] keeps the
bytes and the names (SIGKILL). [SimDisk.Crash] is CrashHost.
MkdirAll MODELLED since rmp #2543. A created directory is registered
in the dirent model and is durable only once its parent is
fsynced, like any other name; an EMPTY directory is
representable and both Stat and Exists report it. It used to
be a no-op, so a directory creation could never be lost and a
partial-publish state was unreachable. MkdirAll over an
EXISTING directory still changes nothing, because it creates
nothing and needs no fsync.
Remove(dir) MODELLED since rmp #2545. Remove follows os.Remove, which
tries unlink and then rmdir: an EMPTY directory is removed
and a NON-EMPTY one returns ENOTEMPTY. It used to succeed
silently on both, so a caller that meant to delete a
directory got success and no deletion. Remove on an ABSENT
path still succeeds, deliberately — the snapshot writer's
best-effort cleanup relies on it.
Truncation MODELLED since rmp #2542. O_TRUNC, the handle Truncate and
TruncatePath all shorten the VISIBLE data at once and the
DURABLE image only at the next successful fsync, so a crash
in between restores the longer prior generation. All three go
through [simFile.truncatePendingTo], so they cannot drift.
They also drop the fault marks for sectors past the new end
of file uniformly (F13); TruncatePath used to keep them.
Sync failure MODELLED since rmp #2540. A failed fsync freezes the file's
durable prefix for good: every later Sync RETURNS SUCCESS and
advances nothing, because a failed write-back marks the pages
clean and drops them (Rebello et al., USENIX ATC 2020;
PostgreSQL's fsyncgate response was to PANIC rather than
retry). It used to leave the data intact so a retry simply
worked, which would have made a "retry once before poisoning"
optimisation look safe under simulation.
DirSync FIXED (rmp #2537), and NARROWER THAN REALITY in the safe
direction. It could not FAIL: the body ended in an
unconditional return nil while its sibling ParentDirSync
already consulted an arm, so four snapshot staging-fsync
error branches were unreachable under simulation — including
the last durability gate before the publish renames. Both
entry points now share ONE body ([SimDisk.dirSyncLocked]) and
one arm mechanism ([dirSyncArm]), so neither the fault nor
its effects can diverge again. What stays narrower is the
SUCCESS path: it durabilises only the entries whose parent is
exactly dir, whereas fsync(2) on a directory forces a journal
commit that makes earlier metadata durable filesystem-wide.
Fewer things become durable than would in reality, so the
harness is stricter, never more forgiving.
SimDisk is an in-memory filesystem with seed-driven fault injection. It backs the durability layer of the simulation: files live entirely in memory, and a per-sector fault bitmap plus a per-Sync fault probability let the simulator reproduce torn writes and failed flushes deterministically.
SimDisk is built in Phase 1 but is not yet wired into the engine (that is Phase 2 work); it must compile, implement the WAL file interface, and be unit-tested standalone.
Concurrency contract ¶
SimDisk's directory operations are guarded by an internal mutex so the file table cannot be corrupted, but the simulation drives it from a single goroutine and the fault decisions draw from the shared single-goroutine Seed; it must not be used concurrently.
The "Sim" prefix is part of the DST harness's deliberate naming scheme (SimDisk / SimFileHandle / SimReport), which reads clearly at call sites and matches the design specification; the apparent stutter is intentional.
func NewSimDisk ¶
NewSimDisk returns an empty in-memory filesystem. faultRate is the probability (clamped to [0,1]) that any individual Sync fails with ErrSimFault and that a freshly written sector is marked faulted. seed drives every fault decision so the fault sequence is reproducible.
func (*SimDisk) AppendCount ¶ added in v0.12.0
AppendCount returns the number of growing writes performed across every handle of this disk since the last SimDisk.ArmTornAppendAt reset (or since construction). A scenario reads it to choose which append to tear.
func (*SimDisk) ArmDirSyncFaultForPath ¶ added in v0.12.0
ArmDirSyncFaultForPath arms a ONE-SHOT durability fault on SimDisk.DirSync: the next directory fsync of dir returns ErrSimFault and makes NO dirent durable, then the arm clears so no further directory fsync is affected. An empty dir disarms. SimDisk.DirSyncFaultCount reports how many times a directory-fsync fault has fired.
Because a SimDisk.ParentDirSync of a child of dir IS an fsync of dir, it fires for that call too; the converse does not hold, since SimDisk.ArmParentDirSyncFaultForPath names one specific child (see the field docs on SimDisk for why the two keys stay distinct).
It is the arm the snapshot publish protocol's directory fsyncs need. Audit finding F3 named four error branches that no fault could reach; this arm reaches the two on the full publish path — store/snapshot/full.go's staging fsync in writeCaptureCore and its indexes/ fsync in writeCapturedIndexes — through a real checkpoint. The remaining two (writer.go's legacy CSR staging fsync and indexes.go's standalone index writer) bind the OS backend at their entry points, so no in-memory disk can reach them however it is faulted; they are driven from inside store/snapshot instead.
The staging fsync is the one that matters most: it is the LAST durability gate before the archive and publish renames, so it must RemoveAll the staging tree and abort rather than publish a directory whose dirents never reached stable storage. It draws nothing from the Seed, so arming never perturbs the reproducible fault stream, and must be called from the controlling goroutine before the operation that will trigger it.
func (*SimDisk) ArmParentDirSyncFaultForPath ¶ added in v0.8.0
ArmParentDirSyncFaultForPath arms a ONE-SHOT durability fault on SimDisk.ParentDirSync: the next call whose childPath equals childPath returns ErrSimFault and makes NO dirent durable, then the arm clears so no further ParentDirSync is affected. An empty childPath disarms.
It is the same primitive as SimDisk.ArmDirSyncFaultForPath — one directory fsync, one shared body, one fire count — differing only in being keyed on the exact childPath rather than on the directory, so it targets a specific fsync robustly where several fsyncs of the SAME parent directory occur in one operation (see the field docs on SimDisk). It models, for example, the post-rename parent-directory fsync of the WAL control file failing inside a checkpoint (wal.Writer.MarkCheckpoint): that failure must fail the checkpoint while the log — and any snapshot published before it — stays intact and recoverable. It draws nothing from the Seed, so arming never perturbs the reproducible fault stream, and must be called from the controlling goroutine before the operation that will trigger it.
func (*SimDisk) ArmRemoveRollbackForPath ¶ added in v0.12.0
ArmRemoveRollbackForPath arms a ONE-SHOT selection of the RESTORED branch of an unlink's crash window: the next SimDisk.Remove or SimDisk.RemoveAll of path is pinned so that a subsequent SimDisk.Crash puts the name — or the whole removed subtree — back, each entry with the dirent durability it carried at the instant of the removal. The arm then clears. An empty path disarms. SimDisk.RemoveRollbackCount reports how many times it fired.
It is the arm that reaches the engine's leftover-artefact tolerance FROM THE REMOVAL SIDE, which no other primitive can: recovery's best-effort RemoveAll("<dir>/snapshot.tmp") and the publish path's stale-backup RemoveAll("<dir>/snapshot.bak") both exist to tolerate an artefact a previous crash left behind, and before rmp #2536 the harness could only reach that state by crashing BEFORE the removal, never by the removal failing to reach disk — which is the case a real deployment hits. SimDisk.RemoveHitCountForPath is how a scenario then proves the tolerant cleanup really found something rather than running as a no-op.
Because the durable metadata mutations of a journalling filesystem are always a prefix of the issued ones, pinning a removal to the restored branch also rolls back every LATER pending record, rename or removal alike; it never rolls back an earlier one.
It draws nothing from the Seed or from the dirent sub-stream, and must be called from the controlling goroutine before the removal it targets.
func (*SimDisk) ArmRemoveWritebackForPath ¶ added in v0.12.0
ArmRemoveWritebackForPath arms a ONE-SHOT unlink WRITE-BACK: the next SimDisk.Remove or SimDisk.RemoveAll of path is treated as having reached stable storage immediately, so no SimDisk.Crash can bring the name back. The arm then clears. An empty path disarms. SimDisk.RemoveWritebackCount reports how many times it fired.
It pins the "the removal stuck" branch of the crash window, and is the exact counterpart of SimDisk.ArmRemoveRollbackForPath. Neither is needed for the model to be sound — an unarmed removal gets a seed-chosen outcome from the same two — but a test that must assert ONE of them unconditionally needs the choice pinned rather than sampled. Pinning it here rather than by fsyncing the parent directory is what keeps the rest of the scenario's dirents untouched: a SimDisk.DirSync would harden every name in that directory too, so the assertion would no longer be about the removal.
Firing it also declares every mutation still pending at that instant durable (see [SimDisk.pinAllDirentUndosLocked]): an unlink that reached the journal implies everything issued before it did, so pinning the prefix is what stops this arm from producing the one interleaving the shared undo log exists to forbid — an unlink kept while a rename issued before it is rolled back.
It draws nothing from the Seed or from the dirent sub-stream, so arming never perturbs either reproducible stream, and it must be called from the controlling goroutine before the removal it targets.
func (*SimDisk) ArmRenameFaultForPath ¶ added in v0.12.0
ArmRenameFaultForPath arms a ONE-SHOT fault on SimDisk.Rename: the next rename whose DESTINATION equals newPath returns ErrSimFault and moves nothing, then the arm clears so no further rename is affected. An empty newPath disarms. SimDisk.RenameFaultCount reports how many times it fired.
It is the rename analogue of SimDisk.ArmSyncFaultAt and SimDisk.ArmParentDirSyncFaultForPath, and closes the last gap in the snapshot publish protocol's fault surface: every other step of
write+fsync components -> fsync staging dir -> archive rename -> publish rename -> fsync parent
could already be made to fail under simulation, but the two renames could not, so the publish path's own archive-restore branch (store/snapshot, the best-effort Rename(bak, dir) after a failed publish rename) was unreachable. Arming the publish destination ("<dir>/snapshot") exercises that restore; arming the archive destination ("<dir>/snapshot.bak") aborts the publish before the live snapshot is touched at all.
The fault fires before the source path is looked up, so it models the rename(2) call failing (EIO), not a missing source. It draws nothing from the Seed, so arming never perturbs the reproducible fault stream, and it must be called from the controlling goroutine before the operation that will trigger it. It defaults disarmed: a SimDisk that never arms it behaves exactly as before.
func (*SimDisk) ArmRenameRevokeBothForPath ¶ added in v0.12.0
ArmRenameRevokeBothForPath arms a ONE-SHOT PHYSICALLY IMPOSSIBLE crash outcome on the next rename whose DESTINATION equals newPath: the crash revokes the new name AND does not restore the old one, so both names are lost. The arm then clears. An empty newPath disarms. SimDisk.RenameRevokeBothCount reports how many times it fired.
No filesystem produces this outcome. rename(2) is atomic, so a crash leaves either the new name or the old one; losing both would mean the file was unlinked, which is not a partial outcome of a rename. It was nevertheless the simulator's DEFAULT until rmp #2514 — SimDisk.Crash revoked every un-fsync'd dirent and nothing put the source name back — which made the snapshot publish protocol look as if a crash between its two renames could destroy both copies of the graph, and made recovery's promote repair unreachable under simulation.
It survives as an explicitly armed fault for exactly one purpose: testing the harness itself — proving that an oracle which would have accepted the impossible outcome now rejects it. Never arm it to reach a durable state a scenario needs; use SimDisk.ArmRenameRollbackForPath for that.
It draws nothing from the Seed or from the rename sub-stream, and must be called from the controlling goroutine before the rename it targets.
func (*SimDisk) ArmRenameRollbackForPath ¶ added in v0.12.0
ArmRenameRollbackForPath arms a ONE-SHOT selection of the ROLLED-BACK branch of a rename's crash window: the next rename whose DESTINATION equals newPath is pinned so that a subsequent SimDisk.Crash undoes it — the new name goes away and the OLD name comes back with the dirent durability it had before the rename. The arm then clears. An empty newPath disarms. SimDisk.RenameRollbackCount reports how many times it fired.
It is the exact counterpart of SimDisk.ArmRenameWritebackForPath: together the two pin the two outcomes a crash immediately after rename(2) can produce. Neither is needed for the model to be sound — an unarmed rename gets a seed-chosen outcome from the same two — but a test that must assert ONE of them unconditionally needs the choice pinned rather than sampled.
Because the durable renames of a journalling filesystem are always a prefix of the issued ones, pinning a rename to the rolled-back branch also rolls back every LATER pending rename; it never rolls back an earlier one.
It draws nothing from the Seed or from the rename sub-stream, so arming never perturbs either reproducible stream, and it must be called from the controlling goroutine before the rename it targets.
func (*SimDisk) ArmRenameWritebackForPath ¶ added in v0.12.0
ArmRenameWritebackForPath arms a ONE-SHOT dirent WRITE-BACK on SimDisk.Rename: the next rename whose DESTINATION equals newPath links that destination with an ALREADY-DURABLE directory entry, as if the kernel had written the rename back to stable storage before the crash. The arm then clears. An empty newPath disarms. SimDisk.RenameWritebackCount reports how many times it fired.
It exists because a crash immediately after a rename has TWO legal outcomes on a real filesystem — the rename reached stable storage, or it did not — and SimDisk.Crash models only the second (every not-yet-fsync'd dirent is revoked). That is sound but incomplete, and the incompleteness is load-bearing: the snapshot publish protocol issues two renames back to back with no fsync between them, so under the revoke-everything model a crash in the publish window drops BOTH the archived backup and the newly published snapshot. No real filesystem produces that outcome, and it is precisely the outcome recovery's interrupted-publish repair (store/recovery, promoting a stranded "<dir>/snapshot.bak" back to the live name) exists to handle — so without this arm that repair path is unreachable under simulation.
Arming it on the archive destination therefore selects the "the archive rename reached disk, the publish rename did not" branch of the window, which is what strands a backup for recovery to promote.
Firing it also declares every mutation still pending at that instant durable (see [SimDisk.pinAllDirentUndosLocked]), because metadata reaches the journal in order. Without that, arming the LATER rename of a pair — the publish rather than the archive — left the crash free to roll the earlier one back while keeping this one, a durable prefix with a hole in it that no filesystem produces. Arming the later step is therefore no longer a way to construct an impossible interleaving; it simply pins more of the window than intended, which SimDisk.PendingRenameCount makes visible.
It draws nothing from the Seed, so arming never perturbs the reproducible fault stream, and it must be called from the controlling goroutine before the rename it targets. It defaults disarmed: a SimDisk that never arms it behaves exactly as before.
func (*SimDisk) ArmSyncFaultAt ¶ added in v0.8.0
ArmSyncFaultAt schedules a ONE-SHOT durability fault: the at-th SimFileHandle.Sync call on this disk (counting from the current [SimDisk.SyncCount]+1) returns ErrSimFault, then the arm clears so no further Sync is affected. It resets the Sync counter to zero so `at` is counted from the moment of arming, letting a scenario arm "the K-th commit's fsync" right before it starts issuing. A non-positive at disarms.
It models an fsync failure on a chosen durable commit: the faulted Sync poisons the WAL writer, which discards its un-synced suffix and fails the commit (the client sees a wire FAILURE, never an ack) while every earlier acked commit stays durable.
The faulted commit is not necessarily the ONLY one that fails. A commit's WAL fsync is a wal.Writer.SyncGroup round, and through the engine that round may carry followers (measured — see the group-commit note in durable_scenarios.go), in which case the poison fails every member of the group by design. This doc claimed the round was "always a solo leader" until rmp #2471; it is not, and a scenario must therefore treat the set of failed commits as a set rather than a singleton. It draws nothing from the Seed, so arming never perturbs the reproducible fault stream. It must be called from the controlling goroutine before the workload that will trigger it begins.
func (*SimDisk) ArmSyncGateAt ¶ added in v0.12.0
ArmSyncGateAt arms a ONE-SHOT rendezvous on the at-th SimFileHandle.Sync call on this disk, counted from the current [SimDisk.SyncCount]+1, and returns the gate. That Sync blocks — outside the disk lock — until SyncGate.Release; every other Sync is unaffected and the arm clears once it fires. A non-positive at disarms and returns nil.
Ordering against ArmSyncFaultAt ¶
Unlike SimDisk.ArmSyncFaultAt this does NOT reset the Sync counter, so that arming a gate never moves a fault ordinal already in place. To gate and fail the SAME Sync, arm the fault FIRST (it resets the counter) and the gate second, with the same ordinal.
It draws nothing from the Seed, so arming never perturbs the reproducible fault stream, and must be called from the controlling goroutine before the work that will trigger it begins.
func (*SimDisk) ArmTornAppendAt ¶ added in v0.12.0
ArmTornAppendAt schedules a ONE-SHOT TORN APPEND: the at-th growing write on this disk (counting from the current append total) is marked so that the next crash leaves only its first keepBytes intact, with the remainder of that record filled with garbage. A non-positive at disarms.
Why this cannot emerge from the durable shadow ¶
A truncated trailing frame is the single most important crash state a write-ahead log must survive, and before this the simulator could not produce one by any route (audit finding F7, rmp #2541). The durable shadow added for F1 does not deliver it: SimDisk.CrashHost reverts each file to data[:durableLen], and durableLen moves only on a SUCCESSFUL Sync — so a crash truncates a file at its last sync boundary, which for a WAL is a FRAME boundary. That is a cleanly LOST trailing frame, never a partial one, and the two exercise different branches of recovery: a reader that handles a clean truncation can still mishandle a frame whose header declares one length and whose body stops short.
The short-count contract in SimFileHandle.Write makes a partial record emergent under a full disk; this arm makes one reachable at any capacity and at a chosen record, which is what a targeted recovery scenario needs.
The kept bytes are the real payload and the remainder is GARBAGE rather than zeroes, per the ALICE finding that zero-fill is an unrealistically benign model of an interrupted write-back. It draws nothing from the seed, so it never perturbs the reproducible fault stream.
func (*SimDisk) CorruptRange ¶ added in v0.8.0
CorruptRange deterministically corrupts n bytes of the ALREADY-DURABLE image of the file at path, starting at byte offset off, by flipping every byte (XOR 0xFF). It is the direct sector-corruption injector for bytes that are already on stable storage — the SimFileHandle.Write fault path only corrupts sectors as they are written, so it cannot damage a frame that was durably committed in an earlier session. Flipping a byte inside a committed WAL frame's header or payload makes that frame fail its CRC32C check on the next replay, modelling a bad disk sector under a durable frame.
It draws NOTHING from the Seed and holds only SimDisk's own mutex, so it never perturbs the reproducible fault stream. It returns an error wrapping fs.ErrNotExist when the file is absent and a range error when [off, off+n) does not lie wholly within the file. It must be called from the controlling goroutine while no handle is mid-write.
func (*SimDisk) Crash ¶ added in v0.6.0
func (d *SimDisk) Crash()
Crash is an alias for SimDisk.CrashHost, the STRONGER of the two crash models. Every scenario that has not consciously chosen otherwise therefore gets power-failure semantics, in which unsynced data is lost; a scenario that genuinely means SIGKILL calls SimDisk.CrashProcess and says so.
func (*SimDisk) CrashHost ¶ added in v0.12.0
func (d *SimDisk) CrashHost()
CrashHost models a host crash — power failure, hard reset, hypervisor kill. Nothing that had not reached stable storage survives. It is the model SimDisk.Crash applies, and it has two halves that must agree with each other.
Data ¶
Every file reverts to its durable image: the bytes covered by a SimFileHandle.Sync that returned nil, and nothing else. A successful write(2) carries no durability whatever — Linux write(2) NOTES: "A successful return from write() does not make any guarantee that data has been committed to disk… The only way to be sure is to call fsync(2)" — and POSIX reaches durability solely through Successfully Transferred, i.e. via fsync/fdatasync/O_SYNC. Until rmp #2535 the model kept every byte ever written, which granted data the guarantee RocksDB's file abstraction reserves for Flush() ("should survive a process crash") while revoking names as if power had been lost — an internal contradiction no real event has, and one that made every "the commit was acked, so the bytes are recovered" assertion pass irrespective of any fsync.
SimDisk.LastCrashDiscardedBytes reports how much this crash actually threw away, which is the non-vacuity observable an oracle needs to prove it was tested at all.
Names ¶
It drops every dirent that is not yet durable, exactly as a real crash within the kernel writeback window loses a create or rename whose parent directory was never fsync'd. A name becomes durable only after a SimDisk.DirSync of its parent; until then it is the load-bearing job of the publish protocol's directory fsyncs to make the name survive, and removing one of those fsyncs makes the corresponding name vanish here — which is what the non-vacuity guard test asserts.
Renames are rolled back, not revoked ¶
A rename is not a create, and revoking its new dirent is not enough: the old name was unlinked when the rename was applied, so revoking alone loses BOTH names. rename(2) is atomic, and a crash after it leaves either the new name or the old one — never neither. Crash therefore ROLLS BACK the renames it does not keep, restoring each source name with the dirent durability it carried at the instant of the rename (see [renameUndo]). A source whose own name was never durable is restored non-durable and then dropped by the ordinary revoke pass below, which is the one case in which losing both names IS legal: that name had never been crash-survivable.
Removals are restored, not applied unconditionally ¶
An unlink is a mutation of the containing directory, in the same durability class as the create and the rename above, so a crash before that directory is fsync'd may legally leave the removed name IN PLACE. Crash therefore restores the removals it does not keep, each entry with the dirent durability it carried at the instant of the unlink (see [unlinkUndo]). Both outcomes are legal here — unlike a rename there is no impossible third one — and the branch that matters is the restore: it is the only way the harness can reach the engine's tolerance of a leftover "<dir>/snapshot.tmp" or "<dir>/snapshot.bak" FROM THE REMOVAL SIDE, rather than by crashing before the removal was ever issued (rmp #2536).
One ordered log, one draw ¶
How much of the pending metadata is kept is drawn ONCE from the disk's dirent sub-stream over the renames and the removals TOGETHER, so it is a reproducible function of the run seed (see the direntSeed field docs) and the mutations that survive are always a PREFIX of the issued ones, as a journalling filesystem guarantees. Drawing per kind would let a crash keep an unlink while rolling back a rename issued before it, which is not an interleaving a filesystem can produce. An armed SimDisk.ArmRenameRollbackForPath or SimDisk.ArmRemoveRollbackForPath pins the boundary at or before that record instead of sampling it, and an armed SimDisk.ArmRenameRevokeBothForPath deliberately produces the impossible both-names-lost outcome for tests of the harness itself. Mutations already made durable — by a SimDisk.DirSync of the affected directory, by SimDisk.ArmRenameWritebackForPath or SimDisk.ArmRemoveWritebackForPath, or by being root-level — are never reversed.
CrashHost mutates the SimDisk in place. It is driven from the single simulation goroutine, but it may run while another goroutine is parked inside a gated SimFileHandle.Sync: that is the phantom-commit window, and the generation stamp on the in-flight fsync (see [SimDisk.completeSync]) is what stops the parked call from re-hardening bytes this crash has just discarded.
func (*SimDisk) CrashProcess ¶ added in v0.12.0
func (d *SimDisk) CrashProcess()
CrashProcess models a SIGKILL of the process (kill -9). The process dies; the kernel does not. Everything the process had already handed the kernel — every byte accepted by a write(2) and every directory entry created by a create(2)/rename(2)/unlink(2) — is still there for the next process to read, whether or not it was ever fsync'd. CrashProcess therefore discards NOTHING: no data revert, no dirent revocation, no rename rollback.
It is the model SimStore.Crash's own documentation used to describe while SimDisk.Crash actually applied dirent revocation, which is a host-crash effect. The two are now separate primitives so a scenario can state which event it means (rmp #2535).
Pending renames stay pending: their dirents are in the page cache but not on stable storage, so a LATER SimDisk.CrashHost can still lose them. It must be driven from the single simulation goroutine.
func (*SimDisk) DirSync ¶ added in v0.6.0
DirSync makes every dirent in directory dir durable: it is the in-memory analogue of fsync(2) on a directory descriptor. The snapshot/csrfile publish protocol calls it on the staging directory before the publish rename and on the parent directory after it; only after DirSync does a freshly created or renamed name survive a SimDisk.Crash. A DirSync of a directory with no entries is a harmless no-op (it models fsync of an empty directory).
When a one-shot fault is armed for dir (SimDisk.ArmDirSyncFaultForPath) it returns ErrSimFault and durabilises NOTHING.
func (*SimDisk) DirSyncFaultCount ¶ added in v0.12.0
DirSyncFaultCount returns how many armed directory-fsync faults actually fired since the disk was created, counting both entry points (SimDisk.DirSync and SimDisk.ParentDirSync) because they are one primitive. It is the reachability observable a non-vacuity gate reads to prove an armed fault really bit rather than being silently ignored because its path never matched.
func (*SimDisk) DurableImage ¶ added in v0.12.0
DurableImage returns a copy of the bytes the file at path would hold after a SimDisk.CrashHost — the durable image, as opposed to SimDisk.ReadFile's live image, which includes writes the process has issued but never fsync'd. It returns an error wrapping fs.ErrNotExist when the file is absent.
func (*SimDisk) DurableSize ¶ added in v0.12.0
DurableSize reports how many leading bytes of the file at path are on stable storage — the length SimDisk.CrashHost would leave it at — and whether the file exists at all. It is the direct read of the durable watermark that lets a test distinguish "the engine fsync'd" from "the bytes merely exist".
func (*SimDisk) FaultRate ¶ added in v0.8.0
FaultRate returns the per-sector / per-Sync fault probability the disk was constructed with (see NewSimDisk). It is an observability accessor for tests that assert the DiskConfig.FaultRate wiring in New; faultRate is immutable after construction, so it reads it without the mutex and mutates nothing.
func (*SimDisk) FaultedSectorCount ¶ added in v0.12.0
FaultedSectorCount returns how many sectors of the file at path currently carry a fault mark. It exists so a test can observe the marks being dropped by a truncation (F13, rmp #2542) without reaching into simFile.
func (*SimDisk) LastCrashDiscardedBytes ¶ added in v0.12.0
LastCrashDiscardedBytes reports how many written-but-never-synced bytes the last SimDisk.CrashHost discarded, counted across the files that SURVIVED the crash (bytes lost with a revoked dirent are a name loss, not a data loss, and are not counted here). SimDisk.CrashProcess always leaves it at zero, because a process crash discards nothing.
It is the non-vacuity observable for every durability oracle: a scenario that asserts "the unsynced tail is gone" must also be able to prove there WAS an unsynced tail, or it is asserting nothing.
func (*SimDisk) LastCrashKind ¶ added in v0.12.0
LastCrashKind reports which crash primitive this disk last ran, or CrashKindNone if it has not crashed. It is the observable a scenario asserts to prove it exercised the model it intended.
func (*SimDisk) LastCrashRemoveOutcome ¶ added in v0.12.0
LastCrashRemoveOutcome reports what the most recent SimDisk.Crash adjudicated about REMOVALS: how many not-yet-durable ones it found pending, and how many of those it restored. The remainder (pending-restored) stuck. Both are zero before the first crash.
It is the shape observable a non-vacuity gate reads: "the crash landed inside the unlink window" is pending > 0, and "the removal did not stick" is restored > 0. A verdict gate must never be conditioned on it — an unmet precondition is a reason to report, not to pass.
func (*SimDisk) LastCrashRenameOutcome ¶ added in v0.12.0
LastCrashRenameOutcome reports what the most recent SimDisk.Crash adjudicated: how many not-yet-durable renames it found pending, and how many of those it rolled back to the old name. The remainder (pending-rolledBack) were kept at the new name. Both are zero before the first crash.
It is the shape observable a non-vacuity gate reads: "the crash landed between the two renames" is pending == 2, and "it took the stranded-backup branch" is rolledBack == 1. A verdict gate must never be conditioned on it — an unmet precondition is a reason to report, not to pass.
func (*SimDisk) MarkDataDurable ¶ added in v0.12.0
MarkDataDurable declares the CURRENT contents of the file at path to be on stable storage, advancing its durable watermark to the live length without issuing a Sync. It returns an error wrapping fs.ErrNotExist when the file is absent.
It is the counterpart of SimDisk.CorruptRange: a harness primitive that edits the durable image directly, for a scenario that SUBSTITUTES a crafted component for a published one and needs the substitution to be what a crash leaves behind. Going through SimFileHandle.Sync instead would draw from the Seed — perturbing the reproducible fault stream — and could itself be failed by the injector, turning a fixture step into a spurious error.
It is NOT a way for engine code to obtain durability it did not fsync for. Nothing outside a test fixture should call it; the durability an engine has is the durability its Syncs earned.
It draws NOTHING from the Seed and holds only SimDisk's own mutex, so it never perturbs the reproducible fault stream.
func (*SimDisk) MkdirAll ¶
MkdirAll creates dir and every missing parent, registering each newly created component in the dirent model. perm is ignored.
What it used to be, and what that cost ¶
It returned nil and registered nothing. Two consequences followed (audit finding F9, rmp #2543). A newly created directory was IMPLICITLY DURABLE, needing no parent fsync, unlike every other name in the model — so a crash could never take one away. And an EMPTY directory was UNREPRESENTABLE, because SimDisk.Stat inferred a directory purely from a key prefix and SimDisk.Exists never saw directories at all — so the state "the snapshot directory exists but is empty or incomplete" could not be reached.
Neither was a live defect: today's recovery probe keys on the manifest FILE rather than on the directory (store/recovery), so it is robust to the gap. The exposure was prospective — any future check of the form "does the snapshot directory exist" would have been silently inert under simulation, and partial-publish states stayed out of reach.
Why only NEWLY created components are registered ¶
A component that already exists — tracked in d.dirs, or implied by a file beneath it — is left exactly as it was. MkdirAll on an existing directory creates nothing and needs no fsync, and registering it afresh as not-yet- durable would let a crash delete a subtree that was already durable. That is not a stricter model; it is a wrong one.
func (*SimDisk) OpenFile ¶
func (d *SimDisk) OpenFile(path string, flag int) (*SimFileHandle, error)
OpenFile opens (creating when os.O_CREATE is set) the file at path and returns a handle positioned per the flags: at end when os.O_APPEND is set, at zero otherwise. When os.O_TRUNC is set the file's contents are discarded. It returns an error wrapping fs.ErrNotExist when the file is absent and os.O_CREATE is not set, and one wrapping fs.ErrExist when os.O_CREATE and os.O_EXCL are both set and the file exists.
func (*SimDisk) ParentDirSync ¶ added in v0.6.0
ParentDirSync makes the dirent of childPath durable by fsyncing its parent directory. It is the analogue of the post-rename parent-directory fsync the publish protocols issue.
It faults when a one-shot arm is set for this exact childPath (SimDisk.ArmParentDirSyncFaultForPath) OR for the parent directory it is about to fsync (SimDisk.ArmDirSyncFaultForPath) — the same body decides both, since this call IS an fsync of that directory.
func (*SimDisk) PendingRemoveCount ¶ added in v0.12.0
PendingRemoveCount returns how many removals are currently recorded as not-yet-durable, i.e. how many a SimDisk.Crash issued now could put back. A scenario reads it to prove it really is INSIDE an unlink window before crashing, instead of assuming the timing worked out.
func (*SimDisk) PendingRenameCount ¶ added in v0.12.0
PendingRenameCount returns how many renames are currently recorded as not-yet-durable, i.e. how many a SimDisk.Crash issued now would have to adjudicate. A scenario reads it to prove it really is INSIDE a rename window before crashing, instead of assuming the timing worked out.
Renames and removals share ONE ordered log (see the direntUndos field docs on SimDisk), but they are counted separately here because a scenario asserts about the operation it issued.
func (*SimDisk) ReadDir ¶ added in v0.16.0
ReadDir lists the immediate children of the directory at dir, sorted by name, as os.ReadDir does. A child is a file directly under dir, or a directory — tracked explicitly through MkdirAll or implied by a file beneath it. A missing dir is reported with an error wrapping fs.ErrNotExist, and a path naming a file with one wrapping syscall.ENOTDIR.
It exists for the store/bulkimport filesystem seam, whose empty-directory check lists the target directory (rmp #2518). It reads the CURRENT image, including names a crash would still revoke, exactly as a directory listing on a live filesystem does.
func (*SimDisk) ReadFile ¶ added in v0.6.0
ReadFile returns a copy of the whole contents of the file at path, or an error wrapping fs.ErrNotExist when absent. The copy keeps the returned slice independent of later writes, mirroring os.ReadFile.
func (*SimDisk) Remove ¶
Remove deletes the file at path. Removing an absent path is a no-op, matching the tolerant cleanup the snapshot writer relies on.
Crash outcome ¶
The unlink is a mutation of the containing DIRECTORY, so until that directory is fsync'd a crash may legally leave the name in place. It is recorded in the shared dirent undo log (see [unlinkUndo] and the direntUndos field docs on SimDisk) so a SimDisk.Crash can restore it; which of the two legal outcomes a given crash selects is drawn from the seed unless an arm pins it (SimDisk.ArmRemoveWritebackForPath, SimDisk.ArmRemoveRollbackForPath).
func (*SimDisk) RemoveAll ¶ added in v0.6.0
RemoveAll deletes path and every file under path/. Removing an absent path is a no-op, matching os.RemoveAll and the staging/backup cleanup the snapshot writer relies on.
Crash outcome ¶
Same as SimDisk.Remove, extended to the subtree: the whole removal is one record in the shared dirent undo log, so a crash either keeps it or restores the entire subtree with each name's original dirent durability. It is one record and not one per name because the caller issued one removal, and a crash that restored half a subtree would model an interleaving os.RemoveAll's callers never see.
func (*SimDisk) RemoveHitCount ¶ added in v0.12.0
RemoveHitCount returns how many removals actually unlinked at least one existing name since construction. A removal of an absent path is a no-op and is not counted.
It is the observable that separates a tolerant cleanup which really cleaned something from one that ran as a no-op — the difference between demonstrating that an engine branch handled a leftover artefact and merely asserting it.
func (*SimDisk) RemoveHitCountForPath ¶ added in v0.12.0
RemoveHitCountForPath returns how many removals of exactly path actually unlinked something (see SimDisk.RemoveHitCount). A scenario brackets it around the operation under test and reads the DELTA, so a cleanup that runs several times over the same name stays attributable.
func (*SimDisk) RemoveRollbackCount ¶ added in v0.12.0
RemoveRollbackCount returns how many armed unlink rollbacks actually fired since construction (see SimDisk.ArmRemoveRollbackForPath). A scenario reads it to prove the branch it pinned was really reached rather than silently ignored because the path never matched.
func (*SimDisk) RemoveWritebackCount ¶ added in v0.12.0
RemoveWritebackCount returns how many armed unlink write-backs actually fired since construction (see SimDisk.ArmRemoveWritebackForPath). A scenario reads it to prove the branch it pinned was really reached rather than silently ignored because the path never matched.
func (*SimDisk) Rename ¶
Rename atomically moves oldPath to newPath, replacing any existing destination. It handles both a single file (oldPath is a file key) and a directory (oldPath has child keys prefixed oldPath+"/"): the snapshot publish protocol renames a whole staging directory onto the live name, so a directory rename must move every child key, re-rooting its prefix. The moved dirent(s) become NOT durable — only a SimDisk.DirSync of the new parent makes the new name crash-survivable — which is what lets the simulator crash in the publish window between the rename and the parent-dir fsync. It returns an error wrapping fs.ErrNotExist when the source is absent.
Crash outcome ¶
A rename whose new name is not yet durable is recorded in an undo log (see [renameUndo]) so that a SimDisk.Crash lands on one of the TWO outcomes a real filesystem can produce — the new name, or the old name — rather than on the impossible third one of losing both. Which of the two a given crash selects is chosen deterministically from the seed unless an arm pins it (SimDisk.ArmRenameWritebackForPath, SimDisk.ArmRenameRollbackForPath, SimDisk.ArmRenameRevokeBothForPath).
func (*SimDisk) RenameFaultCount ¶ added in v0.12.0
RenameFaultCount returns how many armed rename faults actually fired since construction (see SimDisk.ArmRenameFaultForPath). A scenario reads it to prove the fault it armed was really reached rather than silently ignored because the destination never matched.
func (*SimDisk) RenameRevokeBothCount ¶ added in v0.12.0
RenameRevokeBothCount returns how many armed impossible-outcome revocations actually fired since construction (see SimDisk.ArmRenameRevokeBothForPath).
func (*SimDisk) RenameRollbackCount ¶ added in v0.12.0
RenameRollbackCount returns how many armed rename rollbacks actually fired since construction (see SimDisk.ArmRenameRollbackForPath). A scenario reads it to prove the branch it pinned was really reached rather than silently ignored because the destination never matched.
func (*SimDisk) RenameWritebackCount ¶ added in v0.12.0
RenameWritebackCount returns how many armed rename write-backs actually fired since construction (see SimDisk.ArmRenameWritebackForPath). A scenario reads it to prove the crash window it selected was really entered.
func (*SimDisk) SetCapacity ¶ added in v0.6.0
SetCapacity bounds the disk to capacityBytes total bytes across all files, modelling a finite disk. When enospcOnSync is false the out-of-space condition surfaces eagerly at the growing Write/Truncate; when true it surfaces at Sync (delayed allocation). A capacityBytes of 0 removes the bound (the default). It must be called before the store is driven, from the single simulation goroutine; it draws nothing from the seed, so it never perturbs the reproducible fault stream. See the field docs on SimDisk for the model.
func (*SimDisk) Snapshot ¶
Snapshot returns an independent deep copy of every file's contents keyed by path. Mutating the returned maps or slices never affects the live filesystem, so a caller can capture disk state for comparison after a crash.
func (*SimDisk) Stat ¶ added in v0.6.0
Stat reports a minimal fs.FileInfo for a file at path. It returns an error wrapping fs.ErrNotExist when no file is present, which is exactly the probe the snapshot and recovery paths rely on (testing for manifest.json / wal presence). Stat reports a directory when any child key is prefixed path+"/", and — since rmp #2543 — also when the directory was created explicitly through MkdirAll, which is what makes an EMPTY directory representable.
func (*SimDisk) SyncCount ¶ added in v0.8.0
SyncCount returns the number of SimFileHandle.Sync calls performed across every handle of this disk since the last SimDisk.ArmSyncFaultAt reset (or since construction). A concurrent scenario reads it to gate a teardown on durable progress (a bounded condition wait) rather than on wall-clock time.
func (*SimDisk) TruncatePath ¶ added in v0.6.0
TruncatePath resizes the file at path to size bytes (zero-filling on grow), the path-based analogue of os.Truncate used by the csrfile writer. It returns an error wrapping fs.ErrNotExist when the file is absent.
type SimFileHandle ¶
type SimFileHandle struct {
// contains filtered or unexported fields
}
SimFileHandle is an open handle onto a SimDisk file. It implements the WAL file interface (io.Reader, io.Writer, io.Seeker, Sync, Truncate, Close) so it can substitute for *os.File.
Concurrency contract ¶
SimFileHandle is NOT safe for concurrent use; it is driven from the single simulation goroutine.
func (*SimFileHandle) Close ¶
func (h *SimFileHandle) Close() error
Close releases the handle. It is idempotent: a second Close is a no-op.
func (*SimFileHandle) Read ¶
func (h *SimFileHandle) Read(p []byte) (int, error)
Read copies up to len(p) bytes from the current position into p, advancing the position. It returns io.EOF when the position is at or past end of file.
func (*SimFileHandle) Seek ¶
func (h *SimFileHandle) Seek(offset int64, whence int) (int64, error)
Seek repositions the handle per the standard io.Seeker whence values and returns the resulting absolute offset. A negative resulting offset is an error.
func (*SimFileHandle) Stat ¶ added in v0.6.0
func (h *SimFileHandle) Stat() (fs.FileInfo, error)
Stat returns a minimal fs.FileInfo for the open handle, reporting the current file size. The snapshot index writer calls it to record the component's on-disk size in the manifest.
func (*SimFileHandle) Sync ¶
func (h *SimFileHandle) Sync() error
Sync fsyncs the file, granting durability to everything written before the call returns.
A FAILED fsync is permanent, and a retry is a lie (rmp #2540) ¶
When Sync fails, this file's durable prefix is frozen for good: every later Sync on it RETURNS SUCCESS AND ADVANCES NOTHING. That is not a simplification; it is the behaviour the sources describe.
On Linux, a write-back error marks the affected pages CLEAN and reports the error to exactly one fsync caller, after which it is gone — so the dirty data is discarded and a retried fsync returns success over bytes that no longer exist anywhere. The canonical write-up is Rebello et al., "Can Applications Recover from fsync Failures?" (USENIX ATC 2020), which measured this across ext4, XFS and Btrfs and found the page state after a failure differs per filesystem but is never "still dirty and retryable" on ext4 data=ordered. PostgreSQL reached the same conclusion the expensive way and now PANICs rather than retrying (its fsync failure handling, commit 9ccdd7f6 and the "fsyncgate" thread of 2018).
The audit that filed this was explicit that NO man page warns against retrying fsync, so the behaviour is sourced to the measurements and to PostgreSQL's response, never to a man page.
Why the model freezes the whole prefix ¶
The durable image is a contiguous PREFIX. The bytes the failed write-back was carrying are gone, so nothing after them can be made durable either — there is no way to express "durable, then a hole, then durable" in a watermark, and a real reader stopping at the hole would see exactly the same thing.
Bound: a caller that SEEKS BACK and rewrites the lost region could legitimately make it durable again, and that is not modelled. No caller in this module does — the WAL only appends — and modelling it would require tracking which bytes were rewritten after the failure.
There is no live defect this protects against; it protects against a future one ¶
No current caller retries: the WAL poisons fail-stop, and RocksDB's equivalent is structural, with seen_error short-circuiting every later call on the writer. What the model could not express before was the consequence of INTRODUCING a retry. An optimisation of the form "retry the fsync once before poisoning" would have looked perfectly safe under DST and been a data-loss bug in production. [TestSyncRetryAfterFailureLosesData] writes exactly that optimisation and observes the loss.
func (*SimFileHandle) Truncate ¶
func (h *SimFileHandle) Truncate(size int64) error
Truncate resizes the file to size bytes, zero-filling when growing and dropping fault marks for sectors that no longer exist.
func (*SimFileHandle) Write ¶
func (h *SimFileHandle) Write(p []byte) (int, error)
Write copies p to the file at the current position, growing the file as needed, and advances the position. Any byte written into a sector that the fault injector has marked faulted is corrupted deterministically (a single byte in that sector is flipped), modelling a torn or mis-directed write.
type SimListener ¶
type SimListener struct {
// contains filtered or unexported fields
}
SimListener is an in-memory net.Listener that feeds SimConn server-ends to a real bolt/server running under github.com/FlavioCFOliveira/GoGraph/bolt/server.Server.Serve. Each SimListener.Dial creates a connected SimConn pair, queues the server-end for the server's Accept loop, and returns the client-end to the caller (the wire client harness). No OS socket is involved, so the server runs its genuine handshake and message loop over purely in-memory bytes.
Concurrency contract ¶
SimListener is safe for concurrent use: Dial may be called from any number of goroutines (one per simulated connection) while the server calls Accept from its single accept goroutine. This is what lets the concurrent harness open N connections against one server.
func NewSimListener ¶
func NewSimListener(clk clock.Clock) *SimListener
NewSimListener returns an in-memory listener whose connections route deadlines through clk. Hand it to Server.Serve; drive new connections with Dial. clk must be non-nil (clock.Real for ordinary timing, a clock.Fake for deterministic virtual deadlines).
func (*SimListener) Accept ¶
func (l *SimListener) Accept() (net.Conn, error)
Accept implements net.Listener.Accept. It blocks until a connection is dialed or the listener is closed, returning the server-end of the next SimConn pair. After Close it returns ErrSimListenerClosed.
func (*SimListener) Close ¶
func (l *SimListener) Close() error
Close implements net.Listener.Close. It stops further Accept and Dial calls. In-flight connections already accepted by the server are unaffected (they live until the server or harness closes them). It is idempotent.
func (*SimListener) Closed ¶ added in v0.12.0
func (l *SimListener) Closed() bool
Closed reports whether the listener has been closed.
It is the drain-ordering observable rmp #2483 needs. github.com/FlavioCFOliveira/GoGraph/bolt/server.Server.Shutdown closes the listener BEFORE it starts waiting for its connection drain, so a closed listener is positive evidence that a Shutdown call has progressed into that wait rather than not yet started — which is what makes "the store's teardown had not begun" a statement about the server instead of about a Shutdown that had not begun either.
A probe Dial would answer the same question (Dial refuses once closed) but creates a connection when it arrives too early, and a connection accepted into a drain window is counted live by a connection-accounting oracle. This accessor has no such side effect, which is why it exists.
func (*SimListener) Dial ¶
func (l *SimListener) Dial() (*SimConn, error)
Dial creates a new connected SimConn pair, queues the server-end for Accept, and returns the client-end. It blocks if the accept backlog is full (bounded by [defaultAcceptBacklog]) until the server accepts a queued connection or the listener is closed. After Close it returns ErrSimListenerClosed.
type SimReport ¶
type SimReport struct {
// Shrunk, when non-nil, carries the minimal failing reproducer the shrinker
// produced for this failure ([ShrinkTrace]). It is attached by the CLI replay
// path after a deterministic failure is shrunk; a report from a live run
// leaves it nil.
Shrunk *ShrinkResult
// Scenario is the catalogue key of the scenario that produced the failure,
// and Mode the harness it ran under. Both are rendered, because a report is
// read by an operator who has only the log: without them a sighting says
// what broke but not what was running, and a CONCURRENT mode is precisely
// the case where "not bit-reproducible" changes how it must be chased
// (rmp #2347).
Scenario string
FailedOp Op
Violations []Violation
OracleState OracleSnapshot
Seed uint64
FailedTick int64
Mode ExecMode
// Repro is an explicit reproduction instruction for a run whose state is
// NOT determined by the seed alone — typically a scenario driven from a test
// with a configuration that has no command-line form. When it is set,
// [SimReport.String] prints it verbatim instead of synthesising a command.
//
// It exists because the synthesised line was actively misleading: see
// [SimReport.reproLine].
Repro string
}
SimReport is the result of a failed simulation: the seed that produced it, the tick and operation at which the first violation was detected, every violation found at that tick, and a snapshot of the oracle state. A nil *SimReport returned from Simulator.Run means the simulation passed.
The DST harness types share a SimXxx naming scheme by design (see SimDisk in disk.go).
func (*SimReport) String ¶
String renders a human-readable failure report. It always includes a "Reproduce with:" line, which either reproduces the failure or says why it cannot — never a command that runs a different workload (see [SimReport.reproLine]).
IT CAN NEVER RENDER EMPTY, and a report that carries no violation says so LOUDLY rather than rendering a bare header (rmp #2347). A non-nil report means the scenario failed; one that names no violated invariant is itself a reporting defect, and an operator who cannot tell that from a clean run has been told nothing. [TestSimReportNeverRendersEmptyShort] pins both halves.
type SimServer ¶
type SimServer struct {
// contains filtered or unexported fields
}
SimServer runs a real github.com/FlavioCFOliveira/GoGraph/bolt/server.Server over an in-memory SimListener. It exists so the Phase-3 actors drive the GENUINE Bolt wire path — handshake, framing, message loop, streaming — with no OS socket and no reimplementation of the server. New client connections are obtained with SimServer.Dial.
NewSimServer starts the server with server.NoAuthHandler (development/testing mode), which is what the robustness and ACID scenarios want: they abuse the wire and the transaction machinery, not the credential check. The credential surface itself is driven by NewSimServerAuth, which takes an arbitrary server.AuthHandler so the auth scenarios can present a server that genuinely REFUSES a wrong password (rmp #2481) — against a NoAuthHandler a bad-credential probe would pass by admitting everything, which proves nothing. A finite result-row cap is configured on the engine so a single overload query cannot materialise an unbounded result set.
Concurrency contract ¶
SimServer is safe for concurrent use: SimServer.Dial may be called from many goroutines (the concurrent harness opens one connection per goroutine) while the embedded server's accept loop runs. SimServer.Close is idempotent and drains the server before returning.
func NewSimServer ¶
NewSimServer builds a SimServer over the given engine and starts it serving on an in-memory listener whose connection deadlines route through clk. The engine must be non-nil; callers typically pass an engine with a finite result-row cap (see SimEngineForServer). The returned server is already accepting; obtain connections with SimServer.Dial and tear it down with SimServer.Close.
Authentication is server.NoAuthHandler; use NewSimServerAuth to drive a server that validates credentials.
func NewSimServerAuth ¶ added in v0.12.0
func NewSimServerAuth(eng *cypher.Engine, clk clock.Clock, auth server.AuthHandler) (*SimServer, error)
NewSimServerAuth builds a SimServer whose sessions authenticate through auth instead of server.NoAuthHandler, so a scenario can drive the credential surface over the genuine wire: a wrong password, an unknown scheme, LOGOFF followed by a write, and re-authentication.
A nil auth is REFUSED rather than defaulted. The internal constructor treats a nil handler as NoAuthHandler (which is what every pre-existing scenario wants), but silently doing that here would hand a credential-validating scenario a server that admits everything — and every refusal it then failed to observe would look like a passing test. A caller that wants NoAuth asks for it by name with NewSimServer.
The server's own log is discarded, because every rejected credential is reported at ERROR level by design and an auth scenario provokes dozens of them on purpose (see [quietSimLogger]).
func NewSimServerInFlight ¶ added in v0.12.0
NewSimServerInFlight builds a SimServer whose per-connection in-flight cursor cap is maxInFlight instead of server.DefaultMaxInFlightPerConnection, so a scenario can drive the cap to its refusal over the genuine wire rather than by reaching into a server.Session. It is the constructor rmp #2484 needs; before it, nothing in the harness passed Options.MaxInFlightPerConnection at all, so the only cap the DST could ever have reached was 1024 cursors deep inside one transaction.
A non-positive maxInFlight is REFUSED rather than defaulted. The server option treats zero as "take the default", so silently passing it through would hand a cap-driving scenario a cap of 1024 — and the refusal it then failed to observe would read as a passing test rather than as an unreached bound.
Authentication is server.NoAuthHandler and the server log is discarded ([quietSimLogger]), because a store-backed engine carries no result-row cap and server.NewServer warns about that on every construction.
func NewSimServerInboundBudget ¶ added in v0.12.0
func NewSimServerInboundBudget(eng *cypher.Engine, clk clock.Clock, maxInboundDecodeBytes int64) (*SimServer, error)
NewSimServerInboundBudget builds a SimServer whose ENGINE-WIDE inbound-decode ceiling is maxInboundDecodeBytes: one packstream.InboundBudget pool shared by every connection the server accepts.
That sharing is the whole point. The per-message decoded-collection cap (packstream's maxDecodedCollectionBytes, 128 MiB) bounds a SINGLE message; the pool this sets bounds the SUM in flight across the fleet, which is the CWE-770 vector the per-message cap cannot see. The server creates the pool once, in server.NewServer (bolt/server/serve.go:654), and hands the same pointer to every connection's reassembly reader and pooled decoder.
A non-positive value is REFUSED rather than defaulted. Zero means "derive a default" to the server option and -1 means "unlimited" (server.MaxInboundDecodeBytesUnlimited), so silently passing either through would hand a pressure-driving scenario a ceiling of 1 GiB or none at all — and the rejection it then failed to observe would read as a passing test rather than as an unreached bound. This mirrors NewSimServerInFlight's refusal for the same reason.
Authentication is server.NoAuthHandler and the server log is discarded ([quietSimLogger]): a scenario that provokes the ceiling makes the server log "inbound decode memory budget exceeded" at WARN by design (serve.go:1264), so the noise is expected output rather than a signal.
func NewSimServerOwnedCloser ¶ added in v0.12.0
func NewSimServerOwnedCloser( eng *cypher.Engine, clk clock.Clock, closer io.Closer, wrapConn func(net.Conn) net.Conn, ) (*SimServer, error)
NewSimServerOwnedCloser builds a SimServer that OWNS a store-level teardown closer: closer is installed as server.Options.Closer, so the embedded server closes it itself once it has drained every connection — the documented "drain the connections, then close the DB" ordering that store/db.go says a Bolt server provides (store/db.go:54-57). It is the constructor rmp #2483 needs, and nothing in the module passed Options.Closer outside bolt/server's own tests before it.
wrapConn, when non-nil, decorates every connection the server's accept loop receives. It is the only observable in the harness that can time the closer against the connection drain: the per-connection handler's FIRST deferred call is conn.Close (bolt/server/serve.go:1063 and the outer defer at :904), which runs strictly BEFORE the accept-loop wrapper's s.wg.Done (:798), so a decorator counting Close calls sees a connection leave before the WaitGroup the drain waits on can drop to zero. Counting accepts and closes therefore yields a one-sided oracle: a closer entered while a decorated connection is still open is a genuine drain-ordering breach, and the nanosecond window between a connection's Close and its wg.Done can only read as drained, never as a false breach.
Authentication is server.NoAuthHandler and the server log is discarded ([quietSimLogger]), because a store-backed engine carries no result-row cap and server.NewServer warns about that on every construction.
func NewSimServerTxRegistry ¶ added in v0.12.0
func NewSimServerTxRegistry( eng *cypher.Engine, listenerClk, serverClk clock.Clock, maxTxIdle, defaultTxTimeout time.Duration, maxOpenTxPerPrincipal int, ) (*SimServer, error)
NewSimServerTxRegistry builds a SimServer wired for the transaction-registry and idle-reaper surface of rmp #2482: the server's own clock is serverClk, so the message loop's timeout timer and the registry's StartedAt/Elapsed run on VIRTUAL time, while the in-memory listener keeps listenerClk. maxTxIdle, defaultTxTimeout and maxOpenTxPerPrincipal are passed straight through to server.Options; a zero value in any of them takes that option's own documented default, so this constructor can also stand in for NewSimServer with only a clock changed. Authentication is server.NoAuthHandler and the server log is discarded ([quietSimLogger]), because the reaper reports every reap at WARN and a reaper arm provokes them deliberately.
Why the two clocks MUST be different objects ¶
The listener clock is not a spare copy of the server clock: it is the clock every SimConn's blocked I/O uses, and pointing it at the same clock.Fake breaks the instrument the reaper arm depends on. Both halves of this are verified in the code, not assumed:
- EVERY server-side read has a deadline. The reader goroutine calls conn.SetReadDeadline(time.Now().Add(ConnTimeout)) before every single read (bolt/server/serve.go:1109), and a SimServer sets ConnTimeout to 30 s, so the branch is always taken.
- A deadline-bearing blocked read arms a timer ON THE LISTENER CLOCK. halfPipe.waitDeadline does timer := h.clk.NewTimer(h.clk.Until(d)) and spawns a goroutine to broadcast on it (internal/sim/simconn.go:143), where h.clk is the clock handed to NewSimListener and shared by both ends of every SimConn. On a shared fake, every connection parked waiting for its next request therefore registers a timer, and a NewTimer-counting decorator like txClockProbe can no longer attribute its count to the transaction reaper — which is the whole point of counting.
- The deadline instant is WALL-CLOCK. It comes from time.Now(), not from the injected clock, so on a shared fake the comparison h.clk.Now().Before(d) is between virtual and real time. A fake started at time.Now() would then time the connection out as soon as the arm advanced past ConnTimeout — reaping the socket instead of the transaction; a fake started at the Unix epoch would instead put the deadline ~56 years of virtual time away, so the arm's whole advance budget is silently inert against it. Neither is production behaviour.
- Nothing on the server side needs a fake listener. All three of the server's socket deadlines are real-time by construction — the handshake conn.SetDeadline (bolt/server/serve.go:965), the per-read deadline (:1109) and the per-write deadline (:1394) all read time.Now() — so leaving listenerClk on clock.Real leaves connection liveness behaving exactly as it does in production while the transaction machinery runs on virtual time.
The clock the returned SimServer hands to each WireClient from SimServer.Dial is listenerClk, matching the connection it is built on.
func (*SimServer) Close ¶
Close stops accepting new connections, cancels the serve context, and waits for the server to drain. It is idempotent and returns the server's exit error (nil on a clean shutdown).
func (*SimServer) Dial ¶
func (s *SimServer) Dial() (*WireClient, error)
Dial opens a new client connection to the server over the in-memory listener, returning a WireClient ready to negotiate. The caller must Close the client when done. It returns an error only if the listener is closed.
func (*SimServer) DialConn ¶
DialConn opens a new client connection and returns the raw SimConn, for callers (notably the BoltAbuser) that need to write malformed bytes the WireClient would never produce.
func (*SimServer) ListenerAddr ¶ added in v0.12.0
ListenerAddr returns the address of the in-memory listener the embedded server serves on, as net.Listener.Addr renders it.
It exists as the INDEPENDENT REFERENCE for the ROUTE payload arm of rmp #2485. The routing table a client receives is built from the address the accept loop copied off the listener — localAddr = s.ln.Addr().String() at bolt/server/serve.go:1000-1005, handed to newSession and read back by handleRoute as RoutingTable(s.localAddr) (bolt/server/session.go:1751). Reading the listener HERE therefore reaches that same source of truth by a different route than the reply does, so "the routing table names THIS server" is a comparison between two independently obtained values rather than a constant restated. A checker that instead compared the reply against github.com/FlavioCFOliveira/GoGraph/bolt/server.RoutingTable's own output would be comparing that function with itself.
func (*SimServer) Server ¶ added in v0.12.0
Server exposes the embedded server.Server so a scenario can drive its operator API — server.Server.Transactions and server.Server.TerminateTransaction, both of which take the registry's own lock and are safe on a serving server. It is the accessor rmp #2482 needs.
The returned server is owned by the SimServer: tear it down with SimServer.Shutdown (the graceful drain) or SimServer.Close (cancel the serve context and join), never by calling Shutdown on the value returned here. An earlier version of this godoc said SimServer.Close called Shutdown; it never did — Close cancels the serve context and closes the listener, which is the ctx-cancellation stop path, not the drain path (rmp #2483).
Do NOT call SetClock on it ¶
server.Server.SetClock writes s.clk AND replaces s.txReg, and the accept path reads both unguarded (bolt/server/serve.go: sess.setClock(s.clk) and sess.setTxRegistry(s.txReg, remote)). This constructor has already started the serve goroutine by the time it returns, so injecting a clock through this accessor is a data race the detector will report. A scenario that needs a fake clock must have it installed BEFORE Serve starts, i.e. from inside the constructor — which is what NewSimServerTxRegistry does.
func (*SimServer) Shutdown ¶ added in v0.12.0
Shutdown stops the embedded server through server.Server.Shutdown: it stops accepting, drains every active connection, and — when the SimServer was built by NewSimServerOwnedCloser — closes the owned store-level closer on its drain-success branch only. It returns Shutdown's own error verbatim, including the drain-timeout error and a context expiry, so a scenario can adjudicate WHICH branch was taken.
It does NOT join the serve goroutine: on the two failure branches Shutdown leaves a still-blocked server.Server.Serve waiting on the same drain, and that goroutine's own exit path is what eventually performs the post-drain close (bolt/server/serve.go:725-738). Call SimServer.Close afterwards to join it.
Shutdown may be called more than once; the second call observes the same cached close result from the server's own sync.Once.
type SimStore ¶
type SimStore struct {
// contains filtered or unexported fields
}
SimStore is a real GoGraph persistence stack — a WAL-backed txn.Store and a cypher.Engine — whose durability layer is an in-memory SimDisk rather than the OS filesystem. It lets the deterministic simulation harness exercise the genuine WAL append+sync and recovery-replay code paths without touching real disk, so a crash (drop the in-memory engine, keep the SimDisk byte image) and a restart (reopen via real recovery) are fully reproducible from a seed.
The crash/restart boundary is the SimDisk: SimStore.Crash discards the live engine and store but the WAL bytes (and their injected fault state) persist in the SimDisk, and OpenSimStore reopens them through recovery.ReplayWAL — the same replay core that recovery.Open drives over an OS file.
Concurrency contract ¶
SimStore is NOT safe for concurrent use; the simulator drives it from a single goroutine.
func OpenSimStore ¶
OpenSimStore opens (or reopens) a store whose WAL lives in disk under [simWALPath]. When the WAL is absent the store starts empty; when it holds bytes from a prior session, recovery.ReplayWAL rebuilds the graph from the committed WAL prefix before the writer is reopened for further appends.
Reopen-for-append truncates the WAL to the last durable frame boundary (recovery.ReplayResult.WALTailOffset) BEFORE the writer seeks to end (auditor finding F1): a crash between two fsyncs leaves a benign torn tail past the committed prefix, and appending after it would strand every new frame behind junk that every subsequent reader stops at. Truncating to the recovered offset makes the reopened WAL a clean append target.
A reopen that detects genuine corruption (recovery.ReplayResult.IsClean == false) is a hard fault: the function returns an error rather than appending onto the corruption (which would permanently embed it and drop every op past the bad frame), mirroring the production recovery.Open fail-stop contract.
OpenSimStore is the string/float64 SPECIALISATION of [openSimTypedStore] (rmp #2473): it opens the codec-generic core with txn.NewStringCodec and txn.NewFloat64WeightCodec and bolts a cypher.Engine on top. The engine is what pins the specialisation — cypher.NewEngineWithStoreAndSchema takes a *txn.Store[string, float64] and nothing else — so every Cypher-driven scenario necessarily runs on the string key codec, and the codec matrix (codec_matrix.go) drives the other pairs through the typed core directly.
func (*SimStore) Checkpoint ¶ added in v0.6.0
Checkpoint runs ONE synchronous, real checkpoint over the SimDisk: it publishes a self-sufficient snapshot of the live graph to <dir>/snapshot and then prefix-truncates the WAL (<dir>/wal) for the prefix the snapshot folded. It drives the production checkpoint.Checkpointer through its synchronous checkpoint.Checkpointer.RunCheckpoint entry point, with the snapshot publish routed through the SimDisk snapshot seam ([simCheckpointBackend]) and the WAL truncation through the path-backed wal.OpenFS writer this store already holds — so the entire snapshot+WAL+truncate stack exercises the in-memory disk and a subsequent crash recovers via the FULL recovery.OpenFS path.
The critical section runs under the store's real commit serialisation (txn.Store.RunUnderCommitLock) so the snapshot is transaction-boundary consistent and the WAL prefix is reclaimed only after the self-sufficient snapshot is durable (docs/acid-audit.md F3.5). The string-key mapper codec, the engine's constraint specs and its index-definition specs are all wired so a checkpoint that truncates the WAL prefix which first declared a constraint/index cannot lose it (#1464/#1755).
Checkpoint is only meaningful in full-stack mode (the store was opened with a checkpoint directory); on a WAL-only store it returns an error rather than silently doing nothing, since a WAL-only layout has no snapshot directory to recover the truncated prefix from.
func (*SimStore) Clean ¶
Clean reports whether the most recent recovery completed without genuine on-disk corruption (a benign torn tail counts as clean).
func (*SimStore) ClockNow ¶ added in v0.12.0
ClockNow reports the MVCC clock's currently published instant. Read immediately after a reopen it is the RECOVERED clock floor; read during a run it advances by one per published commit, which is what makes it a measure of how much MVCC traffic overlapped a concurrent operation.
func (*SimStore) Close ¶
Close shuts the store down gracefully, flushing and fsyncing the WAL so every acknowledged commit is durable, then releasing the WAL writer. Use it for a clean teardown (end of a run); use SimStore.Crash to model a crash.
func (*SimStore) Config ¶ added in v0.6.0
func (s *SimStore) Config() simStoreConfig
Config returns the store configuration this SimStore was opened with, including the durable layout (the checkpoint directory in full-stack mode). It is preserved across SimStore.Crash so the simulator can reopen the store with the identical layout during crash recovery.
func (*SimStore) Crash ¶
func (s *SimStore) Crash()
Crash models a HOST crash — power failure, hard reset, hypervisor kill. It is an alias for SimStore.CrashHost, so every scenario that has not consciously chosen otherwise gets the stronger of the two models. A scenario that really means SIGKILL calls SimStore.CrashProcess and says so.
func (*SimStore) CrashHost ¶ added in v0.12.0
func (s *SimStore) CrashHost()
CrashHost models a power failure. It discards the in-memory engine, store, and WAL writer WITHOUT a graceful close, and then applies SimDisk.CrashHost to the disk: every file reverts to the bytes a SimFileHandle.Sync returning nil actually made durable, and every not-yet-dir-fsync'd dirent is revoked. What remains is exactly what stable storage held, ready for OpenSimStore to reopen and replay. The SimStore must not be used afterwards.
It deliberately does NOT call s.wlog.Close(): a clean Close would flush and fsync the buffer, which is the opposite of a crash. Dropping the references lets the GC reclaim them; the only durable state is the SimDisk image.
A frame the WAL had written but never fsync'd is therefore GONE, which is what makes an oracle of the form "the commit was acked, so the bytes are recovered" able to fail. Before rmp #2535 the byte image survived untouched and that oracle held whatever the engine did with fsync.
func (*SimStore) CrashProcess ¶ added in v0.12.0
func (s *SimStore) CrashProcess()
CrashProcess models a SIGKILL of the process (kill -9). The process dies and its in-memory engine, store and WAL writer go with it, but the kernel does not: every byte already accepted by a write(2) and every directory entry already created survives for the next process to read, fsync'd or not (see SimDisk.CrashProcess).
Only the WAL writer's own bufio buffer is lost, since that lives in the dead process's address space. Use it when the scenario means "the process was killed", and SimStore.CrashHost when it means "the machine went down" — the two are physically different events and, since rmp #2535, they leave different durable images.
func (*SimStore) Engine ¶
Engine returns the live cypher engine bound to the recovered graph and the WAL-backed store, for the simulator to drive queries through.
func (*SimStore) RecoveredIndexPopulation ¶ added in v0.12.0
func (s *SimStore) RecoveredIndexPopulation() cypher.RecoveredIndexPopulation
RecoveredIndexPopulation reports how the most recent recovery populated each secondary index the reopened engine re-registered: hydrated from the snapshot's indexes/<name>.bin payload, or rebuilt by scanning the recovered graph (rmp #2490). It is the ENGINE-SCOPED count (cypher.Engine.RecoveredIndexPopulation), captured at open, which is what makes it usable as an oracle: the `store.recovery.indexes.*` metrics are process-global, and the swarm runner builds engines on concurrent goroutines, so a metric delta cannot attribute a decision to THIS reopen.
Every field is zero for a store whose recovery re-registered no index — a fresh directory, or one whose durable schema carries no index definition.
func (*SimStore) RecoveredMaxCommitTS ¶ added in v0.12.0
RecoveredMaxCommitTS reports the highest durable MVCC commit instant the recovery observed — over every replayed commit marker and, on the full-stack path, the snapshot's recorded capture instant. Recovery has already raised the graph's MVCC clock to one past it, so it is the durable floor a post-recovery commit must exceed (rmp #2309). It is 0 when nothing durable carried one.
func (*SimStore) ResumedTxnSeq ¶ added in v0.12.0
ResumedTxnSeq reports the transaction sequence this store continued from — recovery.Result.MaxTxnSeq fed back into txn.Options.ResumeTxnSeq, so the first transaction it commits is assigned ResumedTxnSeq+1. It is 0 for a fresh store and for a recovery whose surviving WAL carried no v3 frame.
func (*SimStore) WAL ¶ added in v0.12.0
WAL returns the live WAL writer this store commits through, so an oracle can read the writer's OWN view of durability — wal.Writer.DurableOffset, wal.Writer.Stats and wal.Writer.Poisoned — alongside the durable byte image on the SimDisk (rmp #2472). Reading those two independently is the point: the byte image is what exists, the writer's counters are what it believes, and a watermark defect is a disagreement between them.
The returned writer is OWNED by the store: a caller must only READ it. Appending or syncing through it directly bypasses the transaction layer's sequence minting and apply gate, and SimStore.Close closes it.
type Simulator ¶
type Simulator struct {
// contains filtered or unexported fields
}
Simulator drives the real cypher.Engine against a shadow GraphOracle under a deterministic, single-goroutine, tick-driven loop, verifying ACID and graph invariants after operations.
Concurrency contract ¶
Simulator is NOT safe for concurrent use and spawns no goroutines. Its determinism guarantee depends on a single, totally-ordered stream of draws from one Seed; Simulator.Run must be called from one goroutine.
func New ¶
New builds a Simulator with a fresh in-memory engine, oracle, checker, clock, and (Phase-2-bound, currently unwired) SimDisk, all driven by cfg.Seed. It returns an error only for an invalid configuration.
func (*Simulator) CheckpointCount ¶ added in v0.6.0
CheckpointCount returns how many in-loop checkpoints the run published (always 0 when checkpointing is disabled). Each checkpoint published a self-sufficient snapshot and truncated the WAL prefix it folded.
func (*Simulator) Close ¶
Close releases the simulator's durable resources. In crash mode it gracefully closes the live SimDisk-backed store (flushing and releasing the WAL writer) so no handle or goroutine leaks past the run; in the default in-memory mode it is a no-op. It is safe to call more than once.
func (*Simulator) CrashCount ¶
CrashCount returns how many crash+recovery cycles the run performed (always 0 when crashes are disabled).
func (*Simulator) Disk ¶ added in v0.6.0
Disk returns the SimDisk backing the durable store, for tests that inspect the durable image (e.g. asserting a snapshot directory exists after a checkpoint). It is non-nil whenever a durable store is in use.
func (*Simulator) NodeIDsCompared ¶ added in v0.16.0
NodeIDsCompared returns how many surviving nodes the NodeID stability oracle compared across the run's recoveries (WAL v2 step 3).
func (*Simulator) Oracle ¶
func (s *Simulator) Oracle() *GraphOracle
Oracle returns the simulator's shadow model, for tests that assert on the modelled state after a run.
func (*Simulator) RejectedReads ¶ added in v0.6.0
RejectedReads returns how many read-shaped operations the engine refused (drain error) during the run. Under the mem-pressure scenario it is the non-vacuity guard that a logical-resource budget fired.
func (*Simulator) RejectedWrites ¶ added in v0.6.0
RejectedWrites returns how many write-shaped operations the engine did not commit during the run. Under the disk-full scenario it is the non-vacuity guard that ENOSPC fired; it is 0 for a run that never exhausts the disk.
func (*Simulator) ReplayedOps ¶
ReplayedOps returns the cumulative number of WAL ops recovery replayed across every crash cycle in the run.
func (*Simulator) Run ¶
Run executes the safety-phase tick loop. Each tick advances the clock, selects an actor, asks it for an operation, runs that operation against the engine, applies it to the oracle, and (every CheckEvery ticks) verifies the invariants. On the first violation it returns a populated SimReport; on clean completion it returns (nil, nil). It honours ctx cancellation and deadlines, returning the ctx error if the run is interrupted.
The loop runs entirely on the calling goroutine and spawns none; engine operations are synchronous.
type SlowConsumer ¶
type SlowConsumer struct {
// contains filtered or unexported fields
}
SlowConsumer opens a large result stream and then pulls records very slowly (or stalls entirely), exercising the server's streaming backpressure. Because the SimConn write buffer is bounded ([simConnBufferSize]), a stalled consumer forces the server's record-write to BLOCK once the buffer fills rather than letting it buffer the whole result in memory — the bounded-resource property under a slow reader. When the connection is finally closed (or its read deadline, driven by the injected Clock, elapses) the server must tear the session down without leaking a goroutine.
SlowConsumer runs in the CONCURRENT mode: the slow pulls happen on a real goroutine while the server's writer goroutine is parked on backpressure. The SEED controls the stall timing; interleaving is real (per the hybrid model), so correctness here is backpressure + no-leak, not bit-replay.
Concurrency contract ¶
A SlowConsumer drives one connection it owns; each SlowConsumer.Stall call is independent and may run on its own goroutine.
func NewSlowConsumer ¶
func NewSlowConsumer(clk clock.Clock) *SlowConsumer
NewSlowConsumer returns a SlowConsumer whose stall timing is driven by clk.
func (*SlowConsumer) Stall ¶
func (s *SlowConsumer) Stall(ctx context.Context, srv *SimServer, stallFor time.Duration, onStalled func(*WireClient)) (SlowConsumerResult, error)
Stall opens a large result stream over the server and then deliberately stalls without pulling, holding the stream open for stallFor (measured on the injected clock). While stalled, the server's writer is parked on the bounded SimConn buffer (backpressure). It then closes the connection and returns the result. The connection is always closed before return, so no goroutine leaks.
onStalled, when non-nil, is invoked once while the consumer is stalled, with the client so a caller can inspect the bounded buffer (WireClient.Conn → SimConn.ReadBuffered) and confirm the server did not buffer the whole result.
stallFor stalls are driven by the injected Clock so a Fake makes the stall deterministic; with clock.Real it is a real (short) sleep.
type SlowConsumerResult ¶
type SlowConsumerResult struct {
// RecordsPulled is how many RECORDs the slow consumer drained before it
// stopped. It is always far less than the total result size when the consumer
// stalls, which is the proof the server did not push the whole result eagerly.
RecordsPulled int
// ServerParked reports whether the server's write blocked on the bounded
// SimConn buffer while the consumer stalled (backpressure observed) rather
// than the whole result being buffered ahead of the consumer.
ServerParked bool
// ClosedCleanly reports whether the connection tore down without a transport
// fault other than the expected close/EOF.
ClosedCleanly bool
}
SlowConsumerResult summarises one slow-consumer run. It records whether the server kept its memory bounded while the consumer stalled (proved by the SimConn write-buffer never being allowed to grow past its bound — the server parks on its blocked record write rather than buffering the whole result) and whether the connection torn down without leaking a goroutine.
type StatsEngine ¶ added in v0.12.0
type StatsEngine interface {
PlanEngine
// StatsTrackedPairs reports how many distinct (label, property) pairs the
// engine currently holds planner statistics for; 0 on an engine that never
// completed a refresh.
StatsTrackedPairs() int
}
StatsEngine is the engine surface the statistics-regime checker needs: the plan surface (results + Explain) plus the tracked-pairs observable that proves a rebuild actually published per-(label, property) statistics. The simulator's EngineAdapter satisfies it.
Concurrency contract ¶
Implementations need only be safe for single-goroutine use; the simulator never calls them concurrently.
type StatsRegime ¶ added in v0.12.0
type StatsRegime struct {
// contains filtered or unexported fields
}
StatsRegime is the statistics-driven planning regime checker: it drives CALL db.stats.refresh() at deterministic and seed-chosen points, pins the procedure's row/throttle contract, asserts result identity across every refresh, and accumulates the run's regime statistics. It is stateful so StatsRegime.Finish can assert non-vacuity over the whole run: at least one completed rebuild with a non-zero tracked-pairs observable and at least one exercised refusal, otherwise the statistics path never engaged and the run proved nothing about it.
Concurrency contract ¶
StatsRegime is NOT safe for concurrent use; the simulator drives it from the single simulation goroutine.
func NewStatsRegime ¶ added in v0.12.0
func NewStatsRegime(probes ...ParityProbe) *StatsRegime
NewStatsRegime builds the checker over the fixed probe set whose answers must survive every refresh unchanged.
func (*StatsRegime) CheckRecovered ¶ added in v0.12.0
func (k *StatsRegime) CheckRecovered(tick int64, engine StatsEngine) []Violation
CheckRecovered pins the post-crash statistics regime: the collector is per-engine, in-memory, and never rebuilt by recovery, so a freshly recovered engine must report ZERO tracked pairs until the next explicit refresh. A non-zero count means recovery started rebuilding statistics (a contract change this oracle must be told about) or collector state leaked across the crash. Call it immediately after every recovery, before the post-recovery ExpectRebuild refresh.
func (*StatsRegime) CheckRefresh ¶ added in v0.12.0
func (k *StatsRegime) CheckRefresh(tick int64, engine StatsEngine, expect RefreshExpectation) []Violation
CheckRefresh drives one CALL db.stats.refresh() through engine and returns a violation for each contract breach found:
- the probe battery (both arms of every probe) answering differently immediately before and immediately after the call is a ViolationACIDConsistency — a statistics refresh changed a RESULT;
- a malformed row (not exactly one row, unreadable columns, a detail that does not match its ok verdict) is a ViolationOracleDeviation;
- an outcome contradicting expect (ExpectRebuild refused, or ExpectRefusal rebuilt) is a ViolationOracleDeviation;
- a rebuild that leaves StatsEngine.StatsTrackedPairs at zero, or a refusal that CHANGES it, is a ViolationOracleDeviation.
A probe arm whose Explain rendering changed across the call increments StatsRegime.PlanChanges and is never a violation.
func (*StatsRegime) Finish ¶ added in v0.12.0
func (k *StatsRegime) Finish(tick int64) []Violation
Finish asserts non-vacuity over the whole run: at least one completed rebuild left a non-zero tracked-pairs observable, and at least one refusal exercised the rate limit. A run that never engaged the statistics path — or never proved the throttle refuses — verified nothing about the statistics-driven planning regime and is reported as a ViolationOracleDeviation rather than passing silently. Call it once, at the end of the scenario.
func (*StatsRegime) PlanChanges ¶ added in v0.12.0
func (k *StatsRegime) PlanChanges() int
PlanChanges reports how many probe-arm Explain renderings changed across a refresh over the whole run. A plan change is legal — statistics are a planner input — so this is the scenario's report channel, never a failure.
func (*StatsRegime) Refreshes ¶ added in v0.12.0
func (k *StatsRegime) Refreshes() int
Refreshes reports how many completed rebuilds (ok=true) the run observed.
func (*StatsRegime) Refusals ¶ added in v0.12.0
func (k *StatsRegime) Refusals() int
Refusals reports how many in-window refusals (ok=false) the run observed.
type SurfaceWriter ¶ added in v0.6.0
type SurfaceWriter struct {
// contains filtered or unexported fields
}
SurfaceWriter builds the graph the cypher-surface battery reads: it creates Person nodes that ALWAYS carry name+age+city (so aggregate/filter/grouping invariants have no null ambiguity) and KNOWS edges between existing Persons. It avoids MERGE and SET so the oracle's Person model is unambiguous.
Concurrency contract ¶
SurfaceWriter is NOT safe for concurrent use; it is invoked from the single simulation goroutine.
func (*SurfaceWriter) Name ¶ added in v0.6.0
func (*SurfaceWriter) Name() string
Name returns the actor's identifier.
func (*SurfaceWriter) NextOp ¶ added in v0.6.0
func (w *SurfaceWriter) NextOp(seed *Seed, oracle *GraphOracle) Op
NextOp returns a fresh Person CREATE (name, age, and a city from the seeded c0..c9 vocabulary) or a KNOWS edge between two existing Persons, seed-derived.
type Swarm ¶
type Swarm struct {
// contains filtered or unexported fields
}
Swarm runs many independent seeds of a scenario across a bounded worker pool, time-boxed by run count and/or wall-clock duration, and aggregates the outcomes. Each worker runs ONE seed at a time against its own freshly-built scenario harness; the only shared mutable state is the aggregator, guarded by a mutex. Per-run determinism is whatever the scenario's mode guarantees (a deterministic scenario reproduces bit-for-bit from its derived seed); only the across-run scheduling is concurrent.
Concurrency contract ¶
A Swarm is built with NewSwarm and run once with Swarm.Run, which spawns exactly Workers goroutines and joins them all before returning — it leaks no goroutine. Swarm is not intended for reuse across Swarm.Run calls.
func NewSwarm ¶
func NewSwarm(reg *Registry, cfg *SwarmConfig) (*Swarm, error)
NewSwarm builds a swarm over reg with cfg. It validates that the budget is well-formed (at least one of Runs or Duration is positive) and that the configured scenario resolves, returning an error otherwise so a misconfiguration fails fast rather than running an empty or unbounded swarm.
func (*Swarm) Run ¶
func (s *Swarm) Run(ctx context.Context) (SwarmResult, error)
Run executes the swarm and returns its aggregated result. It spawns the resolved number of worker goroutines, each pulling run indices from a bounded dispatch channel, running the scenario for the derived seed, and pushing the outcome through the mutex-guarded aggregator. Scheduling stops when the run budget is exhausted, the duration budget elapses, or ctx is cancelled; in-flight runs always finish so no goroutine is abandoned. The returned error is non-nil only for a ctx cancellation that interrupted scheduling.
type SwarmConfig ¶
type SwarmConfig struct {
// Clock is the time source the duration budget reads. When nil, the real
// clock is used. The per-run simulations remain seed-driven and never read
// this clock — it bounds only the across-run scheduling, which is concurrent
// and not bit-reproducible by construction.
Clock clock.Clock
// Selector, when non-nil, is consulted before each run to choose the
// scenario for that run (coverage-biased selection, Phase 5 #1565). When
// nil, every run uses Scenario. The selector must be safe for concurrent use
// because workers call it from many goroutines.
Selector ScenarioSelector
// Observe, when non-nil, is called once per completed run with its outcome.
// It runs under the aggregator's lock (so it is serialised across workers)
// and must not block; it is an observation hook for coverage feeding and
// live reporting.
Observe func(SwarmRun)
// Scenario is the name of the catalogue scenario each run executes. It must
// resolve in the registry the swarm was built with.
Scenario string
// MasterSeed seeds the deterministic derivation of every per-run seed, so
// the whole swarm reproduces from this one value. Two swarm runs with the
// same MasterSeed, Scenario, and Runs execute the identical set of seeds.
MasterSeed uint64
// Workers is the worker-pool cap. Values <= 0 are normalised to
// min(GOMAXPROCS, max(1, Runs)). The pool never spawns more than this many
// run goroutines at once.
Workers int
// Runs is the seed-count budget: the swarm executes exactly this many runs
// (each a distinct derived seed) unless Duration elapses first. When Runs <=
// 0 the swarm is duration-bounded only and MUST carry a positive Duration.
Runs int
// Duration is the wall-clock budget. When > 0 the swarm stops scheduling new
// runs once it elapses (in-flight runs finish). Zero means no time bound, in
// which case Runs MUST be positive.
Duration time.Duration
}
SwarmConfig parameterises a Swarm run: the master seed, the scenario to run, the worker cap, and the budget (by run count, wall-clock duration, or both — whichever bound is hit first ends the swarm).
type SwarmResult ¶
type SwarmResult struct {
// Failures lists every failing run in ascending run-index order, so a report
// is deterministic regardless of worker completion order.
Failures []SwarmRun
// MasterSeed is the seed the schedule was derived from (for reproducing the
// whole swarm).
MasterSeed uint64
// Runs is how many runs actually executed (<= the Runs budget; fewer when
// the duration budget cut it short).
Runs int
// Passes is the number of runs that found no violation and no harness error.
Passes int
// Elapsed is the wall-clock time the swarm took.
Elapsed time.Duration
// Workers is the worker cap the swarm actually ran with.
Workers int
}
SwarmResult aggregates a completed swarm: the totals, the wall-clock elapsed, and every failing run with its reproduce line. It is the value the CLI and the integration tests assert on.
func (SwarmResult) FailureCount ¶
func (r SwarmResult) FailureCount() int
FailureCount returns the number of failing runs.
func (SwarmResult) Summary ¶
func (r SwarmResult) Summary() string
Summary renders a one-block human-readable summary: totals, throughput, and every failing seed's reproduce line. It always ends without a trailing newline so the caller controls spacing.
func (SwarmResult) Throughput ¶
func (r SwarmResult) Throughput() float64
Throughput returns runs per second over the elapsed wall-clock, or 0 when no time elapsed (avoids a divide-by-zero in a degenerate empty run).
type SwarmRun ¶
type SwarmRun struct {
// Err is a harness error (setup/transport failure), else nil. A run with a
// non-nil Err counts as a failure distinct from an invariant violation.
Err error
// Report is the failure report when the run found an invariant violation,
// else nil.
Report *SimReport
// Scenario is the scenario name this run executed.
Scenario string
// Index is the run's position in the deterministic schedule (0-based).
Index int
// Seed is the derived per-run seed.
Seed uint64
}
SwarmRun is the outcome of one seed in a swarm: the seed it ran, the scenario name, whether it failed, the failure report (nil on pass), and any harness error (a transport/setup failure that is not itself an invariant violation).
func (SwarmRun) Failed ¶
Failed reports whether the run failed for any reason (an invariant violation or a harness error).
func (SwarmRun) ReproduceLine ¶
ReproduceLine returns a copy-pasteable command that re-runs exactly this (scenario, seed) under the CLI, so a swarm failure is reproducible verbatim.
type SyncGate ¶ added in v0.12.0
type SyncGate struct {
// contains filtered or unexported fields
}
SyncGate is a one-shot rendezvous on a chosen SimFileHandle.Sync call, returned by SimDisk.ArmSyncGateAt. The gated Sync blocks inside the call — with the disk lock RELEASED, so the rest of the disk stays usable — until SyncGate.Release is called, which is what lets a scenario hold the WAL group-commit leader inside its fsync while other committers arrive and become followers.
Concurrency contract ¶
SyncGate is safe for concurrent use. SyncGate.Reached may be received from any goroutine and SyncGate.Release may be called from any goroutine and more than once; the gated Sync itself runs on whichever goroutine issued it.
func (*SyncGate) Fired ¶ added in v0.12.0
Fired reports whether the gated Sync was actually entered. It is the reachability observable the rename arms established (rmp #2465): an ordinal that never matched is a silent no-op, and a scenario that depends on the gate firing must be able to tell "the gate held the leader" from "the gate was never reached" rather than misreading the resulting timing as an engine property.
func (*SyncGate) Reached ¶ added in v0.12.0
func (g *SyncGate) Reached() <-chan struct{}
Reached returns a channel closed when the gated Sync has been entered and is blocked. Receiving from it is how a controlling goroutine learns the leader is parked inside its fsync, rather than guessing with a sleep.
func (*SyncGate) Release ¶ added in v0.12.0
func (g *SyncGate) Release()
Release unblocks the gated Sync, which then returns whatever outcome the other arms selected for it (a SimDisk.ArmSyncFaultAt fault, an ENOSPC, or success). It is idempotent and safe to call even if the gate was never reached.
type Trace ¶
type Trace struct {
// Ops is the ordered operation stream.
Ops []TracedOp
// CrashTicks lists the ticks at which a crash+recovery cycle fired during
// recording, in order. Scripted replay runs against a plain in-memory engine
// and does not re-inject crashes (the violations the shrinker targets are
// oracle/engine divergences reproducible without a crash); the list is
// retained for the report and for completeness.
CrashTicks []int64
// Seed is the seed the recording was driven by (informational; a scripted
// replay does NOT draw from it — it executes Ops directly).
Seed uint64
}
Trace is the full recorded operation stream of a deterministic run, plus a note of which crash ticks fired. Only the deterministic engine-API mode produces a Trace: it is bit-reproducible, so replaying the same Trace against a fresh engine/oracle/checker reaches the identical end-state. Concurrent and liveness modes are not bit-replayable and are not recorded.
Concurrency contract ¶
A Trace is a plain value; callers own their copies and do not share them across goroutines mid-mutation.
func ParallelAggregateTrace ¶ added in v0.10.0
func ParallelAggregateTrace() Trace
ParallelAggregateTrace builds a deterministic, self-contained trace that seeds a graph with UNIQUE-valued (tie-free) numeric properties across three groups, then issues the min / max / count aggregations — scalar and grouped — that exercise the parallel aggregate scan (#2111). With unique extrema the result is order-independent, so the parallel (variant A) and serial (variant B) engines — which build separate graphs whose scan order is not stable across builds (see ParallelAggregateVariantPair) — must still agree on every op. The values mix int and float across groups so the Number-tier ordering is exercised, but no two values are Compare-equal, so the retained min/max representation is unambiguous.
type TraceFault ¶
type TraceFault string
TraceFault names a deterministic fault a scripted replay injects at a specific op, so a trace can carry a reproducible failure for replay verification and shrinking. Faults are a test-only mechanism of the scripted executor; a recording from a real run carries FaultNone on every op unless a fault is explicitly injected.
const ( // FaultNone is the absence of an injected fault (a normal op). FaultNone TraceFault = "" // FaultDropEngineWrite makes the scripted executor APPLY a write to the oracle // but SKIP it on the engine, creating a deterministic oracle-vs-engine // divergence the per-op check detects. It models a lost-write bug and is the // canonical injected violation the shrinking demo reduces. FaultDropEngineWrite TraceFault = "drop-engine-write" )
Trace faults.
type TracedOp ¶
type TracedOp struct {
// Op is the Cypher operation that was issued.
Op Op
// Fault, when non-empty, is a marker injected for replay/shrinking: the
// scripted executor applies the named fault deterministically when it reaches
// this op. The empty string means no fault (a normal op). See [TraceFault].
Fault TraceFault
// Tick is the simulated tick the op ran at during recording. On replay the
// scripted executor re-derives ticks positionally, so this is informational
// (it lets a report point at the original tick).
Tick int64
}
TracedOp is one entry in a recorded Trace: the tick it ran at, the operation issued, and an optional injected fault marker. A trace is the ordered list of these, captured during a deterministic run, and is the unit a scripted replay executes and the shrinker reduces.
type TxnOversizeAttempt ¶ added in v0.12.0
type TxnOversizeAttempt struct {
// Name identifies the attempt in a failure message.
Name string
// Ops is the number of ops the transaction buffered — exactly, by
// construction, since every buffering call appends exactly one op.
Ops int
// Keys are the node keys the transaction created, in issue order.
Keys []string
// Refused reports that Commit returned an error.
Refused bool
// Sentinel reports that the refusal satisfies
// errors.Is(err, txn.ErrTransactionTooLarge) — the typed class a caller needs
// to tell an over-cap refusal from a conflict or an I/O fault.
Sentinel bool
// Err is the commit error rendered for a failure message ("" when none).
Err string
// WALBefore / WALAfter are the durable WAL image lengths on the SimDisk,
// read immediately before and after the attempt.
WALBefore, WALAfter int
// WALIdentical reports that the two durable images compared BYTE-FOR-BYTE
// equal, not merely equal in length. This is the assertion that a refusal
// wrote nothing, as distinct from having written and then truncated.
WALIdentical bool
// OrderBefore / OrderAfter are the live graph's node count across the
// attempt, the in-memory half of the same question.
OrderBefore, OrderAfter uint64
}
TxnOversizeAttempt is what ONE commit attempt did. It holds measurements and no verdict: whether an attempt SHOULD have been refused is derived by the adjudicator from the store's effective cap, never recorded here.
func (TxnOversizeAttempt) String ¶ added in v0.12.0
func (a TxnOversizeAttempt) String() string
String renders one attempt for a failure message or a test log.
type TxnOversizeConfig ¶ added in v0.12.0
type TxnOversizeConfig struct {
// Seed is the master seed for the SimDisk sub-stream. The workload itself is
// fixed: the arms are boundary sizes, not sampled ones.
Seed uint64
// Cap is the per-transaction op cap the store is opened with, in the
// simulator convention: 0 selects [txn.DefaultMaxTxnOps],
// [txn.MaxTxnOpsUnlimited] disables the cap, any positive value is taken
// verbatim. The capped arm passes [txnOversizeCap]; the unlimited arm passes
// [txn.MaxTxnOpsUnlimited] and drives the SAME op counts.
Cap int
// UncappedProducerSeam opens the store with the producer bound left at
// [txn.DefaultMaxTxnOps] while recovery still honours Cap — the simulator
// exactly as it behaved before rmp #2474. It is the sensitivity seam of these
// oracles and must be set only by the test that proves they can fail; see
// simStoreConfig.uncappedProducerSeam.
UncappedProducerSeam bool
}
TxnOversizeConfig parameterises one producer run.
type TxnOversizeEvidence ¶ added in v0.12.0
type TxnOversizeEvidence struct {
// Cap is the value the store was opened with, in the simulator convention.
Cap int
// EffectiveCap is that value RESOLVED the way txn.resolveMaxTxnOps resolves
// it: the op count above which a commit is refused, or 0 for "no cap". It is
// what the adjudicator derives every per-attempt expectation from.
EffectiveCap int
// Attempts is every commit attempt, in the order driven.
Attempts []TxnOversizeAttempt
// PreOversizeWALBytes is the durable WAL length immediately before the FIRST
// attempt that exceeds [txnOversizeCap]. It is the non-vacuity measurement:
// a cap adjudicated against an empty file proves nothing.
PreOversizeWALBytes int
// MaxAttemptOps is the largest transaction the run drove, which the
// non-vacuity gate compares against [txnOversizeCap].
MaxAttemptOps int
// ReopenClean reports that the reopen's recovery found no genuine corruption.
ReopenClean bool
// RecoveredOrder is the reopened graph's live node count, read independently
// of the per-key probes below.
RecoveredOrder uint64
// ModelKeys are the keys of every COMMITTED attempt: what recovery must
// return, and nothing else.
ModelKeys []string
// MissingKeys are the model keys the reopened graph did NOT hold — durable
// loss.
MissingKeys []string
// RefusedKeys are the keys of every refused attempt.
RefusedKeys []string
// ResurrectedKeys are the refused keys the reopened graph DID hold — a
// rejected transaction made durable, the atomicity half of the contract.
ResurrectedKeys []string
}
TxnOversizeEvidence is what a producer run OBSERVED, across the attempts and the reopen that follows them.
func RunTxnOversizeProducer ¶ added in v0.12.0
func RunTxnOversizeProducer(ctx context.Context, cfg TxnOversizeConfig) (TxnOversizeEvidence, error)
RunTxnOversizeProducer opens a store with cfg.Cap, drives the fixed sequence of boundary-sized transactions through it, and reopens it through real recovery — reporting what each attempt did to the durable WAL image and to the live graph, and what survived the reopen.
The store is driven directly through txn.Store rather than through the Cypher engine, because the cap counts OPS and only the transaction layer lets a scenario buffer an exact number of them. A statement's op count is a property of the planner, which would make "exactly at the cap" unreproducible.
func (*TxnOversizeEvidence) Committed ¶ added in v0.12.0
func (e *TxnOversizeEvidence) Committed() int
Committed counts the attempts whose commit returned nil.
func (*TxnOversizeEvidence) String ¶ added in v0.12.0
func (e *TxnOversizeEvidence) String() string
String renders the run for a failure message or a test log.
type TxnOversizeReplayArm ¶ added in v0.12.0
type TxnOversizeReplayArm struct {
// Name identifies the arm in a failure message.
Name string
// RunOps is the number of op frames in the marker-less run that follows the
// committed prefix, before its own commit marker closes it.
RunOps int
// Cap is the replay cap the arm drives, in the simulator convention.
Cap int
// EffectiveCap is Cap resolved to the buffered-op count at which replay
// stops, or 0 for "no cap".
EffectiveCap int
// Bytes is the crafted file's length on the SimDisk.
Bytes int
// TailErr is the replay stop reason rendered for a message ("" when nil).
TailErr string
// Sentinel reports errors.Is(TailErr, recovery.ErrTransactionTooLarge).
Sentinel bool
// Clean is [recovery.ReplayResult.IsClean]: false when the stop reason is
// classified as genuine corruption, which is what makes a fail-stop
// fail-STOP rather than a tolerated tail.
Clean bool
// WALOps is how many ops replay applied to the graph.
WALOps int
// Order is the replayed graph's live node count, read independently of
// WALOps.
Order uint64
// HarnessRefused reports that opening the same image through the simulator's
// own store-open path returned an error rather than a store, which is the
// end-to-end half: a fail-stop that the embedder swallowed would append onto
// the corruption.
HarnessRefused bool
// HarnessSentinel reports that the harness error carries
// [recovery.ErrTransactionTooLarge].
HarnessSentinel bool
}
TxnOversizeReplayArm is one crafted-WAL replay arm: the file that was built and what replaying it under a given cap did.
func (TxnOversizeReplayArm) String ¶ added in v0.12.0
func (a TxnOversizeReplayArm) String() string
String renders one arm for a failure message or a test log.
type TxnOversizeReplayEvidence ¶ added in v0.12.0
type TxnOversizeReplayEvidence struct {
// PriorOps is the committed prefix every arm's file carries.
PriorOps int
// Arms is every arm, in the order driven.
Arms []TxnOversizeReplayArm
}
TxnOversizeReplayEvidence is what the crafted-WAL sweep observed.
func RunTxnOversizeReplay ¶ added in v0.12.0
func RunTxnOversizeReplay(ctx context.Context, seed uint64) (TxnOversizeReplayEvidence, error)
RunTxnOversizeReplay builds a hand-crafted WAL per arm and replays it under that arm's cap, reporting what recovery did with it.
Each arm gets its OWN SimDisk, so an arm's replay (and the harness open that follows it, which truncates a recovered tail) cannot perturb the next.
func (TxnOversizeReplayEvidence) String ¶ added in v0.12.0
func (e TxnOversizeReplayEvidence) String() string
String renders the sweep for a failure message or a test log.
type TypedSchemaConfig ¶ added in v0.12.0
type TypedSchemaConfig struct {
// Seed is the master seed. Every sub-stream (the probe sweep, the durable op
// builder, the crash schedule, the SimDisk) derives from it, so the whole run
// — including [TypedSchemaEvidence.Digest] — is a pure function of this value.
Seed uint64
// MaxTicks bounds the deterministic loop. It must admit at least one full
// fifteen-cell sweep epoch or the coverage gate fires — which is exactly the
// seam the gate's own meta-test drives.
MaxTicks int
// SideNodes is how many nodes the side fixture chains together.
SideNodes int
// NodeEvery and WitnessEvery are the in-loop cadences, in ticks, of the
// ValidateNode boundary battery and of witness arming.
NodeEvery int
WitnessEvery int
// Crash is the crash/recovery schedule, and Checkpoint the in-loop
// checkpoint cadence. Both are fields rather than constants so a test can
// disable one and watch the corresponding non-vacuity gate fire.
Crash CrashConfig
Checkpoint CheckpointConfig
}
TypedSchemaConfig parameterises a typed-schema run. The zero value is not usable; DefaultTypedSchemaConfig fills in the short-layer budgets and [TypedSchemaConfig.normalise] repairs any field a caller left at zero, so a test can override one field without restating the rest.
func DefaultTypedSchemaConfig ¶ added in v0.12.0
func DefaultTypedSchemaConfig(seed uint64) TypedSchemaConfig
DefaultTypedSchemaConfig returns the short-layer configuration for seed.
type TypedSchemaEvidence ¶ added in v0.12.0
type TypedSchemaEvidence struct {
// Coverage[path][verdict] counts how many side-battery writes landed in each
// of the fifteen cells. Every cell is visited once per sweep epoch, so a zero
// means the sweep did not complete even one epoch.
Coverage [tsPathCount][tsVerdictCount]int
// NoMutationChecks[path] counts the rejections whose no-mutation battery ran
// on that path, and AcceptLandedChecks[path] the accepts whose read-back was
// verified.
NoMutationChecks [tsPathCount]int
AcceptLandedChecks [tsPathCount]int
// CrossAccessorChecks counts the rejections whose SECOND per-pair accessor was
// also compared, and KeyInterningChecks the rejections that asserted the
// unregistered key stayed out of the property-key registry.
CrossAccessorChecks int
KeyInterningChecks int
// FusedNoEdgeChecks counts the rejected fused writes that asserted no edge and
// no endpoint node appeared.
FusedNoEdgeChecks int
// NodeVerdicts[verdict] counts the whole-node observations by outcome, and
// the four Validate* counters the individual fixture clauses. They are
// separate because a single "ValidateNode ran" counter cannot distinguish a
// run that only ever saw the OK case.
NodeVerdicts [tsNodeVerdictCount]int
ValidateMidBuild int
ValidateFinalised int
ValidateUnlabelled int
ValidateGhost int
ValidatePreInstall int
// EngineVerdicts[verdict] counts the durable (Cypher-path) writes by outcome,
// and EngineRejectedCreateChecks the rejected CREATEs whose no-node clause ran.
EngineVerdicts [tsVerdictCount]int
EngineRejectedCreateChecks int
// WitnessesArmed is how many (accepted value, then rejected value) witness
// pairs the run committed, and WitnessReadsAfterRecovery how many witness
// verifications ran on a graph that came back through real recovery.
// WitnessCypherReads and WitnessSubstrateReads count the two INDEPENDENT
// channels each verification used, so a channel that silently stopped
// answering is visible rather than absorbed.
WitnessesArmed int
WitnessReadsAfterRecovery int
WitnessCypherReads int
WitnessSubstrateReads int
// The post-recovery pin's five clauses, counted separately: a single counter
// could not tell a run that only reached the first from one that reached all.
PinNoValidatorAccepted int
PinNoValidatorNodeClean int
PinReinstalledRejected int
PinValidateNodeDetected int
PinValidateNodeRepaired int
// Crashes / Checkpoints / ForcedCrashes are the recovery coverage the pin and
// witness clauses depend on.
Crashes int
Checkpoints int
ForcedCrashes int
// The pure-store arm's measurements (the finding in the file header):
// PureStoreArms is how many times it ran, PureStoreRefusedAtBuffer how often
// txn.Tx.SetNodeProperty refused the value before buffering it — which is
// where the guard has lived since rmp #2602, and what replaced counting
// [txn.ErrCommittedNotApplied] from a Commit that no longer has anything left
// to reject —
// PureStoreLiveAbsent how often the LIVE graph was left without the rejected
// value, PureStoreResurrected how often recovery brought it back, and
// PureStoreAcceptedSurvived how often the ACCEPTED sibling value survived (the
// non-vacuity half: without it, "absent" could just mean recovery did
// nothing).
PureStoreArms int
PureStoreRefusedAtBuffer int
PureStoreLiveAbsent int
PureStoreResurrected int
PureStoreAcceptedSurvived int
// SideBatteries and NodeBatteries are how many times each battery ran.
SideBatteries int
NodeBatteries int
// Digest folds every clause's (tick, clause, verdicts) triple. It is the
// scenario's reproducibility claim: same seed, same digest. It folds no
// NodeID and no mapper key, both of which come from a process-global counter
// and are not a function of the seed.
Digest uint64
}
TypedSchemaEvidence is what the run MEASURED, handed back so a test asserts on numbers rather than on the mere absence of a violation, and so the report prints what actually happened.
Following the shape FluentQueryEvidence uses, the checker itself IS the record: the non-vacuity gates in TypedSchemaEvidence.Finish read these very fields, so a test asserting on them cannot drift from what the gates enforce.
func (*TypedSchemaEvidence) Finish ¶ added in v0.12.0
func (e *TypedSchemaEvidence) Finish(tick int64) []Violation
Finish is the terminal assert-something-was-seen gate: it reports a violation for every arm, cell and clause the run did NOT reach.
It is deliberately unconditional on the configuration — it fires just as loudly when a budget was lowered as when a clause was deleted — so it cannot be silenced by the very change it exists to catch.
func (*TypedSchemaEvidence) ReproducibleSummary ¶ added in v0.12.0
func (e *TypedSchemaEvidence) ReproducibleSummary() string
ReproducibleSummary renders exactly the fields that are a pure function of the seed — which here is every field, because nothing this scenario measures depends on a background goroutine. It exists so the determinism test compares what the scenario CLAIMS is reproducible, and so a future field that is NOT reproducible has an obvious place to be excluded from.
func (*TypedSchemaEvidence) String ¶ added in v0.12.0
func (e *TypedSchemaEvidence) String() string
String renders the evidence for a report and for the run's own output.
type TypedSchemaProbes ¶ added in v0.12.0
type TypedSchemaProbes struct {
// contains filtered or unexported fields
}
TypedSchemaProbes is the stateful checker: it runs the write battery, the ValidateNode boundary battery and the post-recovery pin on demand, and accumulates the evidence its terminal non-vacuity gate reads.
Concurrency contract ¶
TypedSchemaProbes is NOT safe for concurrent use. It draws from a Seed and issues reads that need a quiescent view of the graph, so it must be driven from the single simulation goroutine — the same contract CheckSearch and FluentQueryProbes carry, and for the same reason.
func NewTypedSchemaProbes ¶ added in v0.12.0
func NewTypedSchemaProbes(seed *Seed) *TypedSchemaProbes
NewTypedSchemaProbes returns a probe battery drawing its cells and values from seed. The seed must be derived from the run seed and must NOT be the durable workload's, so the durable op stream stays a pure function of the master seed.
func (*TypedSchemaProbes) ArmWitness ¶ added in v0.12.0
ArmWitness commits one durability witness: a Person created with an ACCEPTED age, immediately followed by a REFUSED age on the same node.
The pair is what makes the post-recovery clause non-vacuous. "The refused value is absent" is unfalsifiable on its own — a recovery that replayed nothing would satisfy it — so every witness also carries a value that MUST come back.
func (*TypedSchemaProbes) EngineWrite ¶ added in v0.12.0
func (p *TypedSchemaProbes) EngineWrite( ctx context.Context, sm *Simulator, tick int64, op Op, want tsVerdict, ) []Violation
EngineWrite drives one durable statement and adjudicates its verdict against the model, then — for a refused CREATE — asserts the statement left no node behind.
It advances the GraphOracle only when the statement committed, exactly as [Simulator.applyToOracle] requires, so the harness's own count parity and the crash-boundary durability check stay meaningful for the whole run.
func (*TypedSchemaProbes) Evidence ¶ added in v0.12.0
func (p *TypedSchemaProbes) Evidence() *TypedSchemaEvidence
Evidence returns the accumulating record. The pointer is owned by the probes and is live for the whole run.
func (*TypedSchemaProbes) InstallEngineSchema ¶ added in v0.12.0
InstallEngineSchema binds a FRESH schema.Schema to g's own registries, installs it as g's validator, and records it (and its model) on the probes.
It is called once before the loop and again after every recovery. The schema is rebuilt rather than re-installed because a recovered graph has fresh registries: schema.New mints property-key and label ids through the registries it is handed, so re-installing a schema built over the crashed graph's registries would hand out ids that no longer describe this graph.
func (*TypedSchemaProbes) NodeBattery ¶ added in v0.12.0
func (p *TypedSchemaProbes) NodeBattery( tick int64, side *typedSchemaSide, perturb tsPerturb, ) []Violation
NodeBattery drives the whole-node boundary: the mid-build rejection, the finalised acceptance, the unlabelled and never-interned controls, and the pre-installation fixture whose stored value the schema forbids.
Every clause carries TWO expectations, and they are separate on purpose. The LITERAL expectation is the harness's own precondition — "I set the label and did not set the required property, so this must be refused" — and the MODEL expectation is [typedSchemaModel.predictNode] over the node's actual labels and properties. A perturbation that changes the graph fires the literal clause while leaving the model clause silent, which is the correct attribution: the fixture stopped being what the harness thought, rather than the hook being wrong.
and splitting them would obscure the build sequence they depend on.
five fixture clauses in a fixed order; each is three lines
func (*TypedSchemaProbes) RecoveryPin ¶ added in v0.12.0
func (p *TypedSchemaProbes) RecoveryPin( sm *Simulator, tick int64, epoch int, perturb tsPerturb, ) []Violation
RecoveryPin is the constructed pin for the recovery asymmetry the acceptance criteria require, and it yields FIVE clauses from one probe:
- pin:no-validator — a write the LIVE validator refused is ACCEPTED on the freshly recovered graph. This is the documented limitation, asserted positively rather than only written down: the schema is not among the snapshot's components, so nothing re-installs it.
- pin:node-clean — lpg.Graph.ValidateNode reports clean on the same graph, so the WHOLE-NODE hook is absent too, not merely the per-value one.
- pin:reinstalled — with a fresh schema installed, the IDENTICAL write is refused. Without this clause, clause 1 could pass because the write was never forbidden at all.
- pin:validate-detected — ValidateNode now reports the planted value as a type mismatch, because the value planted while the graph was unvalidated is still stored. This is the kind re-check branch that only the whole-node hook reaches.
- pin:validate-repaired — once the value is repaired, ValidateNode reports clean, so clause 4 measured the value rather than a permanently-refusing node.
The plant is a DIRECT lpg write, so it never reaches the WAL and cannot contaminate the durable image; the repair happens before this function returns, so no checkpoint can capture it either.
func (*TypedSchemaProbes) SideWrite ¶ added in v0.12.0
func (p *TypedSchemaProbes) SideWrite( tick int64, side *typedSchemaSide, spec tsOpSpec, perturb tsPerturb, ) []Violation
SideWrite drives one side-battery write and adjudicates it: the verdict against the model, then either the accept read-back or the full no-mutation battery.
perturb is a TEST-ONLY parameter (see [tsPerturb]); the scenario always passes [tsPerturbNone].
pre/post snapshot inline; splitting it would hide the ordering the clauses depend on.
one linear pass over five paths and three verdicts, with the
func (*TypedSchemaProbes) VerifyWitnesses ¶ added in v0.12.0
func (p *TypedSchemaProbes) VerifyWitnesses( ctx context.Context, sm *Simulator, tick int64, phase string, afterRecovery bool, perturb tsPerturb, ) []Violation
VerifyWitnesses reads every armed witness through BOTH channels and requires the ACCEPTED value and only the accepted value.
afterRecovery distinguishes the reads that matter for the coverage claim (a graph that came back through real recovery) from the terminal read on the live graph, which is a cheaper regression check on the same invariant.
type TypedWriter ¶ added in v0.6.0
type TypedWriter struct {
// contains filtered or unexported fields
}
TypedWriter is the type-coverage actor: it emits [tmplCreateTyped] CREATEs whose parameters span every round-tripping Cypher property kind — string, integer, float, boolean, list, a plain ISO-8601 string, and all six genuine temporal types — with values drawn deterministically from the seed. The oracle records the full property set so the type-coverage checker can verify each kind survives commit and crash/recovery.
Concurrency contract ¶
TypedWriter is NOT safe for concurrent use; it is invoked from the single simulation goroutine.
func (*TypedWriter) Name ¶ added in v0.6.0
func (*TypedWriter) Name() string
Name returns the actor's identifier.
func (*TypedWriter) NextOp ¶ added in v0.6.0
func (w *TypedWriter) NextOp(seed *Seed, _ *GraphOracle) Op
NextOp returns the next Typed-node CREATE with a unique id and a value of every supported kind, all seed-derived so the op stream is a pure function of the seed. A pointer receiver carries the monotone id counter across calls.
The six temporal properties are bound as genuine temporal expr values, so the engine stores them in its kind-tagged form and they read back as temporals; `ts` stays a plain ISO-8601 STRING and is the control that must read back as a string (rmp #2457).
type UpgradeConfig ¶
type UpgradeConfig struct {
// Workload is the actor mix used for the write phase. When nil,
// [WriteHeavyWorkload] is used so the durable image carries real structure.
Workload func(*Seed) *Workload
// IndexSpecs, when non-empty, are created before the write phase and
// cross-checked for consistency after reopen, so the upgrade also guards
// index durability across the boundary.
IndexSpecs []IndexSpec
// Seed drives the deterministic write workload and the SimDisk sub-seed.
Seed uint64
// Ops is the number of write operations applied before the upgrade boundary.
// Values <= 0 default to 400 — enough to populate nodes, edges, and a few
// deletes so the parity check is meaningful.
Ops int
}
UpgradeConfig parameterises an upgrade simulation: the seed driving the write workload, how many write ops to apply before the simulated upgrade boundary, and the workload factory. The write phase is fully deterministic from Seed so a failure reproduces exactly.
type UpgradeResult ¶
type UpgradeResult struct {
// Report is non-nil when the post-reopen parity/durability/index check found
// a violation (data loss, ghost state, or index drift).
Report *SimReport
// WrittenNodes / WrittenEdges are the oracle-modelled counts at close.
WrittenNodes int
WrittenEdges int
// RecoveredNodes / RecoveredEdges are the engine counts after the reopen.
RecoveredNodes int64
RecoveredEdges int64
// ReplayedWALOps is how many committed WAL ops recovery replayed on reopen.
ReplayedWALOps int
}
UpgradeResult summarises an upgrade simulation: the durable counts written, the counts recovered after the reopen, and how many WAL ops the reopen's recovery replayed. A nil UpgradeResult.Report means full parity held.
func RunUpgrade ¶
func RunUpgrade(ctx context.Context, cfg UpgradeConfig) (UpgradeResult, error)
RunUpgrade performs the PRIMARY upgrade simulation: it writes a deterministic workload through a real WAL-backed SimStore on a SimDisk image, closes the store gracefully (every ACKed commit durable), then reopens the SAME durable image through the real recovery path (recovery.ReplayWAL, via OpenSimStore) — the cross-version boundary — and runs the full oracle parity, durability, and (optional) index-consistency check against the recovered engine.
This guards the class of data-compatibility regressions the project has hit before (e.g. the v0.2.0->v0.3.x adjlist recovery panic): if the current recovery code cannot faithfully rebuild the graph from a durable image, the parity check reports the divergence rather than letting it pass silently.
The returned error is a harness failure (store open/close or a write-phase engine error that should not happen on an honest workload); an invariant divergence is carried in the result's Report, not the error.
func (UpgradeResult) Parity ¶
func (r UpgradeResult) Parity() bool
Parity reports whether the upgrade reopened to full parity (no violation).
type Violation ¶
type Violation struct {
Kind ViolationKind
Message string
Op string
Tick int64
}
Violation is a single detected invariant breach, tagged with its kind, a human-readable message, the tick at which it was found, and the operation that immediately preceded it.
func CheckAccessPathParity ¶ added in v0.12.0
func CheckAccessPathParity(tick int64, _ *GraphOracle, engine PlanEngine, probes ...ParityProbe) []Violation
CheckAccessPathParity runs every probe's literal and parameterised arm through engine and returns a violation for each parity breach found:
- a result-multiset divergence between the arms is a ViolationACIDConsistency (the engine answered the same predicate two different ways);
- an access-path divergence (one arm seeks, the other scans) with equal results is a ViolationOracleDeviation — the rmp #2414 class, invisible to every result-only oracle;
- a MustSeek probe whose LITERAL arm does not seek is a ViolationOracleDeviation, because the parity assertion would otherwise hold vacuously on two scans;
- one Profile probe (the first MustSeek probe's literal arm) must report a non-zero db-hit total, so a silently un-instrumented or data-blind profile surfaces instead of passing. A total of zero is a ViolationOracleDeviation when every cell was a figure, and a ViolationVacuousRun when no cell was: since rmp #2760 an operator whose accesses nobody counted renders "?" rather than 0, and a probe that saw only "?" has no oracle rather than a refuted one.
Violation messages name the shape, both plans, and both result summaries; result ids are sorted before rendering so messages are deterministic. The oracle argument is unused (the engine is cross-checked against itself, like CheckIndexConsistency) and kept for signature uniformity.
func CheckCartesianNotification ¶ added in v0.12.0
func CheckCartesianNotification(tick int64, engine *EngineAdapter) []Violation
CheckCartesianNotification asserts the engine's Cartesian-product advisory fires for exactly the queries that build a cross product between disconnected patterns, and stays silent for the ones that do not. See [cartesianNotificationProbes] for the four shapes and why both directions are required.
The check is read-only and population-independent: it inspects the plan-time advisory, which is a function of the query text alone, so it is meaningful even on an empty graph and can never perturb a run. A clean pass returns nil.
func CheckCounterDeclShape ¶ added in v0.12.0
func CheckCounterDeclShape(d ScenarioCounterDecl) []Violation
CheckCounterDeclShape is the SEPARATE, shape-only non-vacuity gate over a declaration. It reads no run: it asks only whether the declaration could ever have failed. A declaration that names nothing, that claims blindness AND names counters, that sets a floor of zero, or that admits a shared counter with no discriminator, is satisfied by any run at all — the exact failure mode rmp #2471 found when a coalescing metric turned out to be reachable by an unrelated path.
It is kept apart from ScenarioCounterDecl.Check for the reason rmp #2470 established: an uninformative declaration must not read as a failing fault.
func CheckCypherSurface ¶ added in v0.6.0
func CheckCypherSurface(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
CheckCypherSurface runs a battery of diverse read queries — aggregation (count/sum), WHERE and WITH...WHERE filters, a pattern-count, UNWIND over range(), OPTIONAL MATCH, and ORDER BY — and asserts each result matches an invariant computed independently from the oracle model. It broadens the DST's coverage of the Cypher read surface beyond the minimal per-tick parity probe, comparing result INVARIANTS (scalar values, the sorted-name sequence) rather than plan-specific row order where it is not determined.
func CheckCypherSurfaceEntity ¶ added in v0.12.0
func CheckCypherSurfaceEntity(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
CheckCypherSurfaceEntity runs the entity-valued function battery over the Person/KNOWS graph, each probe referenced against the oracle model (rmp #2458):
- labels(n) per Person vs the oracle's modelled label set — which also gives SET n:Label / REMOVE n:Label an independent read-back of WHICH labels a node carries, beyond a count;
- properties(n) per Person vs the oracle's modelled property map, compared as a key SET plus a per-key canonical value;
- the MAP PROJECTION over a real node — n{.name, .age}, n{.*}, a selector naming a key the node lacks, and a literal entry — vs the same modelled map, which is the entity-property-resolution path the literal-map probe in the expression battery cannot reach (rmp #2459);
- type(r), startNode(r), endNode(r) and r.eid over every KNOWS edge vs the oracle's modelled endpoints and instance id — an edge whose endpoints the engine transposed reads back transposed. Read three ways: FORWARD, in REVERSE, and UNDIRECTED, because the reverse and undirected reads are the only ones that constrain the direction rmp #2504 lived in;
- the same edge projection driven through BOTH consumers of a pattern comprehension — hoisted by WITH into a RollUpApply, and left to the expression-level fallback by UNWIND — each compared against the same oracle reference, which is what keeps rmp #2505 from returning;
- path materialisation on `MATCH p=(a:Person)-[:KNOWS*1..3]->(b)`: size(nodes(p)) = length(p)+1 and size(relationships(p)) = length(p) per row, the path's first node identical to the anchor and its last node identical to the matched endpoint, plus the per-anchor path-length histogram vs an oracle trail enumeration;
- elementId(n) / elementId(r): stability across two reads, distinctness across entities, and the engine's actual contract (see [checkEntityElementID]).
It runs on the quiescent graph, periodically, after each crash/recovery, and at the end, exactly as the rest of the surface battery does.
func CheckCypherSurfaceExtended ¶ added in v0.8.0
func CheckCypherSurfaceExtended(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
CheckCypherSurfaceExtended broadens the Cypher-surface battery with the read-clause, expression and procedure shapes the earlier DST did not drive, each verified against an invariant computed independently from the oracle model over the Person/KNOWS graph: DISTINCT and count(DISTINCT) (CY5); 3VL boolean AND, list membership IN, IS NULL and <> (CY6); the STARTS WITH / ENDS WITH / CONTAINS / =~ string predicates (CY7); ORDER BY … SKIP … LIMIT pagination (CY8); the min/max/sum aggregates plus the EXACT avg/stDev/stDevP/percentileCont/percentileDisc oracle-arithmetic probes (CY10, rmp #2452 — see [expectedAggregates] for the pinned definitions); EXISTS { } / COUNT { } / pattern-comprehension subqueries (CY13); and the db.* schema introspection procedures (CY16). It runs on the quiescent graph, including after crash/recovery.
func CheckCypherSurfaceGrouped ¶ added in v0.12.0
func CheckCypherSurfaceGrouped(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
CheckCypherSurfaceGrouped runs the grouped-aggregation, DISTINCT-row, UNION and collect probes of the cypher-surface battery, each asserted as full row-set equality against a reference computed independently from the oracle model (rmp #2452):
- grouped count(*) and grouped sum(n.age) BY n.city vs the oracle's per-city histogram ([GraphOracle.personCityStats]), ordered by city;
- a mixed-type grouping key (a CASE yielding FLOAT for some rows and INTEGER for others) vs the oracle's per-age histogram, exercising the exact INTEGER↔FLOAT grouping equivalence the engine pins end-to-end in cypher/intfloat_exact_equality_test.go (rmp #2050): equal int and float keys must land in ONE group, and 2^53-scale keys must not collapse;
- RETURN DISTINCT b.name over a pattern vs the oracle's distinct-target set, plus a WITH DISTINCT mid-pipeline stage feeding a count;
- UNION over graph rows (complementary and OVERLAPPING predicates — the overlap makes the dedup observable) vs the full sorted name set, and UNION ALL of overlapping predicates vs the SUM of the arm cardinalities (duplicate preservation);
- collect(n.name), sorted, vs the oracle's sorted name slice.
Probes that read n.age or n.city are skipped while the model holds a Person without that property (impossible in the surface workload, which always binds both); the DISTINCT and collect probes run unconditionally. It runs on the quiescent graph, periodically, after each crash/recovery, and at the end.
func CheckCypherSurfaceOrdering ¶ added in v0.12.0
func CheckCypherSurfaceOrdering(tick int64, oracle *GraphOracle, engine *EngineAdapter, st *OrderingStats) []Violation
CheckCypherSurfaceOrdering runs the ordering, pagination and multi-part query-structure probes of the cypher-surface battery, each asserted as full row-SEQUENCE equality against a reference computed independently from the oracle model (rmp #2460):
- ORDER BY n.age DESC, n.name ASC — descending primary key with an ascending tie-break, over a workload whose ages collide heavily;
- ORDER BY n.name DESC — the reverse of the battery's ascending probe;
- ORDER BY n.age % 10, n.name — an ordering on an EXPRESSION, whose projected key value is asserted alongside the sequence;
- RETURN n.city, count(*) AS c ORDER BY c DESC, n.city ASC — an ordering on an AGGREGATE;
- Top-vs-Sort equivalence: the same ordering with LIMIT k must return exactly the first k rows of the unlimited ordering, and the two arms must resolve through DIFFERENT physical operators (Top vs Sort) — see [checkTopFusion];
- pagination: the same page written with LITERAL and with PARAMETERISED SKIP/LIMIT, LIMIT 0, a SKIP past the end, and SKIP/LIMIT over a DESC ordering — see [checkOrderingPagination];
- multi-part structure: a two-stage WITH pipeline, "top-k then expand", and WITH … WHERE on an aggregated value — see [checkOrderingMultiPart].
It runs on the quiescent graph, periodically, after each crash/recovery, and at the end. st accumulates the run-level observations the terminal non-vacuity gate reads and may be nil.
The probes are skipped while the model holds a Person without both a string name and an integer age, or two Persons sharing a name: the expected row sequence would then not be a total order, and a probe whose reference is ambiguous can only produce noise. The surface workload binds both properties and issues distinct names, so this is a guard, not an expectation.
func CheckDDLCounters ¶ added in v0.14.2
func CheckDDLCounters(tick int64, query string, before, after ddlSchemaNames, got *exec.QueryCounters) []Violation
CheckDDLCounters compares the engine-reported schema-effect counters of one executed DDL statement against the effect its own registries show it applied, and returns a ViolationOracleDeviation per disagreement.
before and after are [readDDLSchemaNames] snapshots taken immediately either side of the statement. got is the statement's counters, read from the drained cypher.Result; nil is the engine's report for a statement that recorded no schema effect and is treated as an all-zero effect set here, because whether a no-op DDL reports nil or a zero counter set is a distinction the module has not pinned — what IS pinned, and what this checks, is the VALUE of each effect.
The rules:
- each of the four schema counters must equal the corresponding name-set difference: an absorbed IF EXISTS loses nothing and must report nothing, and an applied DROP loses one name and must report one removal;
- every data counter must be zero, so a schema statement cannot leak a node, relationship, property or label effect;
- a statement that changed the schema must report NON-nil counters, or the effect went unreported altogether.
query is carried only for the message; nothing is parsed out of it.
func CheckEdgeProperties ¶ added in v0.6.0
func CheckEdgeProperties(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
CheckEdgeProperties reads every modelled KNOWS edge's properties back through the real engine read path and asserts each round-trips to its modelled value (canonical type-aware compare) and that the edge still exists — pinning the probe to one instance with `WHERE r.eid` when the edge carries a non-zero eid, so a parallel twin can never answer for its sibling. It then walks the deleted-instance tombstones (GraphOracle.DeletedKnowsInstances) asserting each deleted instance is ABSENT while BOTH its endpoint nodes are still alive (DELETE r must never cascade to a node). Run on a quiescent graph including immediately after crash/recovery, so edge properties, per-instance deletions, and surviving twins are all validated against a WAL-recovered (and columnar-tier-rebuilt) graph.
func CheckExprLiterals ¶ added in v0.8.0
func CheckExprLiterals(tick int64, engine *EngineAdapter) []Violation
CheckExprLiterals runs the graph-independent expression battery and asserts each query's canonical result multiset equals its known-constant expectation. It needs no oracle (the answers are pure functions of the query text) and is run inside the cypher-surface scenario — periodically, after each crash/recovery (proving the recovered engine still evaluates every expression shape), and at the end.
func CheckGraphIOGuardDeclShape ¶ added in v0.12.0
func CheckGraphIOGuardDeclShape(decls []GraphIOGuardDecl) []Violation
CheckGraphIOGuardDeclShape is the SEPARATE shape-only non-vacuity gate for the cap declarations. It reads no run: it asks only whether the declarations could ever have failed. A declaration missing its sentinel, or carrying neither a heap bound nor a reason for being unreachable, makes the verdict above meaningless however the run went.
func CheckGraphIOGuards ¶ added in v0.12.0
func CheckGraphIOGuards(r *GraphIOGuardResult) []Violation
CheckGraphIOGuards is the unconditional verdict over one RunGraphIOGuards result: every cap declared reachable was reached with its own sentinel and within its heap bound, the one declared unreachable still is, and every *Ctx reader stopped mid-parse with a typed cancellation and no partial graph.
one flat adjudication per declaration and per arm
func CheckGraphIOSurface ¶ added in v0.12.0
func CheckGraphIOSurface(r *GraphIOSurfaceResult) []Violation
CheckGraphIOSurface is the unconditional verdict over one RunGraphIOSurface result. It runs whatever the run produced and never consults the non-vacuity gate.
one flat adjudication per arm; splitting it would hide the list
func CheckGraphIOSurfaceShape ¶ added in v0.12.0
func CheckGraphIOSurfaceShape(r *GraphIOSurfaceResult) []Violation
CheckGraphIOSurfaceShape is the SEPARATE non-vacuity gate. It asks only whether the run could ever have failed — whether each format arm actually ran, whether the model drove the branches the verdict claims to adjudicate, and whether the mutations were semantically effective. It never decides correctness, so a violation here means the run proved too little, not that the module is wrong.
one flat gate per arm; splitting it would hide the list
func CheckIndexConsistency ¶
func CheckIndexConsistency(tick int64, _ *GraphOracle, engine indexConsistencyEngine, specs ...IndexSpec) []Violation
CheckIndexConsistency performs a THOROUGH (not sampled) index-vs-base-data consistency check for every declared index in specs. For each index on (Label, Property) it:
- full-scans the base data via the engine (MATCH (n:Label) RETURN id(n), n.Property) and builds the authoritative value -> {node id} map directly from the nodes that carry the property;
- for every distinct indexed value, runs the index-seek probe (MATCH (n:Label {Property:$v}) RETURN id(n)) — which the engine resolves through the index when one is present — and asserts the seek returns EXACTLY the node ids the full scan attributed to that value.
A value the seek over-reports (a node id the full scan does not carry) is a torn/orphaned index entry; a value the seek under-reports (a node id the full scan carries but the seek misses) is a stale/lost index entry. Either is a ViolationACIDConsistency (the index disagrees with the base data it indexes). The check is bounded but exhaustive over the CURRENT graph: it walks every node of each indexed label exactly once for the scan and issues one seek per distinct value.
The check probes the engine through the same execution path the workload uses, so it observes whatever the engine would serve a real query — which is precisely the property an index must preserve. It cross-checks the engine against itself (seek path vs scan path) rather than against the oracle, because an index covers DDL-created labels the minimal Phase-1 oracle does not model.
func CheckMergeHandleCollision ¶ added in v0.12.0
func CheckMergeHandleCollision(tick int64, f *mergeHandleFixture, g *lpg.Graph[string, float64], oracle *GraphOracle, engine *EngineAdapter, ) []Violation
CheckMergeHandleCollision is the rmp #2515 detector. It asserts three things, and needs all three: the PRECONDITION still holds, the relationship carries what the model says it carries, and NO Person carries the key at all.
Precondition. The fixture relationship's stable handle must still equal the decoy's node id, and the node with that id must still be the decoy. A regression that moved either would leave the family driving a shape on which a misdirected write has nowhere wrong to land — so it would pass for the wrong reason. Re-running this after every crash is what proves the collision survives recovery rather than assuming handles and node ids are both durable.
Present-direction. The relationship must carry the value the model wrote. This overlaps CheckMergePairRelProps's walk deliberately: that checker reports a LOST whole-entity write (rmp #2510), and this one reports a MISDIRECTED per-property write, which is a different defect with the same first symptom.
Absent-direction. No Person may carry [mergePairRelKey] — no workload template ever writes it to a node — so a Person that has it acquired it by the misdirection. This is the half neither the counters oracle nor the relationship read-back can see: the statement reports `+properties = 1` whether the write landed on the relationship or on the decoy, and a lost write leaves the relationship null with no node holding the value, whereas a misdirected one leaves the relationship null AND the decoy holding it.
func CheckMergePairRelProps ¶ added in v0.12.0
func CheckMergePairRelProps(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
CheckMergePairRelProps reads every modelled PAIRED relationship's property back through the real engine and asserts it matches the model in BOTH directions: the value the whole-entity ON CREATE action wrote where the model carries one, and NULL where it does not.
Both directions are load-bearing. The present-direction is the rmp #2510 regression detector — a lost whole-entity relationship write reads back null. The absent-direction is what stops the checker from degenerating into a tautology satisfied by an engine that writes the property onto every edge: the plain [tmplMergePairPattern] family creates PAIRED edges with NO properties in the same key namespace, so a spurious write is caught too.
Each ordered (name_a, name_b) pair carries at most one PAIRED edge — a second MERGE of the same pair matches rather than creating — so the by-name read is unambiguous even though the family deliberately leaves duplicate Person nodes behind. Running it after crash/recovery is what proves the relationship write is DURABLE and not merely present in memory.
func CheckMergeRel ¶ added in v0.8.0
func CheckMergeRel(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
CheckMergeRel reads every modelled KNOWS edge's r.n counter back through the real engine and asserts it equals the modelled hit count. Running it on the quiescent graph, including immediately after crash/recovery, verifies the MERGE-relationship idempotency (edge-count parity is covered by the shared durability check) and that ON CREATE/ON MATCH SET counter updates round-trip and survive WAL + snapshot recovery.
func CheckMergeZeroDriverAbsent ¶ added in v0.12.0
func CheckMergeZeroDriverAbsent(tick int64, engine *EngineAdapter) []Violation
CheckMergeZeroDriverAbsent asserts that the zero-row-driver family (rmp #2512) created nothing: no Person carries a [mergeZeroKeys] name, and no PAIRED edge joins two of them. It is the read-back half of the guard — the counters oracle catches the phantom effect REPORT on the tick it happens, this catches the phantom STATE, and running it after crash/recovery is what distinguishes a phantom that was merely created from one that was durably persisted.
It also verifies the family's PREMISE rather than assuming it: [mergeZeroAbsentName] must not exist, since a Person by that name would turn the never-matching driver into a one-row driver and make the all-zero expectation wrong. The namespace is disjoint from every name the workload can bind, so this can only fail if that disjointness is ever broken — which is exactly when a silent pass would begin.
func CheckNonDeterministicFuncs ¶ added in v0.12.0
func CheckNonDeterministicFuncs(tick int64, engine *EngineAdapter) []Violation
CheckNonDeterministicFuncs asserts the honest invariants of the three functions whose results are NOT a known constant (rmp #2458). Each is stated as a property the correct implementation must satisfy on every run, never as a captured value:
- rand() — every draw is a FLOAT in [0, 1), and eight draws in one statement are not all identical (a constant generator fires);
- randomUUID() — every value matches the RFC 4122 version-4 textual shape ([reUUIDv4]) and four draws in one statement are pairwise distinct;
- timestamp() — every call WITHIN one statement returns the same integer (the engine freezes the statement clock in cypher/stmt_now_reg.go, which overrides timestamp() alongside the five temporal `now` constructors), the value is epoch MILLISECONDS rather than seconds, and it is non-decreasing ACROSS two successive statements.
func CheckNullSemantics ¶ added in v0.12.0
func CheckNullSemantics(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
CheckNullSemantics runs the NULL / 3VL probe battery against references computed independently from the oracle's age-present vs age-absent partition (rmp #2453). Every scalar probe is exact; the OPTIONAL MATCH probe is full row-set equality including the NULL markers; and the 3VL partition identity is asserted over the ENGINE-returned counts, so the identity itself is observable rather than implied. It runs on the quiescent graph, periodically, after each crash/recovery, and at the end.
func CheckOpCounters ¶ added in v0.12.0
func CheckOpCounters(tick int64, op Op, committed bool, got *exec.QueryCounters, oracle *GraphOracle) []Violation
CheckOpCounters compares the engine-reported per-statement write-effect counters of one executed operation against the effect the GraphOracle predicts for it, and returns a ViolationOracleDeviation per disagreement. It must be called AFTER the op executed (got is read from the drained result) and BEFORE [Simulator.applyToOracle] advances the oracle, because the expected effect is a function of the pre-op model (e.g. whether a MERGE matches, or how many edges a DETACH DELETE takes with it).
The rules, in order:
- An op the engine did not commit is skipped. When the statement failed before producing a result there are no counters at all; when it produced a result whose drain failed, the statement was rolled back and the engine contract deliberately does not pin that result's own counters (rollback restores the graph, not the report — what IS pinned, by cypher's TestQueryCounters_RolledBackStatementReportsNothing, is that no effect leaks into the NEXT statement's counters). The oracle stays frozen for the op, so state parity is still enforced by the per-tick checker.
- A pure read (OpMatch) must report NIL counters — the engine contract distinguishes "no write surface" (nil) from "wrote nothing" (all-zero), so a read that reports counters is a deviation.
- A committed write of a modelled template must report NON-nil counters that equal the derived expectation on every one of the twelve fields. A committed write of an unmodelled template is skipped.
The check is a pure function of its arguments: it draws no randomness, issues no engine query, and its per-op cost is O(1) except for DETACH DELETE, whose incident-edge count is the same O(edges) scan the oracle's own ApplyDelete already pays on the same tick.
func CheckPatternShapes ¶ added in v0.12.0
func CheckPatternShapes(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
CheckPatternShapes runs the multi-hop / undirected / reverse / multi-type / cyclic pattern battery — the join-and-intersect planner family (ExpandInto, ExpandIntersect, hash join, reverse expansion, relationship uniqueness) — and asserts each count(*) equals a reference computed independently from the oracle's adjacency by [computePatternShapeRefs]. The ExpandInto shape (both endpoints bound by property) is probed for a bounded, deterministic selection of existing KNOWS pairs, in BOTH literal and $param form.
func CheckPlanStability ¶ added in v0.12.0
func CheckPlanStability(tick int64, base *PlanBaseline, engine PlanEngine) []Violation
CheckPlanStability re-renders every baseline probe through engine and returns a ViolationOracleDeviation for each rendering whose plan SHAPE is not identical to its captured baseline. It is meant to run after each crash/recovery (the plan cache was rebuilt from scratch) and at the end of a scenario; probes are compared in capture order so messages are deterministic.
The comparison is over [planShape], not over the raw rendering: the cardinality estimates the physical plan carries since rmp #2765 move with the live data these scenarios churn, and a changed estimate is not a changed plan.
func CheckSchemaIntrospection ¶ added in v0.12.0
func CheckSchemaIntrospection(tick int64, model *SchemaModel, engine *EngineAdapter) []Violation
CheckSchemaIntrospection asserts the engine's schema-introspection surfaces against the harness's DDL model:
- SHOW INDEXES and SHOW CONSTRAINTS row sets equal the model's expected rows (full column shape);
- db.indexes() and db.constraints() row sets equal the model too, and the (name, type) enumeration agrees between SHOW and the procedures — two independent surfaces must match each other and the model;
- one SHOW ... YIELD ... WHERE ... RETURN projection per statement kind reproduces the model-side filter (#2044), one of them through a YIELD alias;
- db.schema.visualization() executes and drains without error.
A model mismatch is a ViolationOracleDeviation; a disagreement between the two engine surfaces is a ViolationGraphIntegrity (the engine is internally inconsistent). A clean pass returns nil.
func CheckSchemaMutation ¶ added in v0.8.0
func CheckSchemaMutation(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
CheckSchemaMutation reads every modelled Person back through the real engine and asserts each scalar property equals its modelled value (a removed/never-set property reads NULL — CY1) and that the Vip label membership matches the model, probed via the multi-label pattern (n:Person:Vip {name}) (CY14). Running it on the quiescent graph, including immediately after crash/recovery, verifies every REMOVE / SET-label / SET-map mutation round-trips and survives WAL + snapshot recovery. It reads the shared oracle's own NodeState, so it needs no second model.
It also runs CheckMergePairRelProps, so the whole-entity relationship write of the MERGE-surface family (rmp #2510) is verified at every one of this check's call sites — periodically, immediately after every crash/recovery, and once at the end — rather than needing its own schedule. The PAIRED endpoints are deliberately absent from the name index this function's Person loop walks, so the two probes cover disjoint state.
func CheckSearch ¶ added in v0.6.0
func CheckSearch(tick int64, oracle *GraphOracle, engine Engine) []Violation
CheckSearch runs the search-algorithm battery against the current graph and returns any violations it finds (a clean check returns nil). It performs two independent families of check:
- Structural parity between the engine graph — extracted via the public Cypher read path, the same path the workload uses — and the oracle's shadow model: a full node-set and edge-set equality. This is strictly stronger than the base checker's count-plus-sample probes, and proving it lets the algorithm checks run on the model as a faithful stand-in for the engine's contents.
- Algorithm correctness: each search/ algorithm run on the oracle graph is compared to an independent naive reference on an invariant of the answer (reachable set, partition up to relabelling) rather than a non-unique witness.
- The context-cancellation contract of every public context-accepting entry point in the five search families ([searchCtxCancelViolations]). This family reads no graph: it derives its own fixtures from the tick, because what it adjudicates is whether a cancelled context is honoured, whether the cancellable form computes what the plain form computes, and whether cancellation outranks a terminal sentinel such as search.ErrCycle.
CheckSearch must be called from the single simulation goroutine: it issues read queries that require a consistent, quiescent view of the engine, which the deterministic tick loop guarantees. Running it under the concurrent modes would race the writers and is not supported.
func CheckShortestPath ¶ added in v0.6.0
func CheckShortestPath(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
CheckShortestPath validates the Cypher-level shortestPath() operator against an INDEPENDENT breadth-first shortest-path computed directly from the oracle's KNOWS edge set. For a deterministic, bounded set of (a,b) Person pairs it runs `MATCH p=shortestPath((a)-[:KNOWS*]->(b)) RETURN length(p)` and asserts the engine's hop count equals the BFS distance — or that both agree there is no path. It compares the path-LENGTH invariant, never a specific witness path (shortest paths are not unique). It runs on a quiescent graph, including immediately after crash/recovery, so the operator is validated against a WAL-recovered graph too.
func CheckTypedListPredicates ¶ added in v0.12.0
func CheckTypedListPredicates(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
CheckTypedListPredicates drives the LIST-valued property surface over the `lst` property every Typed node carries, with every expectation computed by the oracle from the list it modelled (rmp #2459).
Before #2459 the DST wrote a list-valued property and only ever read it back WHOLE: no predicate ever indexed, sized, sliced, unwound, reduced, or tested membership in a stored list, so the engine's whole list-expression surface over stored data was unexercised. The probes here close that gap:
- [checkTypedListRow] — subscript, negative subscript, both out-of-range directions, size(), slice, and reduce(), one query per sampled node;
- [checkTypedListUnwind] — UNWIND over the STORED list, aggregated by count(), sum() and collect();
- [checkTypedListMembership] — `WHERE <elem> IN n.lst` over the whole graph, once with an element the model holds and once with one it cannot.
It runs on a quiescent graph — periodically and immediately after each crash/recovery — so the predicates are also proven against a list that survived real recovery.
func CheckTypedProperties ¶ added in v0.6.0
func CheckTypedProperties(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
CheckTypedProperties reads every modelled Typed node back through the real engine read path and asserts, for every property, BOTH that its value round-trips to the modelled value (compared via the canonical expr.Value String() rendering) AND that it reads back with the expected expr KIND ([typedExpectKind]) — that a never-set property reads NULL, and that the node exists at all. It runs on a quiescent graph (the deterministic loop, including immediately after crash/recovery), so a divergence means a property failed to round-trip, changed type, or did not survive recovery.
The kind assertion is what makes the temporal arm honest (rmp #2457). Before it, temporals were written as ISO-8601 STRINGS and compared as strings, so the arm could not distinguish a working temporal round-trip from a broken one: both sides said the same text. A temporal that degrades to an untagged PropString now fails on kind even though its text is unchanged.
func CheckTypedTemporalOrder ¶ added in v0.12.0
func CheckTypedTemporalOrder(tick int64, oracle *GraphOracle, engine *EngineAdapter) []Violation
CheckTypedTemporalOrder asserts that ORDER BY over a TEMPORAL property agrees with an ordering the oracle computes itself from the temporals it modelled. The engine is asked for the Typed ids ordered by (n.d, n.id); the oracle sorts its own modelled (expr.DateValue, id) pairs and the two id sequences must be identical.
This is a genuine oracle, not a self-check: the dates are laid out with a stride ([typedDateDay]) that makes date order differ from id order, so an engine that ignored the temporal ordering — or compared the tagged storage strings byte-wise — would produce a different sequence. Fewer than two modelled nodes cannot order anything, so the check reports nothing.
func CheckVarlenPaths ¶ added in v0.12.0
func CheckVarlenPaths(tick int64, oracle *GraphOracle, engine *EngineAdapter, st *vleStats) []Violation
CheckVarlenPaths runs the variable-length path battery — exact depth, zero length, lower-bound-only, bounded, intermediate-node predicates, multi-type and undirected expansion, and the path functions over VLE rows — and asserts every result equals a reference enumerated independently from the oracle's adjacency by [buildVLEModel] and [enumerateTrails]. It only reads from the engine.
st accumulates what actually fired; pass the same value across a run and hand it to [varlenPathsVacuity] at the end.
type ViolationKind ¶
type ViolationKind string
ViolationKind classifies an invariant breach. The ACID_* kinds map to the module's four transactional guarantees; GRAPH_INTEGRITY covers structural invariants (e.g. an edge whose endpoints are absent); ORACLE_DEVIATION covers any disagreement between the shadow model and the engine that is not more specifically classified.
The kinds fall into TWO families, and an operator triaging a red run reads the kind to decide which of the two it is looking at:
- ENGINE-VERSUS-ORACLE: ACID_ATOMICITY, ACID_CONSISTENCY, ACID_ISOLATION, ACID_DURABILITY, GRAPH_INTEGRITY, ORACLE_DEVIATION, SEARCH_DIVERGENCE. The engine did something the model says it must not. Suspect the engine.
- RUN-NOT-EXERCISED: VACUOUS_RUN. The engine did nothing wrong because it was never driven to the point where it could. Suspect the harness, the seed, the machine, or a co-resident scenario.
Keeping them apart is not cosmetic. On 2026-08-25 thirty-seven bolt-decode-swarm failures were reported as ORACLE_DEVIATION when the cause was a co-resident cpu-starvation scenario clamping GOMAXPROCS (rmp #2613): the clause text was honest, but the KIND pointed the investigation at the engine (rmp #2614).
const ( ViolationACIDAtomicity ViolationKind = "ACID_ATOMICITY" ViolationACIDConsistency ViolationKind = "ACID_CONSISTENCY" ViolationACIDIsolation ViolationKind = "ACID_ISOLATION" ViolationACIDDurability ViolationKind = "ACID_DURABILITY" ViolationGraphIntegrity ViolationKind = "GRAPH_INTEGRITY" ViolationOracleDeviation ViolationKind = "ORACLE_DEVIATION" // ViolationSearchDivergence is a disagreement between a search/ algorithm and // its independent naive reference on the oracle graph (see [CheckSearch]). It // indicates a bug in the traversal/path-finding/analytics code, distinct from // an engine-vs-oracle structural divergence (which is GRAPH_INTEGRITY). ViolationSearchDivergence ViolationKind = "SEARCH_DIVERGENCE" // ViolationVacuousRun reports that a run did not exercise its subject, so the // clauses about that subject held without ever being tested. It is NOT an // engine-versus-oracle disagreement and must never be used for one. // // It is load-bearing rather than advisory: rmp #2588 records that a run whose // honest client never met live pressure must not pass as evidence, so these // clauses still FAIL the run. What changes is what the failure is CALLED — // which is what an operator uses to decide whether the engine or the harness // is the suspect. // // The distinct route rmp #2554 took elsewhere — returning []string so the TYPE // forbids promoting a coverage shortfall to a verdict — is not available here, // precisely because these clauses must keep failing. ViolationVacuousRun ViolationKind = "VACUOUS_RUN" )
Violation kinds.
type VirtualClock ¶
type VirtualClock struct {
// contains filtered or unexported fields
}
VirtualClock is the simulation's logical clock. It models the passage of time as a monotonically-increasing tick counter rather than reading the wall clock, so the simulation never observes real time and stays fully deterministic. One tick represents one unit of simulated time whose duration is fixed at construction (1 tick == 1ms by convention).
VirtualClock deliberately exposes no way to read time.Now: the entire sim package is free of wall-clock reads, which is what lets a given seed replay identically.
Concurrency contract ¶
VirtualClock is NOT safe for concurrent use. It is advanced and read from the single simulation goroutine only.
func NewVirtualClock ¶
func NewVirtualClock(tickSize time.Duration) *VirtualClock
NewVirtualClock returns a clock at tick zero whose every tick advances simulated time by tickSize. A non-positive tickSize is normalised to 1ms so VirtualClock.SimulatedTime always advances.
func (*VirtualClock) Now ¶
func (c *VirtualClock) Now() int64
Now returns the current tick count (the number of ticks elapsed since construction).
func (*VirtualClock) SimulatedTime ¶
func (c *VirtualClock) SimulatedTime() time.Duration
SimulatedTime returns the simulated elapsed time, computed as the tick count multiplied by the per-tick duration.
func (*VirtualClock) Tick ¶
func (c *VirtualClock) Tick() int64
Tick advances the clock by one tick and returns the new tick count.
type WALContiguityConfig ¶ added in v0.12.0
type WALContiguityConfig struct {
// Seed is the disk sub-stream seed.
Seed uint64
// Committers is how many goroutines commit concurrently. The non-vacuity
// gate requires at least [walContiguityMinCommitters].
Committers int
// TxPerCommitter is how many transactions each committer issues.
TxPerCommitter int
// FramesPerTx is how many frames each transaction emits. A value below 2
// makes contiguity vacuous — a one-frame transaction is contiguous by
// definition — and the non-vacuity gate rejects it.
FramesPerTx int
// PerFrameAppend selects the CONTROL arm: each frame is appended with
// [wal.Writer.Append] instead of the whole transaction with
// [wal.Writer.AppendRun]. It is the sensitivity seam — same writer, same
// workload, no run-level lock — and it is expected to interleave.
PerFrameAppend bool
}
WALContiguityConfig parameterises a contiguity run.
type WALContiguityEvidence ¶ added in v0.12.0
type WALContiguityEvidence struct {
// Committers is the concurrency the run drove; FramesPerTx the run's shape.
Committers, FramesPerTx int
// PerFrameAppend records which append path produced this image.
PerFrameAppend bool
// Alternating records that the image was produced by the DETERMINISTIC
// handoff protocol ([RunWALContiguityAlternating]) rather than by racing
// committers, so its layout is exact rather than a scheduling outcome.
Alternating bool
// Transactions is how many distinct transaction ids the image carries, and
// Frames the total frame count.
Transactions, Frames int
// Runs is the number of MAXIMAL contiguous same-transaction blocks. It equals
// Transactions exactly when every transaction is contiguous, and exceeds it
// by one for each extra fragment.
Runs int
// SplitTransactions is how many transactions occupy more than one run — the
// count that must be ZERO for AppendRun's claim to hold.
SplitTransactions int
// WorstFragments is the largest number of fragments any single transaction
// was broken into (1 when every transaction is contiguous).
WorstFragments int
// ShortTransactions is how many transactions carry a frame count other than
// FramesPerTx: a lost or duplicated frame, which would make the contiguity
// census read a different population than the one committed.
ShortTransactions int
// CommitterSwitches counts adjacent run boundaries at which the committing
// goroutine changed. It is the witness that the run genuinely interleaved:
// zero means the committers happened to serialise and the image proves
// nothing about concurrency.
CommitterSwitches int
// TailErr is the durable image's stop condition; non-nil means the census
// read a truncated image.
TailErr error
// ImageLen is the durable image's byte length, reported for evidence only —
// never asserted against a constant.
ImageLen int64
}
WALContiguityEvidence is what a contiguity run OBSERVED in the durable image.
func RunWALContiguity ¶ added in v0.12.0
func RunWALContiguity(ctx context.Context, cfg WALContiguityConfig) (WALContiguityEvidence, error)
RunWALContiguity drives cfg.Committers concurrent committers through a real wal.Writer over a SimDisk and reports how the durable image is laid out.
The transactions are tagged in their payloads, so the census partitions the image by what is actually ON DISK rather than by what any counter claims. No fault is injected: the question is purely one of ordering.
func RunWALContiguityAlternating ¶ added in v0.12.0
func RunWALContiguityAlternating(ctx context.Context, seed uint64, perFrame bool) (WALContiguityEvidence, error)
RunWALContiguityAlternating CONSTRUCTS the frame ordering instead of racing for it. Two committers hand a token back and forth, so at every instant exactly one goroutine is eligible to append and the resulting durable image is fully determined by the protocol — not by the scheduler, the core count, the coverage instrumentation, or the load of the rest of the suite.
Why this replaced a concurrent control arm (rmp #2472, after #2517) ¶
The first version of the control drove eight committers concurrently and asserted that the per-frame path produced at least one split. It did — 31 of 96 transactions on an idle machine — but under `make ci`'s coverage step, with the whole suite running in parallel, the scheduler never overlapped the committers: all four retries measured `committerSwitches=7, split=0`, and the arm reddened a green tree. An assertion on a scheduling outcome measures the MACHINE, which is the defect class rmp #2517 filed. Raising the retry count would have traded a red gate for a slow one and still measured the scheduler.
The two modes, and why the pair is the proof ¶
Both modes run the SAME handoff protocol; only the append API differs, so the difference in the durable image is attributable to the API and to nothing else.
perFrame=true — strict alternation. A takes the token, appends ONE frame, passes the token to B; B appends one frame and passes it back. Because wal.Writer.AppendCtx releases the writer mutex between appends, the partner's frame really does land in the middle, and the image is a0 b0 a1 b1 … — every transaction broken into exactly FramesPerTx fragments. This is a genuine interleaved image produced by the real writer, and it is reproducible byte for byte.
perFrame=false — the same handoff, except each committer emits its whole transaction inside one wal.Writer.AppendRun. A signals (without waiting) once it is INSIDE its run, and B then attempts its own append. B cannot get in: AppendRun holds the writer mutex across the entire run, so B blocks until A is done and the image is a0…a3 b0…b3 — two contiguous runs. The ordering here is decided by the MUTEX, not by the scheduler, so it is equally deterministic.
A one-way signal is used in the second mode on purpose: a full ping-pong would deadlock there, because A would be waiting for B while holding the very mutex B needs. That deadlock is the mechanism under test, so the protocol must not depend on it resolving.
func (WALContiguityEvidence) String ¶ added in v0.12.0
func (e WALContiguityEvidence) String() string
String renders the evidence for a failure message or a test log.
type WALGuardResult ¶ added in v0.12.0
type WALGuardResult struct {
// Skipped reports that the platform can express NOTHING here — Windows has
// no O_NOFOLLOW and privileged symlink creation — so the arm exercised no
// guard at all and the adjudicator makes no claim.
//
// It is deliberately NOT set when only SOME axis could not be exercised. A
// whole-record skip raised LATE discards every guard already measured above
// it, and because the adjudicator returns early on Skipped it then reports
// success having judged nothing. That is precisely how an unavailable LOCK
// sentinel used to throw away a live CWE-59 symlink detection (rmp #2745).
// A partial failure is recorded PER AXIS instead, below.
Skipped bool
// SkipReason explains a whole-record skip.
SkipReason string
// LockGuardsAttempted, SymlinkWALAttempted, SymlinkLockAttempted and
// VictimChecked report which axes were actually exercised. The adjudicator
// judges every attempted axis and stays silent only about the others, so one
// unavailable axis can no longer silence the rest.
//
// They are also this arm's WITNESS: a record that declares no skip and yet
// attempted nothing is itself a violation, because an oracle that observed
// nothing cannot fail and therefore proves nothing.
LockGuardsAttempted bool
SymlinkWALAttempted bool
SymlinkLockAttempted bool
VictimChecked bool
// Unmeasured names each axis that could not be exercised, and why. It is
// reported, never silently dropped: an axis nobody ran is unknown, not clean.
Unmeasured []string
// FirstOpenErr is the error of the first (expected successful) open.
FirstOpenErr error
// SecondOpenErr is the error of a second open against the SAME path while the
// first writer holds the lock, and SecondOpenIsLocked whether it is
// [wal.ErrWALLocked].
SecondOpenErr error
SecondOpenIsLocked bool
// ReopenAfterCloseErr is the error of a third open once the first writer is
// closed. It must be nil: a lock the owner cannot release is a lock that
// strands the WAL.
ReopenAfterCloseErr error
// SymlinkedWALErr is the error of opening a WAL path whose final component is
// a symlink to a victim file, and SymlinkedLockErr the same for the LOCK
// sentinel — a second, distinct O_NOFOLLOW site (lock_unix.go), since the
// lock file is opened before any WAL data is touched.
SymlinkedWALErr, SymlinkedLockErr error
// VictimIntact reports that the victim file's bytes are unchanged after both
// attempts. This is the property that actually matters: an append through the
// link would grow it and the O_TRUNC suffix temp would empty it.
VictimIntact bool
}
WALGuardResult is what the real-directory arm observed of the two process-level guards.
Why this arm leaves the simulated disk ¶
wal.ErrWALLocked comes from flock(2) on a LOCK sentinel and the symlink refusal from O_NOFOLLOW on the final path component. SimDisk is a flat in-memory key table with neither links nor advisory locks, so both guards are STRUCTURALLY unreachable through it — not merely undriven. Modelling them in SimDisk would test the model; opening a real second writer tests the syscall the guard is made of, which is the thing that has to keep working.
The consequence is stated rather than left to be inferred: these two guards are NOT covered by any seeded, crash-injecting scenario in this package, and they never will be while SimDisk has no link or lock semantics. This arm is their only representation here.
func RunWALRealFSGuards ¶ added in v0.12.0
func RunWALRealFSGuards(dir string) (WALGuardResult, error)
RunWALRealFSGuards drives wal.Open's single-writer lock and its symlink refusal against a REAL directory, which the caller owns and cleans up (a test passes t.TempDir()).
The two opens are made from THIS process. flock(2) associates the lock with the open file description rather than the process, so a second open(2) of the same sentinel conflicts with the first even in one process — which is what makes the guard testable without spawning a subprocess.
func (WALGuardResult) String ¶ added in v0.12.0
func (r WALGuardResult) String() string
String renders the result for a failure message or a test log.
type WALLifecycleResult ¶ added in v0.12.0
type WALLifecycleResult struct {
// DurableBeforeTruncate is the offset covered by fsync just before Truncate,
// and TruncateReturned what Truncate reported freeing. They must be equal:
// Truncate documents its return as the bytes in the file at truncation.
DurableBeforeTruncate, TruncateReturned int64
// TruncateErr is Truncate's error on the healthy writer.
TruncateErr error
// DurableAfterTruncate and ImageAfterTruncate are the writer's watermark and
// the file's real length after the truncate; both must be zero.
DurableAfterTruncate, ImageAfterTruncate int64
// StatsBefore / StatsAfter bracket the truncate. Truncate documents that the
// LIFETIME counters are not reset, so these must be equal.
StatsBefore, StatsAfter wal.Stats
// PostTruncateMark is the watermark of the commit issued after the truncate.
// It must equal one frame's worth of bytes: the append restarts at offset 0.
PostTruncateMark int64
// PostTruncateFrames is how many frames the re-read image carries afterwards;
// it must be exactly the one appended after the truncate.
PostTruncateFrames int
// PoisonSyncErr is what the committer whose fsync failed received, and
// PoisonSyncIsClass whether it carries [wal.ErrDurabilityFailed].
PoisonSyncErr error
PoisonSyncIsClass bool
// PoisonedReported is [wal.Writer.Poisoned] after the failure, and
// PoisonedStable whether two consecutive calls returned the identical value.
PoisonedReported error
PoisonedStable bool
// AppendAfterPoisonIsSticky / AppendRunAfterPoisonIsSticky report that the
// two append entry points returned the SAME error value Poisoned() reports.
AppendAfterPoisonIsSticky, AppendRunAfterPoisonIsSticky bool
// SyncBufferedAfterPoison is what SyncBuffered returned on the poisoned
// writer. Measured NIL; see the type doc.
SyncBufferedAfterPoison error
// SyncGroupLostMarkIsSticky reports that a SyncGroup for a watermark the
// poison DISCARDED returns the sticky error, and SyncGroupDurableMarkErr what
// a SyncGroup for an already-durable watermark returned (nil: the
// durability-first rule of rmp #2322).
SyncGroupLostMarkIsSticky bool
SyncGroupDurableMarkErr error
// SyncFailedCount is [wal.Stats.SyncFailed] after the poison: exactly one
// round failed.
SyncFailedCount uint64
// DurableAtPoison is the watermark after the poison; it must still be the
// offset acknowledged BEFORE the failure — the discarded suffix is gone.
DurableAtPoison int64
// AppendedAtPoison is [wal.Stats.Bytes] after the poison. It EXCEEDS
// DurableAtPoison, because the discarded frame was still accepted and counted
// — which is why the watermark oracle asserts durable <= appended and not
// equality.
AppendedAtPoison uint64
// TruncateOnPoisonedErr / TruncateOnPoisonedReturned / ImageAfterPoisonTruncate
// / StillPoisonedAfterTruncate pin the behaviour above; TruncateOnPoisonedIsSticky
// reports that the error is the identical sticky one.
TruncateOnPoisonedErr error
TruncateOnPoisonedIsSticky bool
TruncateOnPoisonedReturned int64
ImageAfterPoisonTruncate int64
StillPoisonedAfterTruncate error
// AppendAfterCloseIsClosed / TruncateAfterCloseIsClosed report that the
// closed check precedes the poison check, and PoisonedAfterClose that the
// sticky error is still legible on a closed writer.
AppendAfterCloseIsClosed, TruncateAfterCloseIsClosed bool
PoisonedAfterClose error
}
WALLifecycleResult is the measured contract of wal.Writer.Truncate and of a POISONED writer. Every field is an observation; [checkWALLifecycle] holds them to the contract.
The poisoned contract as MEASURED, not as assumed ¶
Three of these were surprising enough to be worth naming, and all three were read off a real run before any assertion was written. All three have since been written into the wal.Writer godoc as well (rmp #2525), so these clauses now hold the CODE to what the documentation promises callers rather than merely recording an undocumented observation:
wal.Writer.SyncBuffered on a poisoned writer returns NIL. The poison rewinds the accepted offset to the durable one, so "make everything accepted durable" is already satisfied and the durable-already fast path fires. It is correct — nothing accepted is un-durable — but it means SyncBuffered is NOT a health probe; wal.Writer.Poisoned is.
wal.Writer.Truncate on a poisoned writer returns the IDENTICAL sticky error and touches nothing (WAL v2: it rolls the active segment over, which a poisoned writer refuses). It is pinned here so a change that lets the maintenance helper act on a writer whose durability failed is caught.
After Close, Append and Truncate return wal.ErrWriterClosed rather than the sticky poison: the closed check precedes the poison check. Poisoned() still reports the sticky error.
func RunWALLifecycle ¶ added in v0.12.0
func RunWALLifecycle(ctx context.Context, seed uint64) (WALLifecycleResult, error)
RunWALLifecycle drives the whole-file wal.Writer.Truncate and then the POISONED-writer contract on a fresh writer, over a SimDisk.
The two halves use separate writers on purpose: a truncate is only meaningful on a healthy writer, and the poison is terminal, so folding them into one writer would make the truncate half depend on the order the poison was applied.
func (WALLifecycleResult) String ¶ added in v0.12.0
func (r WALLifecycleResult) String() string
String renders the result for a failure message or a test log.
type WALWatermarkEvidence ¶ added in v0.12.0
type WALWatermarkEvidence struct {
// Label names the arm in a failure message.
Label string
// Commits is how many acknowledged commits were observed.
Commits int
// Samples is one observation per commit, in commit order.
Samples []walWatermarkSample
// Exact reports whether the arm supplied exact expectations.
Exact bool
}
WALWatermarkEvidence is what a watermark run OBSERVED across its commits. It holds measurements and no verdict.
What "expected" means, precisely ¶
Two different statements are made, and they are kept apart because only one of them is available to every caller.
EXACT, when the arm chose the payloads itself: the durable offset after a commit must equal the watermark wal.Writer.AppendRun returned for that commit, and wal.Stats must equal SUM(wal.HeaderSize + len(payload)) over every frame emitted so far. Nothing is inferred; both sides are derived from bytes the harness handed the writer.
RELATIVE, when the payloads belong to the engine: the durable offset must land on a FRAME BOUNDARY of the durable image — that is, it must equal the accumulation SUM(HeaderSize + len(payload)) over some whole number of leading complete frames — and must never exceed the bytes the writer has accepted. This is the invariant wal.Writer.DurableOffset documents ("the value always lands on a frame boundary"), and it is derived from the frames that are actually on disk.
The second form deliberately asserts NO absolute size. rmp #2521 measured that the durable image is not byte-stable across runs in one process: the hidden node key "__cx_"+hex(n) is minted from a process-global counter (cypher/exec/create_node.go), so its width tracks how many nodes the process minted before. An oracle pinning a byte count would be pinning that history. Monotonicity and the frame-boundary relation are invariant under it.
func RunWALWatermarkDirect ¶ added in v0.12.0
func RunWALWatermarkDirect(ctx context.Context, seed uint64) (WALWatermarkEvidence, error)
RunWALWatermarkDirect drives a real wal.Writer over a SimDisk through a sequence of commits whose payloads the harness chooses, so every expectation is EXACT: the durable offset must equal the watermark AppendRun returned, and the lifetime counters must equal the accumulation over the emitted frames.
It also exercises wal.Writer.SyncBuffered on the final commit — the flush path the txn layer takes for a commit that appended nothing of its own — so that method is driven under the same watermark assertions as SyncGroup.
func RunWALWatermarkEngine ¶ added in v0.12.0
func RunWALWatermarkEngine(ctx context.Context, seed uint64) (WALWatermarkEvidence, error)
RunWALWatermarkEngine drives the watermark oracle against the REAL stack: a WAL-backed SimStore whose frames the engine composes, observed after every acknowledged commit.
It is the arm that proves the oracle is size-agnostic. The engine's generated node keys come from a process-global counter, so the durable image is not byte-stable across runs (rmp #2521); the relative clauses — monotonicity, the accepted-bytes ceiling, and the frame-boundary relation — hold regardless, and are exactly what a watermark defect would break.
func (WALWatermarkEvidence) FinalDurable ¶ added in v0.12.0
func (e WALWatermarkEvidence) FinalDurable() int64
FinalDurable is the durable offset the last sample observed, or 0 for an empty run.
func (WALWatermarkEvidence) String ¶ added in v0.12.0
func (e WALWatermarkEvidence) String() string
String renders the evidence for a failure message or a test log.
type WireClient ¶
type WireClient struct {
// contains filtered or unexported fields
}
WireClient speaks the REAL Bolt v5 wire protocol over a SimConn: the 20-byte version handshake, then chunked PackStream request/response messages encoded and decoded with the genuine github.com/FlavioCFOliveira/GoGraph/bolt/proto and github.com/FlavioCFOliveira/GoGraph/bolt/packstream codecs (it does NOT reimplement the wire format). It drives well-formed requests for the honest, overload, and slow-consumer actors and decodes RECORD/SUCCESS/FAILURE/IGNORED responses.
Lock-step determinism ¶
In single-connection use the client writes one request and blocks reading the server's complete terminal response (SUCCESS or FAILURE, after any RECORDs). Because exactly one logical exchange is in flight and the SimConn buffer holds it whole, the byte stream — and therefore the decoded response — is a pure function of the request, so a given seed replays the op stream and the responses identically.
Concurrency contract ¶
A WireClient is NOT safe for concurrent use; the Bolt protocol is itself single-flight per connection (one request, then its response). The concurrent harness gives each goroutine its own WireClient on its own SimConn.
func NewWireClient ¶
func NewWireClient(conn *SimConn, clk clock.Clock) *WireClient
NewWireClient wraps a SimConn with chunked reader/writer framing. clk is retained for deadline-bearing operations; conn and clk must be non-nil.
This is the constructor every DST scenario uses, and the only one whose clients answer WireClient.Conn with a non-nil SimConn.
func NewWireClientNetConn ¶ added in v0.13.0
func NewWireClientNetConn(conn net.Conn, clk clock.Clock) *WireClient
NewWireClientNetConn wraps an ARBITRARY net.Conn with the same framing, so the identical Bolt client can be driven over a real socket as well as over a SimConn.
It exists for the transport A/B of rmp #2711. Every committed Bolt workload runs over the in-memory SimListener, so no published Bolt scaling number had ever been taken over a socket, and the only real-socket comparison in hand used a different client AND a different query — a two-variable difference that cannot attribute anything. Handing this constructor a net.Conn from net.Dial, and NewWireClient a SimConn from the same server's SimListener, leaves the transport as the ONLY difference between the two arms: same encoder, same framing, same request bytes, same server.
The returned client's WireClient.Conn is nil, because there is no SimConn behind it. Use WireClient.NetConn for the transport-agnostic accessor. conn and clk must be non-nil.
Concurrency contract ¶
Identical to NewWireClient: the returned client is NOT safe for concurrent use. Bolt is single-flight per connection, so one goroutine drives one client.
func (*WireClient) AuthenticateAs ¶ added in v0.12.0
func (c *WireClient) AuthenticateAs(principal, credentials string) (any, error)
AuthenticateAs sends the credential-bearing message(s) for the ALREADY negotiated version and returns the response that carried the decision. The credentials travel on the message the version puts them on: LOGON for Bolt >= 5.1 (which authenticates separately from a credential-less HELLO), HELLO itself for <= 5.0.
func (*WireClient) Begin ¶
func (c *WireClient) Begin() (any, error)
Begin sends BEGIN with no extras and returns the response. The server then applies its own defaults: mode "w" and server.Options.DefaultTxTimeout as the transaction's total lifetime bound.
func (*WireClient) BeginExtras ¶ added in v0.12.0
func (c *WireClient) BeginExtras(extra map[string]packstream.Value) (any, error)
BeginExtras sends BEGIN carrying an ARBITRARY extras map and returns the response. It is the single primitive behind WireClient.Begin and WireClient.BeginMode, added for rmp #2485 so a scenario can drive the whole documented BEGIN extras surface — `bookmarks`, `tx_timeout`, `tx_metadata`, `mode`, `db` — instead of only the one key BeginMode reaches.
A nil extras map is passed through, as VERIFIED in the encoder ¶
The map reaches proto.Begin unchanged, nil included, because a nil map and an empty one encode to the SAME bytes — BEGIN with either is b1 11 a0. encodeBegin writes m.Extra as one PackStream value (bolt/proto/messages.go:304-309), and a typed nil map[string]packstream.Value boxed into an interface does NOT reach the encoder's `case nil` arm (bolt/packstream/value.go:73), which matches only an untyped nil interface. It reaches `case map[string]Value` (bolt/packstream/value.go:99-110), whose map header is len(x) and whose body is a range over x — 0 and no iterations for a nil map, by Go's own semantics for len and range. WireClient.Begin therefore stays byte-identical to what it sent before this method existed WITHOUT any normalisation here, and normalising would buy nothing but an allocation. [TestWireClientBeginExtras_NilExtrasNeedsNoNormalisation] measures both spellings and fails if they ever diverge.
The caller owns the map and BeginExtras does not retain it past the call.
func (*WireClient) BeginMode ¶ added in v0.12.0
func (c *WireClient) BeginMode(mode string) (any, error)
BeginMode sends BEGIN carrying the transaction access mode and returns the response. It exists so a scenario can open a READ-ONLY explicit transaction over the genuine wire — the Mode field server.Server.Transactions reports, and the branch that takes cypher's lock-free BeginReadTx path instead of BeginTx.
The wire spelling, as VERIFIED in bolt/server/session.go handleBegin ¶
The key is "mode" in the BEGIN extras and the value is a PackStream string. Only the exact string "r" selects read-only; handleBegin reads
if v, ok := m.Extra["mode"]; ok {
if modeStr, ok := v.(string); ok && modeStr == "r" { mode = "r" }
}
so every other value — "w", a misspelling, a non-string, or an absent key — leaves the default "w". A caller asking for "w" is therefore asking for the default explicitly rather than selecting a second behaviour, and an unknown mode is silently a write transaction rather than an error.
An EMPTY mode OMITS the key entirely rather than sending "mode": "", so WireClient.Begin delegating here is byte-identical on the wire to the empty extras map it sent before, not merely equivalent in the server's eventual decision.
func (*WireClient) Close ¶
func (c *WireClient) Close() error
Close closes the underlying connection.
func (*WireClient) Commit ¶
func (c *WireClient) Commit() (any, error)
Commit sends COMMIT and returns the response.
func (*WireClient) Conn ¶
func (c *WireClient) Conn() *SimConn
Conn returns the underlying SimConn, for callers that need a hard reset (CloseWithError) to model an abrupt disconnect.
It is NIL for a client built by NewWireClientNetConn, which has no SimConn behind it. Every DST scenario builds its clients with NewWireClient and so always gets a non-nil value; a caller that may hold either kind should use WireClient.NetConn instead.
func (*WireClient) Connect ¶
func (c *WireClient) Connect(ctx context.Context) error
Connect drives the full ready-to-query handshake: the wire handshake, a HELLO, and — when the negotiated version is Bolt 5.1+ (which defers authentication to a dedicated LOGON message) — a LOGON. It returns an error if any step does not produce a SUCCESS, leaving the session ready for RUN. It is the convenience path the honest, overload, and slow-consumer actors use; the BoltAbuser drives the lower-level primitives directly.
func (*WireClient) ConnectAs ¶ added in v0.12.0
ConnectAs negotiates a version if one is not negotiated yet and then authenticates with the given basic-scheme credentials, returning the response that carried the authentication decision so the caller can adjudicate it. Unlike WireClient.Connect it does NOT treat a FAILURE as an error — a refused credential is the very outcome an auth probe is measuring — so the returned error is reserved for transport failures.
A caller that needs a SPECIFIC version negotiates it first with WireClient.HandshakeOffering; ConnectAs then keeps it, because re-sending a preamble on a negotiated connection is not a second handshake — it is 20 bytes of garbage arriving where the server expects a chunked message.
func (*WireClient) Discard ¶ added in v0.12.0
Discard sends DISCARD {n:n, qid:-1} and reads to the terminal reply, returning any RECORDs that arrived along the way and the terminal message.
The record slice exists to be asserted EMPTY. DISCARD's contract is that the rows are dropped server-side rather than delivered, so a non-empty slice here is the defect; a helper that could not observe a stray RECORD could not tell DISCARD from PULL. n <= 0 discards the whole remaining stream; n > 0 discards up to n rows and the terminal SUCCESS reports has_more for the remainder (bolt/server/session.go handleDiscard).
Nothing in the harness sent DISCARD before rmp #2484: every call site used PULL -1, so the whole discard path of the session — including its statement-error guard and its own has_more accounting — was driven by no scenario at all.
func (*WireClient) DiscardQID ¶ added in v0.12.0
DiscardQID sends DISCARD {n:n, qid:qid} with an EXPLICIT qid and reads to the terminal reply. As with WireClient.PullQID, a qid >= 0 is expected to be refused with the same code and message shape (bolt/server/session.go:1421-1424).
func (*WireClient) Goodbye ¶
func (c *WireClient) Goodbye() error
Goodbye sends GOODBYE. No response is expected; the server tears the session down.
func (*WireClient) Handshake ¶
Handshake performs the 20-byte Bolt client handshake, offering versions 5.6 down to 5.0 across the four slots, and records the negotiated version. It returns an error if the server rejects negotiation (responds with 0.0) or an I/O error occurs.
func (*WireClient) HandshakeOffering ¶ added in v0.12.0
func (c *WireClient) HandshakeOffering(ctx context.Context, offers ...proto.Version) (proto.Version, error)
HandshakeOffering performs the 20-byte Bolt client handshake offering exactly the given versions, one per slot, and records the negotiated version. It is the explicit-offer counterpart of WireClient.Handshake, which always leads with 5.6: a probe that must reach the pre-5.1 INLINE-auth path (credentials on HELLO) or a specific entity layout has to be able to withhold the newer versions, because the server correctly picks the highest it is offered.
At most four versions may be offered (the preamble has four slots); each is offered with a minor RANGE of zero, i.e. that exact version and no fallback. It returns an error when the server rejects negotiation (responds 0.0).
func (*WireClient) HandshakeOfferingSlots ¶ added in v0.12.0
func (c *WireClient) HandshakeOfferingSlots(_ context.Context, slots [4]BoltOffer) (proto.Version, error)
HandshakeOfferingSlots performs the 20-byte Bolt client handshake writing the four preamble slots EXACTLY as given, and records the negotiated version. It is the single primitive behind WireClient.Handshake and WireClient.HandshakeOffering, which are the two fixed spellings the harness used before rmp #2486; both delegate here, so the preamble is built in one place and [TestWireClientHandshake_PreambleBytesAreUnchanged] can pin the exact 20 bytes each of them puts on the wire.
It exists because neither of those two can express what a version-matrix probe needs. WireClient.HandshakeOffering hard-codes a minor_range of ZERO and packs its offers into slots 0..n-1, so nothing in the harness could send a RANGE offer other than the one WireClient.Handshake hard-codes, and nothing could place an offer in a chosen slot. Both matter: the range is a distinct branch of Negotiate's matching loop (the `sv.Minor >= minMinor` half, bolt/proto/handshake.go:121), and slot placement is what shows the server's choice is driven by ITS preference order rather than by the client's, because Negotiate scans SupportedVersions in the OUTER loop (:109-110).
An all-empty preamble is REFUSED here rather than sent. Negotiate would answer it with the no-common-version rejection, which is indistinguishable from a deliberate probe of that path; a caller that wants the rejection asks for it by offering a version this server does not support.
It returns an error when the server rejects negotiation (responds 0.0).
func (*WireClient) Hello ¶
func (c *WireClient) Hello(extra map[string]packstream.Value) (any, error)
Hello sends a HELLO with scheme="none" (the NoAuth server admits it) and returns the response message (typically *proto.Success). For Bolt 5.1+ the server defers auth to a LOGON message; this client targets the inline (<=5.0-style) HELLO auth the NoAuth handler accepts, which the server honours across the supported versions in the DST harness.
func (*WireClient) Logoff ¶ added in v0.12.0
func (c *WireClient) Logoff() (any, error)
Logoff sends a LOGOFF and returns the response. A successful LOGOFF de-authorises the connection: every subsequent query-bearing or transaction-finalising message must be refused until a fresh LOGON re-authenticates (the CWE-306 gate).
func (*WireClient) Logon ¶
func (c *WireClient) Logon() (any, error)
Logon sends a LOGON with scheme="none" for the Bolt 5.1+ deferred-auth path and returns the response.
func (*WireClient) LogonWith ¶ added in v0.12.0
func (c *WireClient) LogonWith(auth map[string]packstream.Value) (any, error)
LogonWith sends a LOGON carrying the given auth token and returns the response. It is the credential-bearing counterpart of WireClient.Logon, used by the auth-surface scenario to present a right or a wrong password to a server whose github.com/FlavioCFOliveira/GoGraph/bolt/server.AuthHandler actually validates one (rmp #2481).
func (*WireClient) NetConn ¶ added in v0.13.0
func (c *WireClient) NetConn() net.Conn
NetConn returns the underlying transport, whichever kind it is. It is never nil, and it is the accessor to reach for when the client may have been built over a real socket by NewWireClientNetConn.
func (*WireClient) Pull ¶
Pull sends PULL {n:n} and reads up to n RECORDs plus the terminal message. It is used by the SlowConsumer, which pulls in small batches with deliberate stalls between calls, and by the paging arms of rmp #2484.
func (*WireClient) PullAll ¶
func (c *WireClient) PullAll() (records []*proto.Record, terminal any, err error)
PullAll sends PULL {n:-1} and reads every RECORD up to the terminal SUCCESS or FAILURE, returning the records and the terminal message. A FAILURE terminates the pull with the records gathered so far.
func (*WireClient) PullQID ¶ added in v0.12.0
PullQID sends PULL {n:n, qid:qid} with an EXPLICIT qid and reads to the terminal reply. It exists so a scenario can address a stream by qid rather than by the implicit current-stream -1.
A qid >= 0 is expected to be REFUSED: this server keeps exactly one open stream per session (RUN always reports qid = -1, and a second RUN while streaming is an illegal transition), so handlePull answers any non-negative qid with Neo.ClientError.Request.Invalid / "no such query: qid N" (bolt/server/session.go:1240-1243).
func (*WireClient) Recv ¶
func (c *WireClient) Recv() (any, error)
Recv reads and decodes the next response message; exported for actors that read a server-initiated response outside the Request/Pull helpers.
func (*WireClient) RecvRaw ¶
func (c *WireClient) RecvRaw() ([]byte, error)
RecvRaw reads one chunked message and returns its raw bytes without decoding, for the abuser to inspect a FAILURE the standard decoder would also handle.
func (*WireClient) Request ¶
func (c *WireClient) Request(msg any) (any, error)
Request is the LOCK-STEP primitive: it sends one request and reads exactly one response message back. For messages whose reply is a single SUCCESS/FAILURE (HELLO, LOGON, RUN, BEGIN, COMMIT, ROLLBACK, RESET, ROUTE) this is the full terminal exchange. It returns the decoded *proto.Success, *proto.Failure, or *proto.Ignored.
func (*WireClient) Reset ¶
func (c *WireClient) Reset() (any, error)
Reset sends RESET and returns the response.
func (*WireClient) Rollback ¶
func (c *WireClient) Rollback() (any, error)
Rollback sends ROLLBACK and returns the response.
func (*WireClient) Run ¶
Run sends a RUN for query with params and returns the response (a *proto.Success carrying the field metadata, or a *proto.Failure). It does NOT pull records; follow with WireClient.PullAll or WireClient.Pull.
func (*WireClient) Version ¶
func (c *WireClient) Version() proto.Version
Version reports the negotiated protocol version (zero before Handshake).
func (*WireClient) WriteChunkedRaw ¶
func (c *WireClient) WriteChunkedRaw(payload []byte) error
WriteChunkedRaw writes payload as one well-framed chunked message regardless of whether payload decodes to a valid Bolt message. The BoltAbuser uses it to deliver garbage opcodes and wrong-state messages that are correctly framed but semantically invalid, exercising the server's message-level (not framing-level) rejection.
func (*WireClient) WriteRaw ¶
func (c *WireClient) WriteRaw(p []byte) (int, error)
WriteRaw writes raw bytes directly to the connection, bypassing chunked framing. It is the seam the BoltAbuser uses to emit deliberately malformed wire bytes (bad handshakes, truncated chunks, garbage opcodes) the framed send path would never produce.
type WireExchange ¶
type WireExchange struct {
// Op identifies the operation sent (kind + cypher).
Op string
// Response is the terminal response class (SUCCESS, FAILURE:<code>, …).
Response string
}
WireExchange is one decoded request/response pair from a lock-step wire session, rendered to stable strings so two runs can be compared byte-for-byte.
type WireTranscript ¶
type WireTranscript struct {
Exchanges []WireExchange
Seed uint64
}
WireTranscript is the ordered list of exchanges from one lock-step session. Two transcripts produced from the same seed must be equal — the determinism guarantee of the single-connection lock-step Bolt-wire path.
func RunLockStepWire ¶
func RunLockStepWire(seed uint64, nOps int) (WireTranscript, error)
RunLockStepWire drives a deterministic single-connection LOCK-STEP session against a fresh real bolt/server: for the given seed it draws a fixed sequence of nOps honest write operations, sends each over the wire, blocks for the terminal response, and records the exchange. Because exactly one exchange is in flight at a time, the transcript is a pure function of the seed, so two calls with the same seed return equal transcripts (assert with WireTranscript.Equal).
It is the engine behind the cmd/sim `--mode wire` reproducibility demo and the determinism proof for the lock-step path.
func (WireTranscript) Equal ¶
func (t WireTranscript) Equal(other WireTranscript) bool
Equal reports whether two transcripts are byte-identical (same length, same ordered exchanges). It is the reproducibility predicate.
type Workload ¶
Workload is a weighted mix of actors. On each tick the simulator asks the workload to select an actor (by weight) and that actor produces the next operation. The weights need not sum to 1; Workload.SelectActor normalises against their running total.
Concurrency contract ¶
Workload is NOT safe for concurrent use; it is consulted from the single simulation goroutine. The Seed passed to SelectActor is the same shared single-goroutine seed, preserving determinism.
func BadActorWorkload ¶
BadActorWorkload returns a mix that injects a MalformedSender alongside the honest actors (50% writer, 30% reader, 20% malformed), so the safety loop continuously exercises the engine's rejection paths while honest traffic keeps the graph populated. The malformed traffic must never panic, corrupt state, or trip an invariant: each ill-formed op is modelled by the oracle as a no-op, so a clean run sees engine and oracle stay in lock-step across every rejection.
func DefaultWorkload ¶
DefaultWorkload returns a balanced mix: 40% writer, 60% reader. The seed is accepted for symmetry with the other constructors (and to allow future seed-dependent compositions) but the default mix is fixed.
func MemPressureWorkload ¶ added in v0.6.0
MemPressureWorkload returns a mix that drives the engine's bounded-resource contract: an honest writer keeps a real graph populated (50%) while an OverloadReader (50%) issues over-budget reads. With the engine's logical budgets clamped (the mem-pressure scenario's EngineOpts), the overload reads are refused with typed resource-exhausted errors and change no state, so the honest writes and the oracle stay in lock-step throughout — graceful degradation, never a panic or partial result.
func ReadHeavyWorkload ¶
ReadHeavyWorkload returns a 20% writer / 80% reader mix, stressing the read path and isolation.
func SteadyStateWorkload ¶
SteadyStateWorkload returns a mix tuned to keep the modelled graph BOUNDED over a very long run: a writer that creates and deletes in equal measure (via BoundedChurnWriter) plus a reader. It is the long-running scenario's workload — the point there is heap/goroutine stability across millions of small ops, which requires the working set not to grow without bound.
func WriteHeavyWorkload ¶
WriteHeavyWorkload returns an 80% writer / 20% reader mix, stressing the write and recovery paths.
func (*Workload) SelectActor ¶
SelectActor returns one actor chosen with probability proportional to its weight, drawing a single float64 from seed. It panics if the workload has no actors (a programmer error). A non-positive total weight falls back to a uniform first-actor choice rather than dividing by zero.
type XReleaseBuildOptions ¶ added in v0.12.0
type XReleaseBuildOptions struct {
// ForceWALOnly skips the checkpoint-bearing build stage entirely and builds
// the WAL-only helper directly, so the resulting image carries NO snapshot
// directory however capable the tag actually is.
//
// It exists to keep the snapshot oracle falsifiable (rmp #2531). Once every
// tag in the harness's list publishes a snapshot, every arm reports
// SnapshotOpened=true — and a flag that is true in every run observed is
// indistinguishable from a flag hard-wired to true. This option constructs the
// opposite case on demand: a genuine run of the whole pipeline over an image
// that provably has no snapshot, where SnapshotOpened MUST read false.
//
// It is the permanent form of "revert the fix and check the test fails". A
// one-off manual revert proves the oracle discriminates on the day it is done
// and proves nothing thereafter; wiring the negative case into the harness
// keeps it proven on every soak run.
//
// [PriorReleaseHelper.BuildFallbackErr] stays nil under this option: nothing
// failed, so there is no build error to report. CheckpointSupported is false,
// which is the honest description of the binary that was produced.
ForceWALOnly bool
}
XReleaseBuildOptions tunes how a prior-release helper is built.
Source Files
¶
- access_parity.go
- actor.go
- adapter.go
- bolt_auth_surface.go
- bolt_begin_extras.go
- bolt_cert_rotation.go
- bolt_decode_pressure.go
- bolt_shutdown_drain.go
- bolt_stream_semantics.go
- bolt_tx_quota.go
- bolt_tx_registry.go
- bolt_tx_terminate.go
- bolt_version_matrix.go
- boltabuser.go
- bulk_load_oracle.go
- bulkimport_faults.go
- bulkimport_parity.go
- call_yield_where.go
- catalogue.go
- checker.go
- checkpoint_cadence.go
- checkpoint_crash_storm.go
- clock.go
- codec_matrix.go
- codec_matrix_run.go
- concurrent.go
- concurrent_iso.go
- concurrent_tx.go
- constraint.go
- constraint_existence.go
- constraint_kinds.go
- contended_counter_space.go
- count_store.go
- counters_oracle.go
- coverage.go
- crash.go
- crossrelease.go
- crossrelease_compat.go
- crossrelease_run.go
- csrfile_access_matrix.go
- cypher_expr_literals.go
- db_teardown.go
- ddl_checkpoint_crash.go
- ddl_counters_oracle.go
- delete_contract.go
- differential.go
- disk.go
- diskfs.go
- durable_scenarios.go
- edge_properties.go
- fluent_query.go
- foreach.go
- generation_swap.go
- gomaxprocs.go
- graph_io_surface.go
- group_commit.go
- index_diversity.go
- index_hydration_arms.go
- index_intersect_probes.go
- index_seek_results.go
- label_index_scoped.go
- liveness.go
- merge_rel.go
- merge_surface.go
- metrics_oracle.go
- metrics_required.go
- mvcc_clock_recovery.go
- mvcc_contention.go
- mvcc_isolation.go
- mvcc_sessions.go
- mvcc_substrate.go
- mvcc_substrate_arms.go
- nodeid_stability.go
- notifications.go
- null_semantics.go
- oracle.go
- oracle_tx.go
- overloadactor.go
- pagerank_ranker.go
- pattern_shapes.go
- production_profile.go
- report.go
- scenario.go
- schema_introspect.go
- schema_mutation.go
- schemachanger.go
- search_bfs_do.go
- search_bibfs.go
- search_centrality.go
- search_centrality_measures.go
- search_check.go
- search_community.go
- search_ctx_cancel.go
- search_diameter.go
- search_euler.go
- search_extern.go
- search_flow.go
- search_kcore_bcc.go
- search_kshortest.go
- search_matching.go
- search_mst.go
- search_negweight.go
- search_oracle.go
- search_ordering.go
- search_pagerank.go
- search_shapes.go
- search_sssp.go
- search_triangles.go
- seed.go
- shortestpath.go
- shrink.go
- sim.go
- simconn.go
- simlistener.go
- simserver.go
- simstore.go
- slowconsumer.go
- snapshot_codec.go
- snapshot_corruption.go
- stats_regime.go
- storage_fault_scenarios.go
- surface.go
- surface_agg.go
- surface_entity.go
- surface_order.go
- swarm.go
- trace.go
- txn_oversize.go
- type_coverage.go
- typed_schema.go
- upgrade.go
- varlen_paths.go
- wal_txn_counters.go
- wal_writer_surface.go
- wire_lockstep.go
- wire_param_types.go
- wireclient.go
- workload.go