matrixir

package
v1.3.2 Latest Latest
Warning

This package is not in the latest version of its module.

Go to latest
Published: Sep 13, 2026 License: MIT Imports: 18 Imported by: 0

Documentation

Overview

Copyright (c) 2026 Tarek Wasfy

Copyright (c) 2026 Tarek Wasfy

Index

Constants

View Source
const (
	StatementAssign = iota
	StatementPrint
	StatementReturn
	StatementExpression
	StatementClasses
)
View Source
const (
	GrammarBrace = iota
	GrammarIndent
	GrammarEnd
	GrammarSemicolon
	GrammarNewline
	GrammarTypedDeclaration
	GrammarMainWrapper
	GrammarOneBasedIndex
	GrammarIntegerSlashTruncates
	GrammarExclusiveRangeEnd
	GrammarDimensions
)
View Source
const (
	ActionSkip = iota
	ActionBlock
	ActionAssign
	ActionPrint
	ActionReturn
	ActionExpression
	ActionIf
	ActionElse
	ActionWhile
	ActionFor
	ActionFunction
	ActionCall
	ActionIndex
	ActionModule
	ActionObject
	ActionException
	ActionConcurrency
	ActionReflection
	ActionDimensions
)
View Source
const (
	LexIdentifier = iota
	LexNumber
	LexString
	LexStringComment
	LexStringKeyword
	LexBoolean
	LexReturn
	LexPrint
	LexIf
	LexWhile
	LexFor
	LexFunction
	LexObject
	LexException
	LexModule
	LexConcurrency
	LexReflection
	LexArithmetic
	LexDivision
	LexIntegerDivision
	LexComparison
	LexLogical
	LexShortCircuit
	LexBinding
	LexNamedArgument
	LexIndex
	LexGrouping
	LexCall
	LexBlock
	LexComment
	LexicalAxes
)
View Source
const (
	SemProgram = iota
	SemBlock
	SemAssign
	SemPrint
	SemReturn
	SemExpression
	SemIdentifier
	SemNumber
	SemString
	SemBoolean
	SemNull
	SemUnary
	SemBinary
	SemCall
	SemIndex
	SemIf
	SemWhile
	SemFor
	SemFunction
	SemArithmetic
	SemDivision
	SemIntegerDivision
	SemComparison
	SemLogical
	SemBinding
	SemReassignment
	SemScope
	SemClosure
	SemNamedArgument
	SemLazyEvaluation
	SemShortCircuit
	SemOverflow
	SemObject
	SemException
	SemGeneric
	SemOwnership
	SemModule
	SemConcurrency
	SemFFI
	SemReflection
	SemSerialization
	SemIO
	SemEffect
	SemStringComment
	SemStringKeyword
	SemMultiline
	SemGrouping
	SemUnknown
	SemanticDimensions
)

Variables

View Source
var ActionNames = [...]string{
	"skip", "block", "assign", "print", "return", "expression", "if", "else", "while", "for", "function", "call", "index", "module", "object", "exception", "concurrency", "reflection",
}
View Source
var Features = [...]string{
	"arithmetic", "binding", "reassignment", "division", "boolean", "string_comment", "string_keywords", "comparison",
	"multiline", "grouping", "if_else", "while", "for", "function", "index", "integer_division",
	"scope", "closure", "named_args", "lazy_eval", "short_circuit", "overflow", "null_na", "objects",
	"exceptions", "generics", "ownership", "modules", "concurrency", "ffi", "reflection", "serialization",
}
View Source
var GrammarNames = [...]string{
	"brace_blocks", "indent_blocks", "end_blocks", "semicolon", "newline", "typed_declaration", "main_wrapper", "one_based_index", "integer_slash_truncates", "exclusive_range_end",
}
View Source
var Languages = [...]string{"r", "go", "rust", "cpp", "c", "python", "zig", "julia", "nim", "csharp", "java", "kotlin", "swift"}
View Source
var RelationNames = [...]string{"syntax", "control", "data", "binding", "effect"}
View Source
var SemanticNames = [...]string{
	"program", "block", "assign", "print", "return", "expression", "identifier", "number", "string", "boolean", "null",
	"unary", "binary", "call", "index", "if", "while", "for", "function", "arithmetic", "division", "integer_division",
	"comparison", "logical", "binding", "reassignment", "scope", "closure", "named_argument", "lazy_evaluation", "short_circuit",
	"overflow", "object", "exception", "generic", "ownership", "module", "concurrency", "ffi", "reflection", "serialization",
	"io", "effect", "string_comment", "string_keyword", "multiline", "grouping", "unknown",
}

Functions

func BuildLexicalGraph

func BuildLexicalGraph(source, code string) (*Graph, []Lexeme, error)

func ClassifyStatement

func ClassifyStatement(vector Vector) (int, error)

func EvalLexPredicate added in v1.0.6

func EvalLexPredicate(t RealTables, id string, r rune, eof bool) bool

func FeatureIndex

func FeatureIndex(feature string) (int, bool)

func FirstMachineDivergence added in v1.0.6

func FirstMachineDivergence(reference, generic []NormalizedMachineEvent) (int, *NormalizedMachineEvent, *NormalizedMachineEvent)

FirstMachineDivergence returns the first row whose normalized vectors differ.

func FirstMachineDivergenceStrict added in v1.0.6

func FirstMachineDivergenceStrict(reference, generic []NormalizedMachineEvent) (int, *NormalizedMachineEvent, *NormalizedMachineEvent)

FirstMachineDivergenceStrict is used for conformance reports. Unlike the historical compatibility comparator, unknown fields are never wildcards: an omitted parser/lexer state is a real divergence until the oracle and the generic trace both provide the same value.

func LanguageIndex

func LanguageIndex(language string) (int, bool)

func MustFeatureIndex

func MustFeatureIndex(feature string) int

func TraceCanonicalize added in v1.0.6

func TraceCanonicalize(language, source string) (CanonicalProgram, []LegacyTraceRow, error)

TraceCanonicalize records the stable semantic-event/role observations of the existing parser for bootstrap and differential validation.

Types

type AliasSequenceEntry added in v1.0.6

type AliasSequenceEntry struct {
	Language                                string
	ProductionID, ChildIndex, AliasSymbolID int
	AliasSymbolName                         string
}

type CanonicalBindingFact added in v1.0.5

type CanonicalBindingFact struct {
	ID, DeclarationNodeID, ReferenceNodeID, SymbolID, ScopeID, Ordinal, EvidenceRef int
	Kind                                                                            string
	SourceOffset                                                                    int
}

type CanonicalEvent

type CanonicalEvent struct {
	Text   string
	Source int
}

CanonicalEvent preserves structural action order for direct lowerers. It is not a source string field and is deliberately separate from the legacy R diagnostic view.

type CanonicalNode

type CanonicalNode struct {
	Action Vector
	Text   string
	Source int
	// Parent is the canonical action node that owns this action's block.  It is
	// filled while the grammar/block stack is active and lets the semantic
	// event bridge preserve body/statement structure without reparsing text.
	Parent int
	// Close and Post are structural action payload, not R transport text.
	// They make block boundaries available to direct semantic lowerers.
	Close int
	Post  string
	// Branch records that an `else`/`elif` header is a branch of the
	// immediately preceding if construct.  It is parser structure, not source
	// text; the semantic lowering layer uses it to attach the branch body to
	// the existing IfStmt.
	Branch string
}

type CanonicalOperandFact added in v1.0.5

type CanonicalOperandFact struct {
	ID, OwnerNodeID, ReferencedNodeID, ReferencedSymbolID, ReferencedTypeID int
	Role                                                                    string
	Ordinal                                                                 int
	Field                                                                   string
	SourceOffset                                                            int
}

The following records are intentionally language-neutral and optional. Empty entries mean the matrix/token analysis did not prove that fact.

type CanonicalProgram

type CanonicalProgram struct {
	Source         string
	R              string
	Nodes          []CanonicalNode
	Graph          *Graph
	Actions        Matrix
	Grammar        Vector
	Roles          Matrix
	Lexemes        []Lexeme
	Events         []CanonicalEvent
	SemanticEvents []CanonicalSemanticEvent
}

func Canonicalize

func Canonicalize(source, code string) (CanonicalProgram, error)

Canonicalize lowers the supported common subset through token structure and action matrices into R-shaped syntax. It is not a full parser for all source languages; the R expression parser and some header recognizers remain.

type CanonicalRelationFact added in v1.0.5

type CanonicalRelationFact struct {
	Kind                          string
	FromNodeID, ToNodeID, Ordinal int
	Role                          string
}

type CanonicalRoleFact added in v1.0.5

type CanonicalRoleFact struct {
	OwnerNodeID, ChildNodeID, Ordinal int
	Role                              string
}

type CanonicalSemanticEvent added in v1.0.5

type CanonicalSemanticEvent struct {
	ID            int
	Action        string
	Semantic      Vector
	StructureKind string
	Text          string
	SourceOffset  int
	Fields        map[string]string
	ParentID      int
	ChildIDs      []int
	Roles         []CanonicalRoleFact
	Operands      []CanonicalOperandFact
	Symbols       []CanonicalSymbolFact
	Bindings      []CanonicalBindingFact
	Relations     []CanonicalRelationFact
	Evidence      []string
	LanguageFacts map[string]string
	// FactFamily is a structured producer classification. It is deliberately
	// independent of source language and carries no executable source text.
	FactFamily ParsedConstructFamily
}

CanonicalSemanticEvent is the typed, transient frontend contract emitted by MatrixIR. It contains only grammar- and matrix-proven facts and is discarded after a frontend has materialized its UAST facts.

func AnalyzeSemanticExpression added in v1.0.5

func AnalyzeSemanticExpression(language, text string) ([]CanonicalSemanticEvent, error)

AnalyzeSemanticExpression is the shared precedence/grouping pass used by matrix frontends. It emits typed transient events and never constructs a legacy AST.

func AnalyzeSemanticStatement added in v1.0.5

func AnalyzeSemanticStatement(language string, event CanonicalSemanticEvent) ([]CanonicalSemanticEvent, error)

AnalyzeSemanticStatement recognizes the statement forms already selected by the matrix action layer. It is a temporary frontend analysis and does not create legacy Stmt values.

func AnalyzeSemanticStatementTokens added in v1.0.6

func AnalyzeSemanticStatementTokens(language string, event CanonicalSemanticEvent, tokens []Lexeme) ([]CanonicalSemanticEvent, error)

AnalyzeSemanticStatementTokens consumes the grammar tokens attached to a recognised statement. It is deliberately separate from the legacy text helper below: structured MatrixIR paths must never parse Event.Text again.

func AnalyzeSemanticTokens added in v1.0.6

func AnalyzeSemanticTokens(language string, tokens []Lexeme) ([]CanonicalSemanticEvent, error)

AnalyzeSemanticTokens is the structured counterpart of AnalyzeSemanticExpression. Its caller already owns parser/grammar tokens; consequently it never reparses an Event.Text payload.

type CanonicalSymbolFact added in v1.0.5

type CanonicalSymbolFact struct {
	ID, NodeID   int
	Name, Kind   string
	ScopeID      int
	SourceOffset int
}

type CapabilityModel

type CapabilityModel struct {
	Support Matrix // language x feature
	Known   Matrix // language x feature
}

CapabilityModel holds support and observation masks. A zero in Support is meaningful only when the corresponding Known cell is one.

func NewCapabilityModel

func NewCapabilityModel(support, known Matrix) (CapabilityModel, error)

func (CapabilityModel) DeficitMatrices

func (m CapabilityModel) DeficitMatrices(requirements Matrix) (missing Matrix, unknown Matrix, err error)

DeficitMatrices computes D=R*(K-S)^T and Q=R*(1-K)^T. R is a source/program x feature requirement matrix. D counts known missing contracts per target, Q counts unknown required contracts per target.

func (CapabilityModel) Plan

func (m CapabilityModel) Plan(requirements Vector, target string) (TargetPlan, error)

type ExecutionReadyBundle added in v1.2.8

type ExecutionReadyBundle struct {
	Dir       string   `json:"dir"`
	Hash      string   `json:"hash"`
	Languages []string `json:"languages"`
}

ExecutionReadyBundle is the immutable authority selected for a frontend parse. The hash covers the table files that determine machine behaviour; callers can retain it in diagnostics and semantic provenance.

func ResolveExecutionReadyBundle added in v1.2.8

func ResolveExecutionReadyBundle(explicit string, candidates ...string) (ExecutionReadyBundle, error)

ResolveExecutionReadyBundle deterministically selects one complete bundle. An explicit directory wins. Otherwise the first complete candidate in the supplied order is used. No language-specific selection is performed.

func (ExecutionReadyBundle) MarshalJSON added in v1.2.8

func (b ExecutionReadyBundle) MarshalJSON() ([]byte, error)

type ExternalScanInput added in v1.0.6

type ExternalScanInput struct {
	Source       string
	Offset       int
	ValidSymbols []bool
}

type ExternalScanResult added in v1.0.6

type ExternalScanResult struct {
	// Accepted distinguishes a valid external token with symbol id 0 from
	// scanner refusal. Tree-sitter external token index 0 is a real token.
	Accepted       bool
	AcceptedSymbol int
	EndOffset      int
	Serialized     []byte
}

type ExternalScannerAdapter added in v1.0.6

type ExternalScannerAdapter interface {
	Create() (payload uintptr, err error)
	Destroy(payload uintptr) error
	Scan(payload uintptr, input *ExternalScanInput) (ExternalScanResult, error)
	Serialize(payload uintptr) ([]byte, error)
	Deserialize(payload uintptr, state []byte) error
}

ExternalScannerAdapter is the language-neutral representation of the Tree-sitter external-scanner ABI. Implementations are supplied by a temporary, locally compiled scanner host; the parser stores Serialized state per GLR version and never treats ExternalLexState as scanner state.

type ExternalScannerKernel added in v1.0.6

type ExternalScannerKernel struct {
	Language, ScannerPath, SHA256, Class string
	ValidSymbolRefs, ResultSymbols       []string
	Stateful                             bool
}

type FieldMapEntry added in v1.0.6

type FieldMapEntry struct {
	Language                          string
	ProductionID, ChildIndex, FieldID int
	FieldName                         string
	Inherited                         bool
}

type GenericGLRConflictEngine added in v1.0.6

type GenericGLRConflictEngine struct{}

func NewGenericGLRConflictEngine added in v1.0.6

func NewGenericGLRConflictEngine() *GenericGLRConflictEngine

func (*GenericGLRConflictEngine) Resolve added in v1.0.6

type GenericLexerEngine added in v1.0.6

type GenericLexerEngine struct{ Language string }

func NewGenericLexerEngine added in v1.0.6

func NewGenericLexerEngine(language string) *GenericLexerEngine

func (*GenericLexerEngine) Lex added in v1.0.6

func (e *GenericLexerEngine) Lex(source string) []Lexeme

func (*GenericLexerEngine) NextSymbol added in v1.0.6

func (e *GenericLexerEngine) NextSymbol(t RealTables, parseState, pos int, source string) (symbol, next int, ok bool)

func (*GenericLexerEngine) NextToken added in v1.0.6

func (e *GenericLexerEngine) NextToken(t RealTables, parseState, pos int, source string) (LexToken, bool)

NextToken executes Tree-sitter's internal lexer first and invokes the generated keyword lexer only when the main lexer returned the grammar's keyword-capture token. A keyword is promoted only when it is valid in the current parse state or belongs to that state's reserved-word set.

type GenericLexerLREngine added in v1.0.6

type GenericLexerLREngine struct {
	Language     string
	MachineTrace []MachineTraceStep
	// Trace collection is opt-in. Corpus execution only needs ParseStats and
	// ParseNode; copying the complete stack for every lexer/reduction event was
	// the dominant memory cost and made valid GLR cases look like timeouts.
	TraceEnabled bool
}

func NewGenericLexerLREngine added in v1.0.6

func NewGenericLexerLREngine(language string) *GenericLexerLREngine

func (*GenericLexerLREngine) Parse added in v1.0.6

func (e *GenericLexerLREngine) Parse(source string) (CanonicalProgram, error)

Parse executes the shared lexical/semantic pipeline. The existing parser is used only as the bootstrap implementation until generated ACTION/GOTO tables are promoted; callers already receive the canonical event contract.

func (*GenericLexerLREngine) ParseReal added in v1.0.6

func (e *GenericLexerLREngine) ParseReal(source, tableDir string) (ParseStats, bool, error)

ParseReal executes the table-backed lexer/parser path. It intentionally returns execution statistics and does not construct semantic events.

func (*GenericLexerLREngine) ParseRealContext added in v1.0.6

func (e *GenericLexerLREngine) ParseRealContext(ctx context.Context, source, tableDir string) (ParseStats, bool, error)

ParseRealContext executes the real-table parser with cooperative cancellation. Callers that impose a per-input deadline can therefore stop a pathological GLR branch without leaving a parser goroutine behind.

func (*GenericLexerLREngine) ParseRealNodes added in v1.0.6

func (e *GenericLexerLREngine) ParseRealNodes(source, tableDir string) (*ParseNode, ParseStats, bool, error)

ParseRealNodes executes the same real-table path and returns the neutral reduction tree. No legacy parser or source re-tokenizer is involved.

func (*GenericLexerLREngine) ParseRealNodesContext added in v1.0.6

func (e *GenericLexerLREngine) ParseRealNodesContext(ctx context.Context, source, tableDir string) (*ParseNode, ParseStats, bool, error)

ParseRealNodesContext is the cancellable neutral ParseNode entry point.

func (*GenericLexerLREngine) ParseRealTrace added in v1.0.6

func (e *GenericLexerLREngine) ParseRealTrace(source, tableDir string) ([]MachineTraceStep, ParseStats, bool, error)

func (*GenericLexerLREngine) Trace added in v1.0.6

type GenericParserEngine added in v1.0.6

type GenericParserEngine struct {
	Language string
	Tables   *RealTables
}

func NewGenericParserEngine added in v1.0.6

func NewGenericParserEngine(language string) *GenericParserEngine

func (*GenericParserEngine) ActionList added in v1.0.6

func (e *GenericParserEngine) ActionList(id int) []ParseAction

func (*GenericParserEngine) AttachTables added in v1.0.6

func (e *GenericParserEngine) AttachTables(t RealTables)

func (*GenericParserEngine) Dispatch added in v1.0.6

func (e *GenericParserEngine) Dispatch(state, symbol int) (ParseDispatchEntry, bool)

func (*GenericParserEngine) Parse added in v1.0.6

func (e *GenericParserEngine) Parse(source string) (CanonicalProgram, error)

func (*GenericParserEngine) ParseSymbols added in v1.0.6

func (e *GenericParserEngine) ParseSymbols(symbols []int) (ParseStats, bool)

ParseSymbols executes the canonical ACTION/GOTO machine for a token-symbol stream. It is independent of Canonicalize and is used by the bootstrap harness while the source lexer is being connected to symbol IDs.

type GenericProducerEngine added in v1.0.6

type GenericProducerEngine struct{}

func NewGenericProducerEngine added in v1.0.6

func NewGenericProducerEngine() *GenericProducerEngine

func (*GenericProducerEngine) Produce added in v1.0.6

func (*GenericProducerEngine) ProduceParseNode added in v1.2.8

func (e *GenericProducerEngine) ProduceParseNode(language string, tables RealTables, root *ParseNode) []CanonicalSemanticEvent

ProduceParseNode is the language-neutral producer boundary for the execution-ready parser. It deliberately consumes only parser facts and symbol metadata; no source spelling, diagnostics, or language switch is consulted. The event shape is intentionally the same CanonicalSemanticEvent contract used by the existing MatrixIR frontend.

type Graph

type Graph struct {
	Language string
	Nodes    []Node
	Edges    [RelationCount]SparseMatrix
}

func NewGraph

func NewGraph(language string) (*Graph, error)

func (*Graph) AddNode

func (g *Graph) AddNode(vector Vector, text, symbol string, sourceAt int) int

func (*Graph) AddNodes

func (g *Graph) AddNodes(nodes []Node) []int

AddNodes appends a batch with consecutive graph-owned IDs and copied semantic vectors. Relation matrices grow their dimensions without copying any edges. Invalid vectors are rejected before the graph is changed; incoming IDs are ignored because IDs are positions in this graph's relation matrices.

func (*Graph) Connect

func (g *Graph) Connect(relation Relation, from, to int) error

func (*Graph) NodeMatrix

func (g *Graph) NodeMatrix() Matrix

func (*Graph) RelationCounts

func (g *Graph) RelationCounts() Vector

func (*Graph) Requirements

func (g *Graph) Requirements() Vector

Requirements performs X*W, where X is the node matrix and W projects semantic dimensions onto the shared 32-feature contract. Counts are reduced to a binary program requirement vector after multiplication.

func (*Graph) SemanticRequirements

func (g *Graph) SemanticRequirements() Vector

type LegacyTraceRow added in v1.0.6

type LegacyTraceRow struct {
	Language       string
	Token          string
	TokenClass     string
	LexerContext   string
	ParserContext  string
	Lookahead      string
	ConsumedSymbol string
	ProducedNode   int
	ChildRole      string
	FieldRole      string
	OperatorRole   string
	BindingRole    string
	ControlRole    string
	SemanticEvent  int
}

LegacyTraceRow is a structured teacher observation. It contains no source text and is suitable for joining against offline lexer/parser tables.

type LexCharacterRange added in v1.0.6

type LexCharacterRange struct {
	SetID      string
	Start, End int
}

type LexDispatchEntry added in v1.0.6

type LexDispatchEntry struct {
	Language, LexerFunction                                              string
	LexState, TransitionOrdinal, NextState, AcceptSymbolID, MapCodepoint int
	PredicateKind, Start, End, PredicateID                               string
	EOF, Skip, Immediate                                                 bool
}

type LexModeEntry added in v1.0.6

type LexModeEntry struct{ ParseState, LexState, ExternalLexState, ReservedWordSetID int }

type LexPredicateInstruction added in v1.0.6

type LexPredicateInstruction struct {
	PredicateID        string
	Ordinal            int
	Opcode, Arg1, Arg2 string
}

type LexToken added in v1.0.6

type LexToken struct {
	SymbolID, Start, End int
	Text                 string
}

type Lexeme

type Lexeme struct {
	Class    TokenClass
	Text     string
	Start    int
	End      int
	Depth    int
	Axes     Vector
	Semantic Vector
}

func Tokenize

func Tokenize(source, code string) []Lexeme

type MachineTraceStep added in v1.0.6

type MachineTraceStep struct {
	ParserState, LexState, Lookahead, Symbol, ActionList int
	Action                                               string
	// ActionIndex and the reduction metadata are copied from the selected
	// table row so differential traces can be compared without diagnostics.
	ActionIndex, StackVersion, ProductionID, DynamicPrecedence int
	Stack                                                      []int
}

GenericLexerLREngine is the language-neutral execution facade for the offline table set. Tables are identified by language; semantic output is a CanonicalProgram and never a legacy AST.

type Matrix

type Matrix struct {
	Rows int
	Cols int
	Data []float64
}

Matrix is a dense row-major matrix. The semantic graphs are sparse in meaning, but dense storage keeps dimensions and multiplication auditable at the current program sizes. Storage can later change without changing the IR.

func ActionSemanticProjection

func ActionSemanticProjection() Matrix

ActionSemanticProjection converts an action vector into the shared semantic basis. It is the executable counterpart of the V9 structure-action matrix.

func GrammarProfileMatrix

func GrammarProfileMatrix() Matrix

GrammarProfileMatrix is the V10 source-language grammar contract. Parser behavior is selected by multiplying a language basis vector by this matrix.

func LexicalRuleMatrix

func LexicalRuleMatrix() Matrix

func LoweringMatrix

func LoweringMatrix() Matrix

LoweringMatrix is target-language x semantic statement class. It is the shared structural rule table used before any target-specific renderer runs.

func MatrixFromRows

func MatrixFromRows(rows [][]float64) (Matrix, error)

func NewMatrix

func NewMatrix(rows, cols int) Matrix

func SemanticFeatureProjection

func SemanticFeatureProjection() Matrix

func (Matrix) At

func (m Matrix) At(row, col int) float64

func (Matrix) BooleanClosure

func (m Matrix) BooleanClosure() (Matrix, error)

BooleanClosure computes the transitive closure with boolean addition and multiplication. It deliberately does not add identity edges.

func (Matrix) Hadamard

func (m Matrix) Hadamard(right Matrix) (Matrix, error)

func (Matrix) Multiply

func (m Matrix) Multiply(right Matrix) (Matrix, error)

func (Matrix) Row

func (m Matrix) Row(row int) Vector

func (Matrix) Set

func (m Matrix) Set(row, col int, value float64)

func (Matrix) Threshold

func (m Matrix) Threshold() Matrix

func (Matrix) Transpose

func (m Matrix) Transpose() Matrix

type Node

type Node struct {
	ID       int
	Vector   Vector
	Text     string
	Symbol   string
	SourceAt int
}

type NormalizedMachineEvent added in v1.0.6

type NormalizedMachineEvent struct {
	Event             string `json:"event"`
	ActionType        string `json:"action_type"`
	ParserState       int    `json:"parser_state"`
	LexState          int    `json:"lex_state"`
	Lookahead         int    `json:"lookahead"`
	Symbol            int    `json:"symbol"`
	ActionList        int    `json:"action_list"`
	ActionIndex       int    `json:"action_index"`
	StackVersion      int    `json:"stack_version"`
	ProductionID      int    `json:"production_id"`
	DynamicPrecedence int    `json:"dynamic_precedence"`
	StackStates       []int  `json:"stack_states,omitempty"`
}

NormalizedMachineEvent is the common, source-independent trace row used by the differential runner. Unknown values are -1; they are compared as unknowns rather than being inferred from diagnostics.

func NormalizeGenericTrace added in v1.0.6

func NormalizeGenericTrace(in []MachineTraceStep) []NormalizedMachineEvent

NormalizeGenericTrace converts the table engine trace into the shared machine-event format. The trace contains one row per parser-machine step; lexical lookahead and the selected action remain explicit fields.

func NormalizeOracleTrace added in v1.0.6

func NormalizeOracleTrace(text string) []NormalizedMachineEvent

NormalizeOracleTrace consumes the temporary runtime's TRACE_INTERNAL and TRACE PARSE lines. It deliberately uses only machine fields present in the trace and never parses source text or diagnostics.

type ParseAction added in v1.0.6

type ParseAction struct {
	Language                                                              string
	ActionListID, Ordinal, Count                                          int
	Kind                                                                  string
	TargetState, LhsSymbolID, ChildCount, DynamicPrecedence, ProductionID int
	Reusable                                                              bool
}

type ParseDispatchEntry added in v1.0.6

type ParseDispatchEntry struct {
	Language                             string
	ParseState, SymbolID                 int
	SymbolName, SymbolKind, DispatchKind string
	NextState, ActionListID              int
}

type ParseNode added in v1.0.6

type ParseNode struct {
	SymbolID, ProductionID, Start, End int
	ParserState                        int
	DynamicPrecedence                  int
	Extra                              bool
	Missing                            bool
	Children                           []*ParseNode
	Fields                             map[int][]int
	// FieldChildren preserves the structural child ordinals selected by the
	// production field map. Fields remains the compatibility symbol view.
	FieldChildren map[int][]int
}

ParseNode is the neutral result of a table reduction. It contains no language-specific AST and is suitable as input to the producer registry.

type ParseStats added in v1.0.6

type ParseStats struct {
	Shift, Reduce, Goto, Accept, Forks                            int
	LastParserState, LastLexState, LastSymbolID, LastActionListID int
	LastProductionID                                              int
}

type ParsedConstructFamily added in v1.0.6

type ParsedConstructFamily string

ParsedConstructFamily is the bounded structured grammar output used before FrontendSemanticFacts. It is a parser fact, not an IR and maps only to existing UAST structures and execution primitives.

const (
	ParsedContainer  ParsedConstructFamily = "CONTAINER"
	ParsedIteration  ParsedConstructFamily = "ITERATION"
	ParsedClosure    ParsedConstructFamily = "CLOSURE_FUNCTION_VALUE"
	ParsedIndexSlice ParsedConstructFamily = "INDEX_SLICE"
)

func ParsedFamilyForStructure added in v1.0.6

func ParsedFamilyForStructure(kind string) ParsedConstructFamily

type ProducerClass added in v1.0.6

type ProducerClass struct {
	Name          string
	Family        ParsedConstructFamily
	RequiredRoles []string
}

ProducerClass is the language-neutral quotient of structured parser facts.

func ProducerClassForEvent added in v1.0.6

func ProducerClassForEvent(e CanonicalSemanticEvent) (ProducerClass, bool)

func ProducerClasses added in v1.0.6

func ProducerClasses() []ProducerClass

type Production added in v1.0.6

type Production struct {
	Language                     string
	ID                           int
	FieldMapSlice, AliasSequence string
}

type RealTables added in v1.0.6

type RealTables struct {
	Parse                []ParseDispatchEntry
	ParseIndex           map[parseDispatchKey]ParseDispatchEntry
	StateRows            map[int][]ParseDispatchEntry
	Actions              []ParseAction
	ActionsByList        map[int][]ParseAction
	Lex                  []LexDispatchEntry
	Modes                []LexModeEntry
	ModeByState          map[int]LexModeEntry
	Predicates           map[string][]LexPredicateInstruction
	CharacterSets        map[string][]LexCharacterRange
	Accepts              map[int]int
	AcceptsByFunction    map[string]int
	Productions          []Production
	FieldMap             []FieldMapEntry
	FieldMapByProduction map[int][]FieldMapEntry
	PrimaryStates        map[int]int
	InitialState         int
	IdentifierSymbol     int
	ErrorSymbol          int
	KeywordCaptureSymbol int
	HasKeywordLexer      bool
	ReservedWords        map[int]map[int]bool
	Symbols              map[int]SymbolMetadata
	Aliases              map[int][]AliasSequenceEntry
	ExternalScanners     *ExternalScannerKernel
	// ExternalValid is the lossless ts_external_scanner_states matrix keyed by
	// the serialized external-lex state.  It is optional for older exports; in
	// that case the parser derives a conservative valid-symbol vector from the
	// dispatch table.
	ExternalValid map[int][]bool
}

func LoadExecutionReadyTables added in v1.0.6

func LoadExecutionReadyTables(dir, language string) (RealTables, error)

func LoadRealTables added in v1.0.6

func LoadRealTables(dir, language string) (RealTables, error)

LoadRealTables reads the lossless execution-ready export and filters one language partition.

type Relation

type Relation int
const (
	Syntax Relation = iota
	Control
	Data
	Binding
	Effect
	RelationCount
)

type SparseMatrix

type SparseMatrix struct {
	Rows int
	Cols int
	// contains filtered or unexported fields
}

SparseMatrix stores only nonzero entries, indexed by row then column. Its coordinates do not depend on dimensions, so graph growth never copies edges. Small semantic/projection matrices deliberately retain the dense Matrix type.

func NewSparseMatrix

func NewSparseMatrix(rows, cols int) SparseMatrix

func (SparseMatrix) At

func (m SparseMatrix) At(r, c int) float64

func (SparseMatrix) BooleanClosure

func (m SparseMatrix) BooleanClosure() (SparseMatrix, error)

BooleanClosure enumerates reachable coordinates, not an n*n temporary array. Dense reachability still requires quadratic output; sparsity is no guarantee that the mathematical result itself is sparse. No identity edges are added.

func (SparseMatrix) Dense

func (m SparseMatrix) Dense() Matrix

func (SparseMatrix) Each

func (m SparseMatrix) Each(f func(int, int, float64))

Each uses stable coordinate order, including in floating-point reductions.

func (*SparseMatrix) Grow

func (m *SparseMatrix) Grow(rows, cols int)

func (SparseMatrix) MarshalJSON

func (m SparseMatrix) MarshalJSON() ([]byte, error)

Sparse serialization is explicit; legacy dense audit fields call Dense at the boundary only. Encoding must never allocate rows*cols implicitly.

func (SparseMatrix) Multiply

func (m SparseMatrix) Multiply(b SparseMatrix) (SparseMatrix, error)

func (SparseMatrix) NonZeros

func (m SparseMatrix) NonZeros() int

func (SparseMatrix) Set

func (m SparseMatrix) Set(r, c int, v float64)

func (SparseMatrix) Sum

func (m SparseMatrix) Sum() float64

func (SparseMatrix) Transpose

func (m SparseMatrix) Transpose() SparseMatrix

func (*SparseMatrix) UnmarshalJSON

func (m *SparseMatrix) UnmarshalJSON(data []byte) error

UnmarshalJSON accepts only the explicit COO representation written above. It rejects duplicate or out-of-range coordinates so a serialized semantic graph cannot silently change meaning while it is loaded.

type SymbolMetadata added in v1.0.6

type SymbolMetadata struct {
	Language                  string
	ID                        int
	Name, DisplayName, Kind   string
	Visible, Named, Supertype bool
}

SymbolMetadata mirrors the generated symbol table. It is runtime metadata, not a second AST: visibility/named/supertype flags are needed when reducing hidden and aliased nodes.

type TargetPlan

type TargetPlan struct {
	Target     string
	Required   Vector
	Supported  Vector
	Missing    Vector
	Unknown    Vector
	RequiredN  int
	SupportedN int
	MissingN   int
	UnknownN   int
}

type TokenClass

type TokenClass int
const (
	TokenIdentifier TokenClass = iota
	TokenNumber
	TokenString
	TokenOperator
	TokenDelimiter
	TokenNewline
	TokenComment
	TokenClassCount
)

type TokenStructure

type TokenStructure struct {
	Classes Matrix
	Pairs   SparseMatrix
	Code    Vector
}

TokenStructure keeps class selection and delimiter incidence separate from semantic guesses. Strings/comments have zero weight in the code projection. Pairing is a syntax operation; multiplication alone cannot infer a grammar.

func AnalyzeTokenStructure

func AnalyzeTokenStructure(tokens []Lexeme) (TokenStructure, error)

type Vector

type Vector []float64

func ActionSemantic

func ActionSemantic(action Vector) (Vector, error)

func AnalyzeExpression

func AnalyzeExpression(source, text string) Vector

AnalyzeExpression creates a shared semantic vector without rewriting text. It is quote-aware and language-neutral; source-specific syntax is mapped to the same semantic axes. It is a structural scanner, not the translation engine and not a claim of full parsing.

func Basis

func Basis(size, index int) Vector

func FeatureRequirements

func FeatureRequirements(semantic Vector) (Vector, error)

func GrammarProfile

func GrammarProfile(language string) (Vector, error)

func LexicalAxesFor

func LexicalAxesFor(source string, class TokenClass, text string) Vector

func MissingLowerings

func MissingLowerings(target string, required Vector) (Vector, error)

func StatementRequirements

func StatementRequirements(vectors []Vector) (Vector, error)

func (Vector) Dot

func (v Vector) Dot(right Vector) (float64, error)

func (Vector) Or

func (v Vector) Or(right Vector) (Vector, error)

Jump to

Keyboard shortcuts

? : This menu
/ : Search site
f or F : Jump to
y or Y : Canonical URL