models

package
v0.1.47 Latest Latest
Warning

This package is not in the latest version of its module.

Go to latest
Published: Sep 1, 2026 License: BSD-2-Clause Imports: 9 Imported by: 0

Documentation

Index

Constants

This section is empty.

Variables

This section is empty.

Functions

func CalcRefHash added in v0.1.39

func CalcRefHash(outValue *string, keyvals ...interface{})

CalcRefHash exposes the shared SHA1 hashing used for deterministic document ids to other packages (the Elasticsearch writer uses it to key the file<->leak reference documents by file_id + leak_id).

Types

type Credential

type Credential struct {
	ID     uint `json:"id" gorm:"primarykey"`
	FileID uint `json:"file_id" gorm:"index:idx_cred"`

	Rule string    `json:"rule"`
	Time time.Time `json:"time"`

	UserDomain string `json:"user_domain"`
	Username   string `json:"username"`
	Password   string `json:"password"`

	CPF string `json:"cpf"`

	Url       string `json:"url"`
	UrlDomain string `json:"url_domain"`

	Severity int     `json:"severity"`
	Entropy  float32 `json:"entropy"`

	NearText string `json:"near_text"`
}

func (Credential) LeakDoc added in v0.1.39

func (cred Credential) LeakDoc() map[string]interface{}

func (Credential) LeakID added in v0.1.39

func (cred Credential) LeakID() string

func (Credential) LeakType added in v0.1.39

func (cred Credential) LeakType() string

func (Credential) MarshalJSON

func (cred Credential) MarshalJSON() ([]byte, error)

Custom Marshaller for Credential

func (Credential) RefDoc added in v0.1.39

func (cred Credential) RefDoc() map[string]interface{}

func (*Credential) Sanitize added in v0.1.30

func (cred *Credential) Sanitize()

Sanitize removes null bytes and invalid UTF-8 sequences from all string fields

type Document added in v0.1.36

type Document struct {
	ID     uint `json:"id" gorm:"primarykey"`
	FileID uint `json:"file_id" gorm:"index:idx_document"`

	Time time.Time `json:"time"`

	Raw    string `json:"raw"`
	Number string `json:"number"`
	IsCPF  bool   `json:"is_cpf"`
	IsCNPJ bool   `json:"is_cnpj"`

	Source   string `json:"source"`
	FileName string `json:"file_name"`
	Line     string `json:"line"`

	NearText string `json:"near_text"`
}

func (Document) LeakDoc added in v0.1.39

func (d Document) LeakDoc() map[string]interface{}

func (Document) LeakID added in v0.1.39

func (d Document) LeakID() string

func (Document) LeakType added in v0.1.39

func (d Document) LeakType() string

func (Document) MarshalJSON added in v0.1.36

func (d Document) MarshalJSON() ([]byte, error)

Custom Marshaller for Document

func (Document) RefDoc added in v0.1.39

func (d Document) RefDoc() map[string]interface{}

func (*Document) Sanitize added in v0.1.36

func (d *Document) Sanitize()

Sanitize removes null bytes and invalid UTF-8 sequences from all string fields

type Email

type Email struct {
	ID     uint `json:"id" gorm:"primarykey"`
	FileID uint `json:"file_id" gorm:"index:idx_email"`

	Time time.Time `json:"time"`

	Domain string `json:"domain"`
	Email  string `json:"email"`

	NearText string `json:"near_text"`
}

func (Email) LeakDoc added in v0.1.39

func (eml Email) LeakDoc() map[string]interface{}

func (Email) LeakID added in v0.1.39

func (eml Email) LeakID() string

func (Email) LeakType added in v0.1.39

func (eml Email) LeakType() string

func (Email) MarshalJSON

func (eml Email) MarshalJSON() ([]byte, error)

Custom Marshaller for URL

func (Email) RefDoc added in v0.1.39

func (eml Email) RefDoc() map[string]interface{}

func (*Email) Sanitize added in v0.1.30

func (eml *Email) Sanitize()

Sanitize removes null bytes and invalid UTF-8 sequences from all string fields

type File

type File struct {
	ID uint `json:"id" gorm:"primarykey"`

	Provider  string    `json:"provider"` //IntelX, ...
	FilePath  string    `json:"file_path"`
	FileName  string    `json:"file_name"`
	Name      string    `json:"name"`
	Date      time.Time `json:"date"`
	Bucket    string    `json:"bucket"`
	MediaType string    `json:"media_type"`
	IndexedAt time.Time `json:"indexed_at"`

	Size        uint   `json:"size"`
	ProviderId  string `json:"provider_id"`
	MIMEType    string `json:"mime_type"`
	Fingerprint string `json:"fingerprint" gorm:"unique;not null"`

	Content string `json:"content"`

	// Failed flag set if the result should be considered failed
	Failed       bool   `json:"failed"`
	FailedReason string `json:"failed_reason"`

	Credentials []Credential `json:"credentials" gorm:"constraint:OnDelete:CASCADE"`
	Emails      []Email      `json:"emails" gorm:"constraint:OnDelete:CASCADE"`
	URLs        []URL        `json:"urls" gorm:"constraint:OnDelete:CASCADE"`
	Phones      []Phone      `json:"phones" gorm:"constraint:OnDelete:CASCADE"`
	Documents   []Document   `json:"documents" gorm:"constraint:OnDelete:CASCADE"`
}

Name,Date,Bucket,Media,Content Type,Size,System ID

func (*File) BeforeCreate added in v0.1.42

func (file *File) BeforeCreate(tx *gorm.DB) (err error)

BeforeCreate targets the upsert at the fingerprint instead of the primary key. The connection-level OnConflict set by DbWriter has no Columns, so gorm resolves it to ON CONFLICT ("id") — which never fires for a new row and lets the insert hit uni_files_fingerprint instead. The fingerprint is the real identity of a file here (the Elasticsearch writer uses it as the document _id), so a re-import of the same content must update the existing row.

Declared on the model rather than on the connection so the cascaded inserts into credentials/urls/emails/phones/documents keep their own conflict target; those tables have no fingerprint column.

func (File) Clone added in v0.1.1

func (file File) Clone() *File

func (File) MarshalJSON

func (file File) MarshalJSON() ([]byte, error)

Custom Marshaller for File

func (*File) Sanitize added in v0.1.30

func (file *File) Sanitize()

Sanitize removes null bytes and invalid UTF-8 sequences from all string fields

type Finding

type Finding struct {
	// Rule is the name of the rule that was matched
	RuleID      string
	Description string

	StartLine   int
	EndLine     int
	StartColumn int
	EndColumn   int

	Line string `json:"-"`

	Match string

	// Secret contains the full content of what is matched in
	// the tree-sitter query.
	Secret string

	// File is the name of the file containing the finding
	File        string
	SymlinkFile string
	Commit      string
	Link        string `json:",omitempty"`

	// Entropy is the shannon entropy of Value
	Entropy float32

	Author  string
	Date    string
	Message string
	Tags    []string

	// unique identifier
	Fingerprint string

	Credential Credential
	Email      Email
	Url        URL
	Phone      Phone
	Document   Document
}

Finding contains information about strings that have been captured by a tree-sitter query.

type LeakIndexable added in v0.1.39

type LeakIndexable interface {
	LeakID() string
	LeakType() string
	LeakDoc() map[string]interface{}
	RefDoc() map[string]interface{}
}

LeakIndexable is implemented by every leak type (Credential, URL, Email, Phone, Document). It powers the restructured Elasticsearch writer, which splits each leak into three concerns:

  • LeakID: a content-only fingerprint used as the _id in the global, deduplicated leak index. It must NOT depend on the file it was found in nor on a timestamp, so the same leak seen in different files/imports collapses to a single document.
  • LeakDoc: the intrinsic leak fields (the value itself). This is all that gets stored in the leak index — no file reference.
  • RefDoc: the occurrence-specific context (near_text, line, source...) that belongs to the monthly file<->leak reference index, not to the leak itself.
  • LeakType: a discriminator string for the reference index.

type Phone added in v0.1.36

type Phone struct {
	ID     uint `json:"id" gorm:"primarykey"`
	FileID uint `json:"file_id" gorm:"index:idx_phone"`

	Time time.Time `json:"time"`

	Country string `json:"country"`
	Raw     string `json:"raw"`
	Phone   string `json:"phone"`

	Source   string `json:"source"`
	FileName string `json:"file_name"`
	Line     string `json:"line"`

	NearText string `json:"near_text"`
}

func (Phone) LeakDoc added in v0.1.39

func (p Phone) LeakDoc() map[string]interface{}

func (Phone) LeakID added in v0.1.39

func (p Phone) LeakID() string

func (Phone) LeakType added in v0.1.39

func (p Phone) LeakType() string

func (Phone) MarshalJSON added in v0.1.36

func (p Phone) MarshalJSON() ([]byte, error)

Custom Marshaller for Phone

func (Phone) RefDoc added in v0.1.39

func (p Phone) RefDoc() map[string]interface{}

func (*Phone) Sanitize added in v0.1.36

func (p *Phone) Sanitize()

Sanitize removes null bytes and invalid UTF-8 sequences from all string fields

type URL

type URL struct {
	ID     uint `json:"id" gorm:"primarykey"`
	FileID uint `json:"file_id" gorm:"index:idx_url"`

	Time time.Time `json:"time"`

	Domain string `json:"domain"`
	Url    string `json:"url"`

	NearText string `json:"near_text"`
}

func (URL) LeakDoc added in v0.1.39

func (u URL) LeakDoc() map[string]interface{}

func (URL) LeakID added in v0.1.39

func (u URL) LeakID() string

func (URL) LeakType added in v0.1.39

func (u URL) LeakType() string

func (URL) MarshalJSON

func (u URL) MarshalJSON() ([]byte, error)

Custom Marshaller for URL

func (URL) RefDoc added in v0.1.39

func (u URL) RefDoc() map[string]interface{}

func (*URL) Sanitize added in v0.1.30

func (u *URL) Sanitize()

Sanitize removes null bytes and invalid UTF-8 sequences from all string fields

Jump to

Keyboard shortcuts

? : This menu
/ : Search site
f or F : Jump to
y or Y : Canonical URL