Documentation
¶
Overview ¶
Package xml is a pull reader for XML documents shaped like the parts of an Office Open XML package: one encoding, no DTD, deep repetition of a few element shapes, and a caller that wants a handful of the elements and none of the rest.
It is not a replacement for encoding/xml. It does not resolve namespaces, does not decode into arbitrary Go values, and does not accept a DOCTYPE. In exchange the reader allocates nothing in steady state: every name, attribute and text it returns is a slice into its own buffer, valid until the next call that advances the reader, and a caller copies only what it keeps.
Reading ¶
Next advances one token at a time. On a StartElement the caller reads the name and the attributes it wants, then either descends into the children, reads the text content with ElementText, captures the whole subtree with RawElement, or discards it with Skip. NextChild walks the direct children of an element and skips whatever the caller did not consume, so a decoder for one element is a loop over its children with a switch on the name:
el := r.Element()
for {
ok, err := r.NextChild(el)
if err != nil || !ok {
return err
}
switch {
case r.NameIs("v"):
text, err := r.ElementText()
...
case r.NameIs("is"):
err = inline.DecodeXMLFrom(r)
}
}
Names and namespaces ¶
Names are matched as written, prefix included: "w:p" is "w:p". Office XML generators emit canonical prefixes, so matching the qualified name is enough for that input and costs nothing. The xmlns attributes are visible through NextAttr for a caller that needs to resolve them.
Text ¶
Text and Attr return the bytes as written, entities included. A Value knows whether it holds any, and unescapes into a caller-owned buffer, so the common case of text with no ampersand is a slice and no copy.
Bounds ¶
The buffer grows only to hold one token, or one capture, and never past Options.MaxBufferBytes; nesting stops at Options.MaxDepth. A document that needs more is refused, not accommodated.
Example ¶
package main
import (
"fmt"
"strings"
"github.com/shibukawa/tinygodriver/encoding/xml"
)
// A cell of a worksheet, read by hand: the attributes it wants, the child
// it wants, and nothing kept as a string.
type cell struct {
Ref string
Style int64
Shared bool
Value float64
Index int64
}
func (c *cell) DecodeXMLFrom(r *xml.Reader) error {
*c = cell{} // the caller reuses one cell, and an absent attribute must not keep the last value
ref, _ := r.Attr("r")
c.Ref = ref.String()
if s, ok := r.Attr("s"); ok {
c.Style, _ = s.Int()
}
if t, ok := r.Attr("t"); ok {
c.Shared = t.Equal("s")
}
for name := range r.Children(r.Element()) {
if !xml.Equal(name, "v") {
continue
}
v, err := r.ElementText()
if err != nil {
return err
}
if c.Shared {
c.Index, err = v.Int()
} else {
c.Value, err = v.Float()
}
if err != nil {
return err
}
}
return r.Err()
}
func main() {
const sheet = `<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<worksheet xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">
<dimension ref="A1:B2"/>
<sheetData>
<row r="1"><c r="A1" t="s"><v>0</v></c><c r="B1" s="2"><v>12.5</v></c></row>
<row r="2"><c r="A2"><f>B1*2</f><v>25</v></c></row>
</sheetData>
<mergeCells count="1"><mergeCell ref="A1:B1"/></mergeCells>
</worksheet>`
r := xml.NewReader(strings.NewReader(sheet), xml.Options{})
var c cell
for k := range r.Tokens() {
if k != xml.StartElement {
continue
}
switch {
case r.NameIs("sheetData"):
for range r.Children(r.Element()) { // each <row>
for range r.Children(r.Element()) { // each <c>
if err := r.Decode(&c); err != nil {
panic(err)
}
if c.Shared {
fmt.Printf("%s style=%d shared string #%d\n", c.Ref, c.Style, c.Index)
} else {
fmt.Printf("%s style=%d value=%g\n", c.Ref, c.Style, c.Value)
}
}
}
case r.NameIs("mergeCell"):
ref, _ := r.Attr("ref")
fmt.Printf("merged %s\n", ref)
}
}
if err := r.Err(); err != nil {
panic(err)
}
}
Output: A1 style=0 shared string #0 B1 style=2 value=12.5 A2 style=0 value=25 merged A1:B1
Index ¶
- Constants
- Variables
- func Equal(b []byte, s string) bool
- func Unescape(dst, src []byte) []byte
- type Decodable
- type Element
- type Kind
- type Options
- type Reader
- func (r *Reader) Attr(name string) (Value, bool)
- func (r *Reader) Children(e Element) iter.Seq[[]byte]
- func (r *Reader) Decode(d Decodable) error
- func (r *Reader) Depth() int
- func (r *Reader) Element() Element
- func (r *Reader) ElementText() (Value, error)
- func (r *Reader) Err() error
- func (r *Reader) Kind() Kind
- func (r *Reader) LocalName() []byte
- func (r *Reader) LookupNamespace(prefix []byte) (string, bool)
- func (r *Reader) Name() []byte
- func (r *Reader) NameIs(s string) bool
- func (r *Reader) Namespace() string
- func (r *Reader) Next() (Kind, error)
- func (r *Reader) NextAttr() (name []byte, value Value, ok bool)
- func (r *Reader) NextChild(e Element) (bool, error)
- func (r *Reader) Offset() int64
- func (r *Reader) Prefix() []byte
- func (r *Reader) RawElement() ([]byte, error)
- func (r *Reader) Reset(src io.Reader)
- func (r *Reader) ResetBytes(data []byte)
- func (r *Reader) Skip() error
- func (r *Reader) Text() Value
- func (r *Reader) Tokens() iter.Seq[Kind]
- type SyntaxError
- type Value
- func (v Value) AppendTo(dst []byte) []byte
- func (v Value) Bool() (bool, error)
- func (v Value) Equal(s string) bool
- func (v Value) EqualFold(s string) bool
- func (v Value) Float() (float64, error)
- func (v Value) HasEntities() bool
- func (v Value) Int() (int64, error)
- func (v Value) String() string
- func (v Value) Uint() (uint64, error)
Examples ¶
Constants ¶
const XMLNamespace = "http://www.w3.org/XML/1998/namespace"
XMLNamespace is the namespace the xml prefix is bound to without a declaration.
Variables ¶
var ( // ErrTruncated is returned when the input ends inside a token or inside an // open element. ErrTruncated = errors.New("xml: unexpected end of input") // ErrTooLarge is returned when a token or a capture does not fit in // Options.MaxBufferBytes. ErrTooLarge = errors.New("xml: token exceeds MaxBufferBytes") // ErrTooDeep is returned when nesting exceeds Options.MaxDepth. ErrTooDeep = errors.New("xml: nesting exceeds MaxDepth") // ErrDoctype is returned for a DOCTYPE unless Options.AllowDoctype is set. ErrDoctype = errors.New("xml: DOCTYPE refused") // ErrEncoding is returned when the XML declaration names an encoding other // than UTF-8. ErrEncoding = errors.New("xml: declared encoding is not UTF-8") // ErrNotStart is returned by the element-level calls when the reader is // not positioned on a StartElement. ErrNotStart = errors.New("xml: not positioned on a start element") )
Functions ¶
func Equal ¶
Equal reports whether b holds exactly the bytes of s. It allocates on neither compiler: the Go compiler elides the conversion in string(b) == s, but TinyGo does not, and copies b for every comparison, every switch on string(b) and every map index by string(b). Compare names through this, NameIs or Value.Equal when the binary is a TinyGo one.
Types ¶
type Decodable ¶
Decodable is implemented by a type that reads itself from an element. The reader is positioned on the element's StartElement when the method is called, and the method returns with the reader on the matching EndElement; NextChild and ElementText both end there.
type Element ¶
type Element struct {
// contains filtered or unexported fields
}
Element identifies an element the reader has entered, for NextChild. It is a value, held on the caller's stack.
type Kind ¶
type Kind uint8
Kind is the kind of token the reader is positioned on.
const ( // None is the kind before the first Next and after an error. None Kind = iota // StartElement is an opening tag. A self-closing tag is a StartElement // followed by an EndElement, as encoding/xml reports it. StartElement // EndElement is a closing tag. EndElement // Text is character data between tags, as written: entities are not // decoded and whitespace is not trimmed. Text // CData is the content of a CDATA section. It contains no entities. CData // Comment is the body of a comment. Comment // ProcInst is a processing instruction, including the XML declaration, // whose target is "xml". ProcInst // Directive is a DOCTYPE, reported only when Options.AllowDoctype is set. Directive // EOF is reported once at the end of a well-formed document. EOF )
type Options ¶
type Options struct {
// BufferSize is the initial buffer, 64 KiB by default. It grows only when
// one token does not fit, and only up to MaxBufferBytes.
BufferSize int
// MaxBufferBytes bounds the buffer, and so the largest single token, the
// largest ElementText and the largest RawElement. 1 MiB by default.
MaxBufferBytes int
// MaxDepth bounds element nesting, 1024 by default.
MaxDepth int
// AllowDoctype reports a DOCTYPE as a Directive instead of refusing it.
// An internal subset is refused either way.
AllowDoctype bool
}
Options bounds a Reader. The zero value selects the defaults noted on each field.
type Reader ¶
type Reader struct {
// contains filtered or unexported fields
}
Reader reads XML tokens from an io.Reader or from a byte slice.
Every slice a Reader returns aliases its buffer and is valid until the next call that advances the reader: Next, Skip, NextChild, ElementText, RawElement and Decode. Callers that keep a name or a value copy it.
A Reader is not safe for concurrent use. Reuse one across documents with Reset rather than allocating one per document.
func NewBytesReader ¶
NewBytesReader returns a Reader over data, which it neither copies nor modifies. BufferSize and MaxBufferBytes do not apply: the buffer is data.
func (*Reader) Attr ¶
Attr returns the value of the named attribute of the current StartElement, as written. The start tag was indexed as it was scanned, so this is a comparison against each of the element's few names, not a rescan.
func (*Reader) Children ¶
Children iterates the direct child elements of e, yielding each one's qualified name with the reader positioned on its StartElement. It is NextChild as a range loop, with the same skipping of whatever the body does not consume, and it ends silently on an error:
for name := range r.Children(r.Element()) {
switch {
case xml.Equal(name, "v"):
text, err := r.ElementText()
...
}
}
The name is compared through Equal rather than string(name) because the conversion allocates under TinyGo; see Equal.
if err := r.Err(); err != nil {
return err
}
A break leaves the reader on the child's StartElement; the element e is then still open, and NextChild or Skip finish it.
func (*Reader) Decode ¶
Decode reads the current element into d and checks that d consumed exactly that element.
func (*Reader) Depth ¶
Depth reports how many elements are open. On a StartElement it counts that element; on its EndElement it no longer does.
func (*Reader) Element ¶
Element returns a handle to the element the reader is in, for NextChild. On a StartElement that is the element itself; anywhere else it is the innermost open element.
func (*Reader) ElementText ¶
ElementText advances from a StartElement to its EndElement and returns the element's own character data with entities decoded. Text in child elements is not included; the children are skipped. When the text is one run with no entity the result aliases the buffer; otherwise it is assembled in a scratch buffer the next call reuses.
func (*Reader) Err ¶
Err returns the error that stopped the reader, or nil. The iterators end silently on an error, so a loop over Tokens or Children checks Err after the loop, as a bufio.Scanner caller checks Scanner.Err.
func (*Reader) LocalName ¶
LocalName returns the name after the prefix, or the whole name when there is none.
func (*Reader) LookupNamespace ¶
LookupNamespace returns the namespace bound to prefix at the current position. An empty prefix looks up the default namespace. The xml prefix is always bound.
func (*Reader) Name ¶
Name returns the qualified name of a StartElement or EndElement, or the target of a ProcInst, as written.
func (*Reader) Namespace ¶
Namespace returns the namespace the current element's name is in: the one its prefix is bound to, or the default namespace when it has none. It is empty when nothing binds the prefix. Names are still matched as written; this is for the caller that must tell one vocabulary from another under the same prefix, or an unprefixed name under a default namespace from one without.
func (*Reader) Next ¶
Next advances to the next token and reports its kind. After an error every later call returns the same error.
func (*Reader) NextAttr ¶
NextAttr returns the attributes of the current StartElement in document order, one per call, and reports false after the last. It restarts on each new token.
func (*Reader) NextChild ¶
NextChild advances to the next direct child element of e and reports whether there was one. Anything between children is passed over, and a child the caller did not consume is skipped, so a loop over NextChild ends on the EndElement of e however much of each child it read.
func (*Reader) RawElement ¶
RawElement advances from a StartElement to its EndElement and returns the bytes of the whole element, tags included, as written. The result aliases the buffer, so the element must fit in MaxBufferBytes.
func (*Reader) ResetBytes ¶
ResetBytes points the Reader at data, keeping its options. Slices returned before the call still alias the old input.
func (*Reader) Skip ¶
Skip advances from a StartElement to its matching EndElement, reading nothing in between. It scans the subtree raw, tracking only tags, quotes and depth, so it neither indexes attributes nor checks that end tags match inside what it skips; the reader is left on the end tag with Name set, exactly as Next would leave it.
func (*Reader) Text ¶
Text returns the body of a Text, CData, Comment, ProcInst or Directive token as written. For Text the entities are still encoded; see Value.
func (*Reader) Tokens ¶
Tokens iterates the remaining tokens, ending at EOF or at an error. The reader is the loop variable's context: inside the body Name, Attr, Text and the element-level calls all refer to the token just yielded.
for k := range r.Tokens() {
if k == xml.StartElement && r.NameIs("sheetData") {
r.Decode(&sheet)
}
}
if err := r.Err(); err != nil {
return err
}
type SyntaxError ¶
SyntaxError reports malformed input at a byte offset from the start of the document.
func (*SyntaxError) Error ¶
func (e *SyntaxError) Error() string
type Value ¶
type Value []byte
Value is attribute or text content as it appears in the document, entities included. Its methods decode on demand, so content with no ampersand, which is nearly all of it, is never copied. A Value aliases the reader's buffer and is valid until the reader advances.
Decoding replaces the five predefined entities and numeric character references, and normalizes line ends as XML requires of text: "\r\n" and a lone "\r" both become "\n". The further whitespace normalization XML applies to attribute values, tabs and newlines to spaces, is not done; Office writers escape such characters as references, which decoding leaves as the characters they name.
func (Value) Bool ¶
Bool parses an xsd:boolean: "1", "0", "true" or "false". Office attributes use the digits.
func (Value) EqualFold ¶
EqualFold reports whether the decoded content equals s under ASCII case folding.
func (Value) Float ¶
Float parses a floating point number. A plain decimal of at most 15 significant digits and at most 22 fraction digits, which is every number a spreadsheet writer emits for a cell, is converted by one exact division and is bit-identical to strconv's answer; anything else goes to strconv.
func (Value) HasEntities ¶
HasEntities reports whether decoding would change the bytes: an entity or character reference, or a carriage return.
func (Value) Int ¶
Int parses a decimal integer. Numbers carry no entities, so the raw bytes are parsed directly.