Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 14 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -283,3 +283,17 @@ and observability shape. Where the shipped implementation deliberately
diverged from an original design decision, the superseded passage is
marked as historical in place — the supersession notes, MIGRATION.md,
and the package godocs describe the behaviour that actually ships.

### Schema conformance

The XML models in `internal/feed/xml` and `internal/api/xml` are written
by hand. `internal/schemacheck` keeps them honest against the
[oddsfeedschema](https://github.com/oddin-gg/oddsfeedschema) XSDs, which
stay the single source of truth: the test downloads the schema from
GitHub at test time (ref `main` by default, `ODDSFEEDSCHEMA_REF` to pin
a branch, tag or commit, `ODDSFEEDSCHEMA_DIR` to use a local checkout).
Every attribute and element the schema declares must have a Go field and
every Go tag must exist in the schema. Known deviations live in a ledger
in the test with the reason for each; a new deviation, or a ledger entry
the code has outgrown, fails `go test ./...`. Without network the test
skips locally and fails in CI.
131 changes: 131 additions & 0 deletions internal/schemacheck/compare.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,131 @@
package schemacheck

import (
"fmt"
"sort"
"strings"
)

// finding is one slot present on exactly one side.
type finding struct {
// kind is one of the *Only constants below.
kind string
// path is "/root/child/…@attr" for attributes, "/root/child/…" for
// elements — the key the known-drift ledger is written in.
path string
}

const (
// schemaOnlyAttr: the XSD declares the attribute, the Go model has no
// field for it — the SDK drops it on decode.
schemaOnlyAttr = "schema-only attribute"
// schemaOnlyElem: the XSD declares the element, the Go model has no
// field for it — the SDK drops the whole subtree.
schemaOnlyElem = "schema-only element"
// sdkOnlyAttr: the Go model decodes an attribute the XSD does not
// declare — dead field, or the schema is behind the producer.
sdkOnlyAttr = "sdk-only attribute"
// sdkOnlyElem: as sdkOnlyAttr for a child element.
sdkOnlyElem = "sdk-only element"
// schemaOnlyText: the XSD gives the element text content (a simple
// type, simpleContent, or mixed) and the Go model has no chardata
// field for it — the text is dropped on decode. Path suffix "#text".
schemaOnlyText = "schema-only text"
// sdkOnlyText: the Go model reads text content the XSD does not
// declare.
sdkOnlyText = "sdk-only text"
)

func (f finding) String() string { return f.kind + " " + f.path }

// compare walks both shapes in lockstep from path and reports every slot
// only one side has. Elements both sides have are recursed into; an
// element only one side has is reported once, not expanded.
func compare(path string, schema, sdk *shape) []finding {
var out []finding
switch {
case schema.text && !sdk.text:
out = append(out, finding{schemaOnlyText, path + "#text"})
case sdk.text && !schema.text:
out = append(out, finding{sdkOnlyText, path + "#text"})
}
for _, a := range schema.sortedAttrs() {
if !sdk.attrs[a] {
out = append(out, finding{schemaOnlyAttr, path + "@" + a})
}
}
for _, a := range sdk.sortedAttrs() {
if !schema.attrs[a] {
out = append(out, finding{sdkOnlyAttr, path + "@" + a})
}
}
for _, e := range schema.sortedElems() {
child, ok := sdk.elems[e]
if !ok {
out = append(out, finding{schemaOnlyElem, path + "/" + e})
continue
}
out = append(out, compare(path+"/"+e, schema.elems[e], child)...)
}
for _, e := range sdk.sortedElems() {
if _, ok := schema.elems[e]; !ok {
out = append(out, finding{sdkOnlyElem, path + "/" + e})
}
}
return out
}

// ledgerEntry is one acknowledged deviation: the finding's path plus the
// reason it is tolerated. The tests fail on any finding that is not in
// the ledger AND on any ledger entry that no longer matches a finding, so
// the ledger is always exactly the current drift.
type ledgerEntry struct {
path string
reason string
}

// matches reports whether the entry covers path. An entry starting with
// "**" matches by suffix — "**/sport@ref_id" covers that attribute on
// every <sport> wherever it is nested — so one line can acknowledge a
// deviation on a shared Go type instead of one per context.
func (l ledgerEntry) matches(path string) bool {
if rest, ok := strings.CutPrefix(l.path, "**"); ok {
return strings.HasSuffix(path, rest)
}
return l.path == path
}

// reconcile splits findings into unexpected ones (not in the ledger) and
// stale ledger entries (nothing matched them). Both lists are sorted.
func reconcile(findings []finding, ledger []ledgerEntry) (unexpected []finding, stale []ledgerEntry) {
used := make([]bool, len(ledger))
for _, f := range findings {
matched := false
for i, l := range ledger {
if l.matches(f.path) {
used[i] = true
matched = true
}
}
if !matched {
unexpected = append(unexpected, f)
}
}
for i, l := range ledger {
if !used[i] {
stale = append(stale, l)
}
}
sort.Slice(unexpected, func(i, j int) bool { return unexpected[i].path < unexpected[j].path })
sort.Slice(stale, func(i, j int) bool { return stale[i].path < stale[j].path })
return unexpected, stale
}

// describe renders findings one per line for a failure message.
func describe(fs []finding) string {
s := ""
for _, f := range fs {
s += fmt.Sprintf("\n %s", f)
}
return s
}
147 changes: 147 additions & 0 deletions internal/schemacheck/gomodel.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,147 @@
package schemacheck

import (
"encoding"
"encoding/xml"
"fmt"
"reflect"
"strings"
"time"
)

// goShape reduces a Go XML model type to a shape by reading its
// encoding/xml struct tags the way encoding/xml itself would:
//
// - `xml:"name,attr"` is an attribute;
// - `xml:"name"` (or an untagged exported field) is a child element,
// recursed into unless the field's type decodes as text;
// - `xml:"a>b>c"` is a chain of nested elements;
// - `xml:",chardata"`, `xml:",cdata"` and `xml:",innerxml"` mark text;
// - `xml:"-"`, XMLName and `xml:",any"` are skipped;
// - anonymous embedded structs are flattened into the parent;
// - pointers and slices are looked through.
//
// A type decodes as text when it is a basic kind, implements
// encoding.TextUnmarshaler or xml.Unmarshaler, or is time.Time — the
// SDK's utils.Date / DateTime / Timestamp are time.Time aliases with
// UnmarshalText, so they stop the recursion instead of exposing
// time.Time's internals as elements.
func goShape(t reflect.Type) (*shape, error) {
t = deref(t)
if t.Kind() != reflect.Struct || isTextType(t) {
leaf := newShape()
leaf.text = true
return leaf, nil
}
out := newShape()
if err := addStructFields(out, t, t.Name()); err != nil {
return nil, err
}
return out, nil
}

var (
textUnmarshaler = reflect.TypeFor[encoding.TextUnmarshaler]()
xmlUnmarshaler = reflect.TypeFor[xml.Unmarshaler]()
timeType = reflect.TypeFor[time.Time]()
xmlNameType = reflect.TypeFor[xml.Name]()
)

func deref(t reflect.Type) reflect.Type {
for t.Kind() == reflect.Pointer || t.Kind() == reflect.Slice || t.Kind() == reflect.Array {
t = t.Elem()
}
return t
}

func isTextType(t reflect.Type) bool {
if t == timeType || t.ConvertibleTo(timeType) && t.Kind() == reflect.Struct {
return true
}
pt := reflect.PointerTo(t)
return pt.Implements(textUnmarshaler) || pt.Implements(xmlUnmarshaler) ||
t.Implements(textUnmarshaler) || t.Implements(xmlUnmarshaler)
}

func addStructFields(out *shape, t reflect.Type, where string) error {
for i := 0; i < t.NumField(); i++ {
f := t.Field(i)
if !f.IsExported() && !f.Anonymous {
continue
}
if f.Type == xmlNameType {
continue
}
tag, hasTag := f.Tag.Lookup("xml")
if tag == "-" {
continue
}
name, opts, _ := strings.Cut(tag, ",")
if f.Anonymous && !hasTag {
// Embedded struct: encoding/xml promotes its fields.
et := deref(f.Type)
if et.Kind() == reflect.Struct && !isTextType(et) {
if err := addStructFields(out, et, where+"."+f.Name); err != nil {
return err
}
continue
}
}
switch {
case hasOpt(opts, "attr"):
if name == "" {
name = f.Name
}
out.attrs[name] = true
case hasOpt(opts, "chardata"), hasOpt(opts, "cdata"), hasOpt(opts, "innerxml"):
out.text = true
case hasOpt(opts, "any"), hasOpt(opts, "comment"):
// Catch-alls carry no name to compare.
default:
if name == "" {
name = f.Name
}
child, err := goShape(f.Type)
if err != nil {
return fmt.Errorf("%s.%s: %w", where, f.Name, err)
}
if err := out.addPath(strings.Split(name, ">"), child); err != nil {
return fmt.Errorf("%s.%s: %w", where, f.Name, err)
}
}
}
return nil
}

func hasOpt(opts, want string) bool {
for _, o := range strings.Split(opts, ",") {
if o == want {
return true
}
}
return false
}

// addPath attaches child under a `>`-separated element path, creating
// the intermediate (attribute-less) elements encoding/xml implies.
func (s *shape) addPath(path []string, child *shape) error {
cur := s
for _, seg := range path[:len(path)-1] {
if seg == "" {
return fmt.Errorf("empty segment in xml path %q", strings.Join(path, ">"))
}
next, ok := cur.elems[seg]
if !ok {
next = newShape()
cur.elems[seg] = next
}
cur = next
}
leaf := path[len(path)-1]
if existing, ok := cur.elems[leaf]; ok {
existing.merge(child)
} else {
cur.elems[leaf] = child
}
return nil
}
Loading
Loading