diff --git a/CONTEXT-MAP.md b/CONTEXT-MAP.md index bc13f84f2..af93fdfbd 100644 --- a/CONTEXT-MAP.md +++ b/CONTEXT-MAP.md @@ -4,6 +4,9 @@ - [EQL](./packages/eql/CONTEXT.md) — defines the PostgreSQL objects that store and query encrypted values. +- [Stack Encrypt](./packages/stack-encrypt/CONTEXT.md) — encrypts values under + per-value ZeroKMS data keys and derives searchable terms from them, for Rust + and, through a WASI guest, Go. ## Relationships @@ -12,3 +15,6 @@ - **EQL → ORM adapters**: EQL defines durable encrypted column domains and disposable query machinery; adapters create application columns and derived search indexes against that surface. +- **Stack Encrypt → EQL**: Stack Encrypt seals values and renders the ZeroKMS + descriptor; EQL's `eql-bindings` transcodes those into EQL payloads, and its + `Identifier` (a table and a column) is a two-segment Stack Encrypt `Label`. diff --git a/Cargo.lock b/Cargo.lock index 937d64f14..ba86403c9 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3274,7 +3274,7 @@ dependencies = [ [[package]] name = "stack-encrypt" -version = "0.1.0" +version = "0.2.0" dependencies = [ "base64ct", "cllw-ore", @@ -3299,10 +3299,11 @@ dependencies = [ [[package]] name = "stack-encrypt-derive" -version = "0.1.0" +version = "0.2.0" dependencies = [ "proc-macro2", "quote", + "serde_json", "stack-encrypt", "syn 3.0.3", "tokio", diff --git a/docs/plans/stack-encrypt-go-bindings.md b/docs/plans/stack-encrypt-go-bindings.md index fa00c797e..2045d1bb9 100644 --- a/docs/plans/stack-encrypt-go-bindings.md +++ b/docs/plans/stack-encrypt-go-bindings.md @@ -351,8 +351,8 @@ way: field when the record is sealed with `encrypt_into` (no caller context). A list is what the Rust derive produces when it *extends* every field's context with the caller's: `encrypt_into_with_context(row, 7u64)` seals - `users/email` under `("users/email", 7u64)`, descriptor - `users/email|7u64`, and the plan spells that as `["users/email", 7u64]` + the `users.email` field under `(("users", "email"), 7u64)`, descriptor + `(users/email)/7u64`, and the plan spells that as `[["users", "email"], 7u64]` — the same bytes on the AAD side (a list is an `AadPiece::List`, PAE of its parts like a tuple) and on the PRF side (leaves carry vitaminc's own typed encodings, lists are `PrfContext::pae`). Rows sealed from Rust @@ -460,7 +460,8 @@ ct, _ := cipher.Encrypt(ctx, user, aad) // map[string]any of stackencrypt.Seal pt, _ := client.Decrypt(ctx, ct, aad) // any keyset rows, _ := cipher.EncryptRecords(ctx, users) // one ZeroKMS call for the slice -probe, _ := cipher.Term(ctx, uint32(34), stackencrypt.MustContext("users/age"), stackencrypt.Equality) +age, _ := stackencrypt.ParseLabel("users/age") // the field's name, as its `label=` tag spells it +probe, _ := cipher.Term(ctx, uint32(34), age.Context(), stackencrypt.Equality) ``` - `stackencrypt.Sealed` — the Phase 2 leaf; `driver.Valuer` + `sql.Scanner` @@ -471,8 +472,8 @@ probe, _ := cipher.Term(ctx, uint32(34), stackencrypt.MustContext("users/age"), ```go type User struct { ID int64 `stash:"plain"` - Age uint32 `stash:"context=users/age,index=eq;ore"` - Email string `stash:"context=users/email,index=eq;match"` + Age uint32 `stash:"label=users/age,index=eq;ore"` + Email string `stash:"label=users/email,index=eq;match"` } ``` diff --git a/languages/golang/stackencrypt/cipher.go b/languages/golang/stackencrypt/cipher.go index 79fbc6bad..3d92aace6 100644 --- a/languages/golang/stackencrypt/cipher.go +++ b/languages/golang/stackencrypt/cipher.go @@ -97,7 +97,7 @@ func (cph *Cipher) DecryptElement(ctx context.Context, ct any, aad []byte) (any, // round trip: term derivation is asynchronous in the Rust crate, and a // ZeroKMS backend that derives terms server-side settles the same way. func (cph *Cipher) Term(ctx context.Context, value any, context Context, kind TermKind, opts ...Option) (any, error) { - if context.node == nil { + if context.isZero() { return nil, fmt.Errorf("stackencrypt: term context is empty") } var o termOptions diff --git a/languages/golang/stackencrypt/context.go b/languages/golang/stackencrypt/context.go index 49c968de1..f99cf6dab 100644 --- a/languages/golang/stackencrypt/context.go +++ b/languages/golang/stackencrypt/context.go @@ -4,6 +4,7 @@ import ( "bytes" "errors" "fmt" + "reflect" ) // Context is the encryption context a record field or a term probe binds: @@ -14,16 +15,22 @@ import ( // // A Context is a part or a list of parts. A part is a string, a byte slice // or an integer (int32, int64, uint32, uint64; Go's int is sent as int64). -// [NewContext] makes a one-part context — the bare part, the shape a Rust -// `#[derive(EncryptFrom)]` field is sealed under when the caller supplies no -// context of its own. [Context.With] extends it as Rust's NonEmpty::with -// does: the result is the two-element list [previous, part], nesting to the -// left, so NewContext("users/age").With(uint64(7)) is the context a row -// sealed with encrypt_into_with_context(row, 7u64) binds for that field. +// [NewContext] makes a one-part context — the bare part, what a Rust +// `#[stash(context = "..")]` literal binds. [Context.With] extends it as +// Rust's NonEmpty::with does: the result is the two-element list +// [previous, part], nesting to the left. So NewContext("users").With("age") +// is the pair a Rust `struct = .., context = "users"` derive binds its `age` +// field under — and what ParseLabel("users/age") binds — rendering the ZeroKMS +// descriptor users/age; extended With(uint64(7)) it is what a row sealed +// with encrypt_into_with_context(row, 7u64) binds for that field. // A one-element list is not the bare part, and this type cannot spell one. // // A Context owns its parts: a byte-slice part is copied in, so a caller's // buffer reused once the Context is built does not change it. +// +// Compare two Contexts with [Context.Equal]. Do not use == and do not use a +// Context as a map key: a list context holds a slice, and Go panics when it +// compares those. type Context struct { node any } @@ -108,3 +115,32 @@ func checkPart(part any) error { return fmt.Errorf("stackencrypt: %T is not a context part (string, []byte or integer)", part) } } + +// Equal reports whether c and other are the same context: the same parts, +// in the same order, with the same types, so a probe built from one matches +// terms written under the other. This is the supported comparison; == on +// two Contexts panics when either holds a list. +func (c Context) Equal(other Context) bool { return reflect.DeepEqual(c.node, other.node) } + +// isZero reports whether c is the zero Context, which binds nothing: what a +// plan field without a context, or a zero Label, carries. +func (c Context) isZero() bool { return c.node == nil } + +// flatContext is the context a [Label] binds: one segment is the bare part, +// as NewContext makes it; two or more are a flat list of the segments. The +// segments are plain by construction, so no part check is needed, and a +// list is never built from one part (a one-element list is a different +// context from the bare part, and this type cannot spell one). +func flatContext(segments []string) Context { + switch len(segments) { + case 0: + return Context{} + case 1: + return Context{node: segments[0]} + } + parts := make([]any, len(segments)) + for i, s := range segments { + parts[i] = s + } + return Context{node: parts} +} diff --git a/languages/golang/stackencrypt/example/explicit/main.go b/languages/golang/stackencrypt/example/explicit/main.go index 6dee61e4c..41ab2e500 100644 --- a/languages/golang/stackencrypt/example/explicit/main.go +++ b/languages/golang/stackencrypt/example/explicit/main.go @@ -30,7 +30,7 @@ import ( type user struct { ID int64 `stash:"-"` - Email string `stash:"context=users/email,index=eq"` + Email string `stash:"label=users/email,index=eq"` } type config struct { @@ -148,7 +148,11 @@ func run(ctx context.Context, cfg config, secrets secrets) error { if err != nil { return fmt.Errorf("encrypting records: %w", err) } - probe, err := cipher.Term(ctx, "bob@example.com", stackencrypt.MustContext("users/email"), stackencrypt.Equality) + emailCtx, err := stackencrypt.ParseLabel("users/email") + if err != nil { + return fmt.Errorf("the probe's label: %w", err) + } + probe, err := cipher.Term(ctx, "bob@example.com", emailCtx.Context(), stackencrypt.Equality) if err != nil { return fmt.Errorf("deriving a probe: %w", err) } diff --git a/languages/golang/stackencrypt/example/main.go b/languages/golang/stackencrypt/example/main.go index 88db70002..acfe03572 100644 --- a/languages/golang/stackencrypt/example/main.go +++ b/languages/golang/stackencrypt/example/main.go @@ -22,12 +22,12 @@ import ( ) // A record type. The `stash` tag is the Go stand-in for Rust's -// `#[derive(EncryptFrom)]`: `context=` is the field's own encryption +// `#[derive(EncryptFrom)]`: `label=` is the field's own encryption // context, `index=` the terms to derive beside the ciphertext. type user struct { ID int64 `stash:"-"` - Email string `stash:"context=users/email,index=eq;match"` - Age uint32 `stash:"context=users/age,index=eq;ore"` + Email string `stash:"label=users/email,index=eq;match"` + Age uint32 `stash:"label=users/age,index=eq;ore"` } func main() { @@ -142,7 +142,11 @@ func recordsAndTerms(ctx context.Context, cipher *stackencrypt.Cipher) ([]stacke // being searched for. It never touches the ciphertext — matching is what // the term is for. fmt.Println() - probe, err := cipher.Term(ctx, "bob@example.com", stackencrypt.MustContext("users/email"), stackencrypt.Equality) + emailCtx, err := stackencrypt.ParseLabel("users/email") + if err != nil { + return nil, fmt.Errorf("the probe's label: %w", err) + } + probe, err := cipher.Term(ctx, "bob@example.com", emailCtx.Context(), stackencrypt.Equality) if err != nil { return nil, fmt.Errorf("deriving a probe: %w", err) } @@ -155,7 +159,11 @@ func recordsAndTerms(ctx context.Context, cipher *stackencrypt.Cipher) ([]stacke // A term is bound to its context. The same value under another field's // context is a different term, which is what stops a match in one column // from being a match in another. - wrong, err := cipher.Term(ctx, "bob@example.com", stackencrypt.MustContext("users/name"), stackencrypt.Equality) + nameCtx, err := stackencrypt.ParseLabel("users/name") + if err != nil { + return nil, fmt.Errorf("the probe's label: %w", err) + } + wrong, err := cipher.Term(ctx, "bob@example.com", nameCtx.Context(), stackencrypt.Equality) if err != nil { return nil, fmt.Errorf("deriving a probe: %w", err) } diff --git a/languages/golang/stackencrypt/guest/Cargo.lock b/languages/golang/stackencrypt/guest/Cargo.lock index e80577f68..7488fea75 100644 --- a/languages/golang/stackencrypt/guest/Cargo.lock +++ b/languages/golang/stackencrypt/guest/Cargo.lock @@ -1978,7 +1978,7 @@ dependencies = [ [[package]] name = "stack-encrypt" -version = "0.1.0" +version = "0.2.0" dependencies = [ "base64ct", "cllw-ore", @@ -1998,7 +1998,7 @@ dependencies = [ [[package]] name = "stack-encrypt-derive" -version = "0.1.0" +version = "0.2.0" dependencies = [ "proc-macro2", "quote", diff --git a/languages/golang/stackencrypt/guest_test.go b/languages/golang/stackencrypt/guest_test.go index 9c54e03de..c71f59ffb 100644 --- a/languages/golang/stackencrypt/guest_test.go +++ b/languages/golang/stackencrypt/guest_test.go @@ -611,8 +611,8 @@ func mustHex(s string) []byte { } type recordRow struct { - Age uint32 `stash:"context=users/age,index=eq;ore"` - Email string `stash:"context=users/email,index=eq;match"` + Age uint32 `stash:"label=users/age,index=eq;ore"` + Email string `stash:"label=users/email,index=eq;match"` } // A record decrypted under a plan that names a field it does not carry is @@ -621,7 +621,7 @@ func TestMismatchedPlanIsRefusedBeforeTheGuest(t *testing.T) { ctx := context.Background() c := rawInstance(t) record := EncryptedRecord{"Age": {Ciphertext: Sealed(fixtureLeaf)}} - plan, err := NewPlan(FieldPlan{Field: "Email", Name: "email", Context: "users/email"}) + plan, err := NewPlan(FieldPlan{Field: "Email", Name: "email", Context: label(t, "users/email").Context()}) if err != nil { t.Fatal(err) } @@ -654,8 +654,8 @@ func TestGuestAcceptsEveryEncodingThisPackageBuilds(t *testing.T) { Email string } plan, err := NewPlan( - FieldPlan{Field: "Age", Context: "users/age", Terms: []TermKind{Equality, Ore}}, - FieldPlan{Field: "Email", Context: "users/email", Terms: []TermKind{Equality, Match}}, + FieldPlan{Field: "Age", Context: label(t, "users/age").Context(), Terms: []TermKind{Equality, Ore}}, + FieldPlan{Field: "Email", Context: label(t, "users/email").Context(), Terms: []TermKind{Equality, Match}}, ) if err != nil { t.Fatal(err) @@ -710,7 +710,7 @@ func TestGuestRefusesMalformedInputsBeforeState(t *testing.T) { c := rawInstance(t) def := c.DefaultKeyset() type badRow struct { - Age float64 `stash:"context=users/age,index=eq"` + Age float64 `stash:"label=users/age,index=eq"` } calls := map[string]func() error{ "float under equality": func() error { _, err := def.Term(ctx, 1.5, MustContext("k"), Equality); return err }, diff --git a/languages/golang/stackencrypt/label.go b/languages/golang/stackencrypt/label.go new file mode 100644 index 000000000..98ad01f0c --- /dev/null +++ b/languages/golang/stackencrypt/label.go @@ -0,0 +1,137 @@ +package stackencrypt + +import ( + "errors" + "fmt" + "slices" + "strings" + "unicode" +) + +// Label names the data a field or a probe binds: a table and a column +// ("users/email"), a document path ("documents/v2/body"), any name a direct +// consumer chooses. ZeroKMS binds the data key to that name and writes it in +// its log, spelled exactly as given. It is the Go form of Rust's +// stack_encrypt::Label; EQL's identifier, a table and a column, is a Label of +// two segments ([plan.Identifier]). +// +// # Naming and scoping +// +// A context carries two kinds of information, and each has one spelling: +// +// - A name says WHAT the data is. Spell it as a Label. +// - A scope says WHICH slice of that data: a tenant, a row. Spell it by +// extending the name's context with [Context.With], or with the +// [ExtendContext] option on a record call. +// +// In practice: +// +// What you mean Spelling ZeroKMS log +// the users.email column label, _ := ParseLabel("users/email") users/email +// that column, tenant 7 label.Context().With(uint64(7)) (users/email)/7u64 +// a deeper name ParseLabel("documents/v2/body") documents/v2/body +// a one-part name ParseLabel("users"), the same as NewContext("users") users +// +// Do not build a name with With, and do not put a scope into a Label. The +// renderer keeps the two apart: a name is one flat list, a scope nests. So +// (users/email)/7u64 is never read as a three-segment name, and +// documents/v2/body is never read as a scoped column. +// +// A two-segment Label binds the same context a Rust +// `#[stash(struct = .., context = "")]` derive gives a field. That is +// what lets a Go label open a row a Rust derive wrote, and a probe built from +// the label match the terms the derive produced. +// +// # Segments +// +// Every segment is plain — non-empty, no control or invisible format +// characters (zero-width and bidirectional marks), none of '/', '(' or ')', +// not beginning with "b64:", a digit or '-' — which is exactly +// the text the descriptor renders verbatim. So a Label's [Label.String] is +// its descriptor, [ParseLabel] reads that string back losslessly (no +// segment can contain the separator), and a string that is not a label is +// refused with a [LabelError] naming the segment, never escaped silently. +type Label struct { + segments []string +} + +// labelSeparator joins a label's segments: the descriptor's own separator. +const labelSeparator = "/" + +// NewLabel makes a label from its segments, each checked to be plain. +func NewLabel(segments ...string) (Label, error) { + if len(segments) == 0 { + return Label{}, ErrEmptyLabel + } + for i, s := range segments { + if err := checkSegment(i, s); err != nil { + return Label{}, err + } + } + return Label{segments: slices.Clone(segments)}, nil +} + +// ParseLabel reads a label from its rendered form, segments separated by +// '/': the inverse of [Label.String]. "users//email" and "users/" are +// refused (an empty segment), as is "" (one empty segment). +func ParseLabel(s string) (Label, error) { + return NewLabel(strings.Split(s, labelSeparator)...) +} + +// Segments returns the label's segments, in order; at least one. +func (l Label) Segments() []string { return slices.Clone(l.segments) } + +// String renders the label as its descriptor: the segments joined by '/'. +func (l Label) String() string { return strings.Join(l.segments, labelSeparator) } + +// Context is the label as the context a field or probe binds. A zero +// Label gives the zero Context, which every call refuses as "needs a +// context". +func (l Label) Context() Context { return flatContext(l.segments) } + +// ErrEmptyLabel is [NewLabel]'s refusal of no segments at all. +var ErrEmptyLabel = errors.New("stackencrypt: a label needs at least one segment") + +// LabelError says why a string is not a [Label] segment. Index is the +// segment's position, counting from zero. +type LabelError struct { + Index int + Reason string +} + +func (e *LabelError) Error() string { + return fmt.Sprintf("stackencrypt: label segment %d %s", e.Index, e.Reason) +} + +// checkSegment is the one definition of plain text, the same as Rust's +// Label::check_segment: what passes here is what the descriptor renders +// verbatim. +func checkSegment(index int, s string) error { + if s == "" { + return &LabelError{Index: index, Reason: "is empty"} + } + if strings.HasPrefix(s, "b64:") || s[0] == '-' || (s[0] >= '0' && s[0] <= '9') { + return &LabelError{Index: index, Reason: "begins like another descriptor form (b64:, a digit or -)"} + } + for _, r := range s { + if r == '/' { + return &LabelError{Index: index, Reason: "contains '/', the separator"} + } + if unicode.IsControl(r) || r == '(' || r == ')' { + return &LabelError{Index: index, Reason: fmt.Sprintf("contains %q, which the descriptor reserves", r)} + } + if strings.ContainsRune(invisible, r) { + return &LabelError{Index: index, Reason: fmt.Sprintf("contains %q, an invisible format character", r)} + } + } + return nil +} + +// invisible is the format characters with no glyph of their own: the soft +// hyphen, the Arabic letter mark, the Mongolian vowel separator, the +// zero-width characters, the bidirectional embeddings, overrides and +// isolates, and the byte-order mark. unicode.IsControl covers only Cc; +// these are Cf. A name containing one prints like another name in the +// ZeroKMS log, so they are refused beside the control characters. The same +// list as Rust's Label::INVISIBLE; the shared fixture holds the two together. +const invisible = "\u00ad\u061c\u180e\u200b\u200c\u200d\u200e\u200f\u202a\u202b\u202c\u202d\u202e\u2060\u2061\u2062\u2063\u2064\u2066\u2067\u2068\u2069\ufeff" diff --git a/languages/golang/stackencrypt/label_test.go b/languages/golang/stackencrypt/label_test.go new file mode 100644 index 000000000..ae531c282 --- /dev/null +++ b/languages/golang/stackencrypt/label_test.go @@ -0,0 +1,241 @@ +package stackencrypt + +import ( + "encoding/json" + "errors" + "os" + "path/filepath" + "reflect" + "strings" + "testing" +) + +// A label's string is its descriptor and reads back losslessly; as a +// context, one segment is the bare part, two are the pair With builds, and +// more are a flat list — the same shapes Rust's Label takes. +func TestLabelRendersAsItsDisplayAndBindsTheMatchingContext(t *testing.T) { + pair, err := NewContext("users") + if err != nil { + t.Fatal(err) + } + if pair, err = pair.With("age"); err != nil { + t.Fatal(err) + } + for text, want := range map[string]any{ + "users": "users", + "users/age": []any{"users", "age"}, + "documents/v2/body": []any{"documents", "v2", "body"}, + "naïve/with space": []any{"naïve", "with space"}, + } { + l, err := ParseLabel(text) + if err != nil { + t.Fatalf("ParseLabel(%q): %v", text, err) + } + if l.String() != text { + t.Errorf("ParseLabel(%q).String() = %q", text, l.String()) + } + if got := l.Segments(); strings.Join(got, "/") != text { + t.Errorf("ParseLabel(%q).Segments() = %q", text, got) + } + if got := l.Context().value(); !reflect.DeepEqual(got, want) { + t.Errorf("ParseLabel(%q).Context() = %#v, want %#v", text, got, want) + } + if again, err := NewLabel(l.Segments()...); err != nil || !reflect.DeepEqual(again, l) { + t.Errorf("NewLabel(segments of %q) = %#v, %v", text, again, err) + } + } + if got := label(t, "users/age").Context().value(); !reflect.DeepEqual(got, pair.value()) { + t.Errorf("a two-segment label is not the With pair: %#v vs %#v", got, pair.value()) + } + if got := label(t, "users").Context().value(); !reflect.DeepEqual(got, MustContext("users").value()) { + t.Errorf("a one-segment label is not the bare part: %#v", got) + } + // A literal containing '/' is one part, not the pair: the two spell + // different contexts, as in Rust. + if reflect.DeepEqual(MustContext("users/age").value(), label(t, "users/age").Context().value()) { + t.Error(`NewContext("users/age") and label(t, "users/age") bind the same context`) + } + if got := (Label{}).Context(); !got.isZero() { + t.Errorf("zero Label's Context = %#v, want the zero Context", got) + } +} + +// Every way a segment is not plain is refused and named, matching Rust's +// LabelError variants: a label never renders escaped. +func TestLabelRefusesSegmentsThatWouldNotRenderVerbatim(t *testing.T) { + if _, err := NewLabel(); !errors.Is(err, ErrEmptyLabel) { + t.Errorf("NewLabel() = %v, want ErrEmptyLabel", err) + } + for _, tc := range []struct { + segments []string + index int + reason string + }{ + {[]string{""}, 0, "is empty"}, + {[]string{"users", ""}, 1, "is empty"}, + {[]string{"users", "a/b"}, 1, "separator"}, + {[]string{"b64:x"}, 0, "another descriptor form"}, + {[]string{"users", "7"}, 1, "another descriptor form"}, + {[]string{"-x"}, 0, "another descriptor form"}, + {[]string{"a(b"}, 0, "reserves"}, + {[]string{"a)b"}, 0, "reserves"}, + {[]string{"a\tb"}, 0, "reserves"}, + {[]string{"a\u0085b"}, 0, "reserves"}, + } { + _, err := NewLabel(tc.segments...) + var le *LabelError + if !errors.As(err, &le) { + t.Errorf("NewLabel(%q) = %v, want a LabelError", tc.segments, err) + continue + } + if le.Index != tc.index || !strings.Contains(le.Reason, tc.reason) { + t.Errorf("NewLabel(%q) = %v, want segment %d %q", tc.segments, err, tc.index, tc.reason) + } + } + for _, text := range []string{"", "/", "users/", "/age", "users//age"} { + if _, err := ParseLabel(text); err == nil { + t.Errorf("ParseLabel(%q) succeeded; want a refusal", text) + } + } +} + +// The segment rule is one rule in two languages. Rust's +// plain_text_and_label_segments_are_one_rule reads the same fixture, so a +// change to either implementation that the other does not follow fails here +// or there. +func TestLabelSegmentRuleMatchesTheSharedFixture(t *testing.T) { + raw, err := os.ReadFile(filepath.Join("..", "..", "..", "packages", "stack-encrypt", "tests", "fixtures", "label_segments.json")) + if err != nil { + t.Fatal(err) + } + var fixture struct { + Plain []string `json:"plain"` + NotPlain []string `json:"not_plain"` + } + if err := json.Unmarshal(raw, &fixture); err != nil { + t.Fatal(err) + } + if len(fixture.Plain) == 0 || len(fixture.NotPlain) == 0 { + t.Fatalf("fixture is empty: %+v", fixture) + } + for _, ok := range fixture.Plain { + if _, err := NewLabel(ok); err != nil { + t.Errorf("NewLabel(%q) = %v, want ok", ok, err) + } + } + for _, bad := range fixture.NotPlain { + if _, err := NewLabel(bad); err == nil { + t.Errorf("NewLabel(%q) succeeded; want a refusal", bad) + } + } +} + +// label is ParseLabel for a label the test knows to be valid: the fixture +// form of the error-returning constructor, since there is no panicking one. +func label(t testing.TB, s string) Label { + t.Helper() + l, err := ParseLabel(s) + if err != nil { + t.Fatal(err) + } + return l +} + +// A struct tag names a field's own context as a label or as one part, +// never both, and a plan built by hand needs a non-zero Context. +func TestTagsSpellALabelOrOneContextPart(t *testing.T) { + type tagged struct { + Email string `stash:"label=users/email"` + Notes string `stash:"context=notes/v1"` + } + p, err := PlanFromTags(reflect.TypeOf(tagged{})) + if err != nil { + t.Fatal(err) + } + fields := p.Fields() + if got := fields[0].Context.value(); !reflect.DeepEqual(got, []any{"users", "email"}) { + t.Errorf("label=users/email bound %#v", got) + } + // context= is one part: the '/' is text, as a Rust literal's is. + if got := fields[1].Context.value(); !reflect.DeepEqual(got, "notes/v1") { + t.Errorf("context=notes/v1 bound %#v, want the one part", got) + } + for name, typ := range map[string]reflect.Type{ + "both": reflect.TypeOf(struct { + A string `stash:"label=t/a,context=a"` + }{}), + "label twice": reflect.TypeOf(struct { + A string `stash:"label=t/a,label=t/b"` + }{}), + "bad label": reflect.TypeOf(struct { + A string `stash:"label=t/a/"` + }{}), + "empty context": reflect.TypeOf(struct { + A string `stash:"context="` + }{}), + "no context": reflect.TypeOf(struct { + A string `stash:"index=eq"` + }{}), + } { + if _, err := PlanFromTags(typ); err == nil { + t.Errorf("%s: PlanFromTags succeeded; want a refusal", name) + } + } + if _, err := NewPlan(FieldPlan{Field: "A"}); err == nil || !strings.Contains(err.Error(), "needs a context") { + t.Errorf("NewPlan without a context = %v", err) + } +} + +// A repeated option is reported as what the author wrote, not as a mix of +// the two keys. +func TestARepeatedOwnContextOptionNamesItself(t *testing.T) { + _, err := PlanFromTags(reflect.TypeOf(struct { + A string `stash:"context=a,context=b"` + }{})) + if err == nil || !strings.Contains(err.Error(), "context= and context= both given") { + t.Errorf("err = %v, want the repeated option named", err) + } +} + +// A Label owns its segments, as a Context owns its parts: NewLabel copies +// the slice in and Segments copies it out, so neither side can change the +// name the data is keyed under through a shared array. +func TestLabelOwnsItsSegments(t *testing.T) { + segments := []string{"users", "email"} + l, err := NewLabel(segments...) + if err != nil { + t.Fatal(err) + } + segments[1] = "phone" + if l.String() != "users/email" { + t.Errorf("label followed the caller's slice: %s", l) + } + out := l.Segments() + out[0] = "accounts" + if l.String() != "users/email" { + t.Errorf("label followed the returned slice: %s", l) + } +} + +// Equal is the supported comparison: == on two list contexts panics. +func TestContextEqualComparesParts(t *testing.T) { + pair := label(t, "users/email").Context() + if !pair.Equal(label(t, "users/email").Context()) { + t.Error("equal labels compare unequal") + } + if pair.Equal(label(t, "users/phone").Context()) || pair.Equal(MustContext("users/email")) { + t.Error("different contexts compare equal") + } + if !MustContext("users").Equal(label(t, "users").Context()) { + t.Error("a one-segment label is not the bare part") + } + if (Context{}).Equal(pair) || !(Context{}).Equal(Context{}) { + t.Error("the zero Context compares wrongly") + } + defer func() { + if recover() == nil { + t.Error("== on list contexts did not panic; Equal's reason to exist is gone, revisit its doc") + } + }() + _ = pair == label(t, "users/email").Context() +} diff --git a/languages/golang/stackencrypt/live_test.go b/languages/golang/stackencrypt/live_test.go index f3e4e50d2..de8c405a3 100644 --- a/languages/golang/stackencrypt/live_test.go +++ b/languages/golang/stackencrypt/live_test.go @@ -111,8 +111,8 @@ func liveClient(t *testing.T) *Client { type liveUser struct { ID int64 `stash:"-"` - Age uint32 `stash:"context=users/age,index=eq;ore"` - Email string `stash:"context=users/email,index=eq;match"` + Age uint32 `stash:"label=users/age,index=eq;ore"` + Email string `stash:"label=users/email,index=eq;match"` } func TestLiveValueRoundTrip(t *testing.T) { @@ -183,7 +183,7 @@ func TestLiveRecordsAndTerms(t *testing.T) { t.Fatalf("records = %+v", records) } - probe, err := cipher.Term(ctx, uint32(34), MustContext("users/age"), Equality) + probe, err := cipher.Term(ctx, uint32(34), label(t, "users/age").Context(), Equality) if err != nil { t.Fatal(err) } @@ -228,7 +228,7 @@ func TestLiveRecordsAndTerms(t *testing.T) { if err != nil { t.Fatal(err) } - scoped, err := cipher.Term(ctx, "bob@example.com", MustContext("users/email"), Equality, tenant7) + scoped, err := cipher.Term(ctx, "bob@example.com", label(t, "users/email").Context(), Equality, tenant7) if err != nil { t.Fatal(err) } @@ -241,7 +241,7 @@ func TestLiveRecordsAndTerms(t *testing.T) { if scoped.(EqualityTerm).Equal(records[1]["Email"].Equality) { t.Error("tenant probe equals the unextended term") } - unscoped, err := cipher.Term(ctx, "bob@example.com", MustContext("users/email"), Equality) + unscoped, err := cipher.Term(ctx, "bob@example.com", label(t, "users/email").Context(), Equality) if err != nil { t.Fatal(err) } @@ -261,8 +261,8 @@ func TestLiveExplicitPlanRoundTrip(t *testing.T) { Email string } plan, err := NewPlan( - FieldPlan{Field: "Age", Context: "users/age", Terms: []TermKind{Equality, Ore}}, - FieldPlan{Field: "Email", Context: "users/email", Terms: []TermKind{Equality, Match}}, + FieldPlan{Field: "Age", Context: label(t, "users/age").Context(), Terms: []TermKind{Equality, Ore}}, + FieldPlan{Field: "Email", Context: label(t, "users/email").Context(), Terms: []TermKind{Equality, Match}}, ) if err != nil { t.Fatal(err) @@ -290,7 +290,7 @@ func TestLiveExplicitPlanRoundTrip(t *testing.T) { // A plan naming a field the record does not carry is refused before // any key is requested. - other, err := NewPlan(FieldPlan{Field: "Email", Name: "email", Context: "users/email"}) + other, err := NewPlan(FieldPlan{Field: "Email", Name: "email", Context: label(t, "users/email").Context()}) if err != nil { t.Fatal(err) } diff --git a/languages/golang/stackencrypt/plan/message.go b/languages/golang/stackencrypt/plan/message.go index 9b827226a..2e818dbec 100644 --- a/languages/golang/stackencrypt/plan/message.go +++ b/languages/golang/stackencrypt/plan/message.go @@ -4,7 +4,6 @@ import ( "errors" "fmt" "reflect" - "strings" "github.com/cipherstash/stack/languages/golang/stackencrypt" ) @@ -78,9 +77,6 @@ func (m Message) Build(facts []Fact) (stackencrypt.Plan, error) { if m.table == "" { return stackencrypt.Plan{}, fmt.Errorf("plan: %s: a message needs a Table", messageName(m, facts)) } - if strings.Contains(string(m.table), "/") { - return stackencrypt.Plan{}, fmt.Errorf("plan: %s: table %q contains '/', which would make its column identities ambiguous", messageName(m, facts), m.table) - } var fields []stackencrypt.FieldPlan var errs []error // An EQL identity is one column's context: two fields sharing one would @@ -167,20 +163,16 @@ func (m Message) field(f Fact) (stackencrypt.FieldPlan, string, bool, error) { } else if d.identity != "" { identity = d.identity } - // A '/' in an EQL column would make an identity-shaped context - // ambiguous ("a/b" under "t" reads as "a" under "t/b" would), and the - // column is the identity until the day it is renamed. A Custom - // target's column is only the record key. - if !custom { - for _, name := range []string{column, identity} { - if strings.Contains(name, "/") { - return none(fmt.Errorf("%w: column %q contains '/', which would make its identity ambiguous", ErrInvalid, name)) - } - } - } - context := d.target.Context(Identifier{Table: string(m.table), Column: identity}) - if context == "" { - return none(fmt.Errorf("%w: target %v gives an empty context", ErrInvalid, d.target)) + // An EQL target's context is Identifier.Label(), which refuses a table + // or column that would not render as itself — a '/' among them, since it + // would read as two names. A Custom target's column is only the record + // key, so it may contain anything. + context, err := d.target.Context(Identifier{Table: string(m.table), Column: identity}) + if err != nil { + // Both errors stay reachable: ErrInvalid for the policy's caller, and + // the target's own (a *stackencrypt.LabelError, say) for one that + // wants to know which name was wrong. + return none(fmt.Errorf("%w: target %v: %w", ErrInvalid, d.target, err)) } return stackencrypt.FieldPlan{ Field: f.goField(), diff --git a/languages/golang/stackencrypt/plan/plan_test.go b/languages/golang/stackencrypt/plan/plan_test.go index 37db29402..f0cfaac29 100644 --- a/languages/golang/stackencrypt/plan/plan_test.go +++ b/languages/golang/stackencrypt/plan/plan_test.go @@ -43,10 +43,10 @@ func TestPolicyBuildsThePlan(t *testing.T) { // Columns are the schema's spelling of the Go field: what the Rust // derive binds and the database names. want := []se.FieldPlan{ - {Field: "Email", Name: "email", Context: "individuals/email", Terms: []se.TermKind{se.Equality, se.Match}}, - {Field: "Name", Name: "name", Context: "individuals/name"}, + {Field: "Email", Name: "email", Context: label(t, "individuals/email").Context(), Terms: []se.TermKind{se.Equality, se.Match}}, + {Field: "Name", Name: "name", Context: label(t, "individuals/name").Context()}, // The per-message rule wins over the base's government_id rule. - {Field: "MedicareNo", Name: "medicare_number", Context: "individuals/medicare_number", Terms: []se.TermKind{se.Equality, se.Ore}}, + {Field: "MedicareNo", Name: "medicare_number", Context: label(t, "individuals/medicare_number").Context(), Terms: []se.TermKind{se.Equality, se.Ore}}, } if got := p.Fields(); !reflect.DeepEqual(got, want) { t.Fatalf("fields =\n%+v\nwant\n%+v", got, want) @@ -146,7 +146,7 @@ func TestUnclassifiedFieldsAreLeftOutUnlessNamed(t *testing.T) { if err != nil { t.Fatal(err) } - want := []se.FieldPlan{{Field: "Notes", Name: "notes", Context: "individuals/notes"}} + want := []se.FieldPlan{{Field: "Notes", Name: "notes", Context: label(t, "individuals/notes").Context()}} if got := p.Fields(); !reflect.DeepEqual(got, want) { t.Fatalf("fields = %+v, want %+v", got, want) } @@ -174,8 +174,9 @@ func TestColumnPinSurvivesRenames(t *testing.T) { t.Fatal(err) } f1, f2 := p1.Fields()[0], p2.Fields()[0] - if f1.Context != "individuals/medicare_number" || f2.Context != f1.Context { - t.Fatalf("contexts %q, %q: want both individuals/medicare_number", f1.Context, f2.Context) + want := label(t, "individuals/medicare_number").Context() + if !f1.Context.Equal(want) || !f2.Context.Equal(f1.Context) { + t.Fatalf("contexts %v, %v: want both individuals/medicare_number", f1.Context, f2.Context) } if f2.Name != "medicare_number" || f2.Field != "MedicareNo" { t.Fatalf("pinned field = %+v", f2) @@ -185,8 +186,8 @@ func TestColumnPinSurvivesRenames(t *testing.T) { if err != nil { t.Fatal(err) } - if got := unpinned.Fields()[0].Context; got != "individuals/medicare_no" { - t.Fatalf("unpinned context = %q", got) + if got := unpinned.Fields()[0].Context; !got.Equal(label(t, "individuals/medicare_no").Context()) { + t.Fatalf("unpinned context = %v", got) } } @@ -214,7 +215,7 @@ func TestIdentityKeepsTheContextThroughAColumnRename(t *testing.T) { t.Errorf("%s: %v", name, err) continue } - want := []se.FieldPlan{{Field: "MedicareNo", Name: tc.key, Context: tc.context, Terms: []se.TermKind{se.Equality}}} + want := []se.FieldPlan{{Field: "MedicareNo", Name: tc.key, Context: label(t, tc.context).Context(), Terms: []se.TermKind{se.Equality}}} if got := p.Fields(); !reflect.DeepEqual(got, want) { t.Errorf("%s: fields = %+v, want %+v", name, got, want) } @@ -236,9 +237,9 @@ func TestContextsByTarget(t *testing.T) { t.Fatal(err) } want := []se.FieldPlan{ - {Field: "Email", Name: "email", Context: "users/email", Terms: []se.TermKind{se.Equality}}, + {Field: "Email", Name: "email", Context: label(t, "users/email").Context(), Terms: []se.TermKind{se.Equality}}, // A custom target's context is its own; the pin names the record key only. - {Field: "Blob", Name: "blob_v1", Context: "tenant-blobs/v1", Terms: []se.TermKind{se.Ope}}, + {Field: "Blob", Name: "blob_v1", Context: se.MustContext("tenant-blobs/v1"), Terms: []se.TermKind{se.Ope}}, } if got := p.Fields(); !reflect.DeepEqual(got, want) { t.Fatalf("fields =\n%+v\nwant\n%+v", got, want) @@ -276,7 +277,13 @@ func TestBuildRefusesMalformedDecisions(t *testing.T) { "identity on plain": {"t", plan.When(plan.Field("a"), plan.Plaintext(), plan.Identity("c")), plan.ErrInvalid, "Plaintext"}, "identity on custom": {"t", plan.When(plan.Field("a"), plan.Encrypt(plan.Custom("ctx")), plan.Identity("c")), plan.ErrInvalid, "context is fixed"}, "slash in identity": {"t", plan.When(plan.Field("a"), plan.Encrypt(plan.EQL()), plan.Column("c"), plan.Identity("x/y")), plan.ErrInvalid, "contains '/'"}, - "slash, renamed": {"t", plan.When(plan.Field("a"), plan.Encrypt(plan.EQL()), plan.Column("x/y"), plan.Identity("c")), plan.ErrInvalid, "contains '/'"}, + // Identifier.Label() refuses more than '/': every reason a segment is + // not plain, named as the table or the column identity it came from. + "digit in table": {"2024_events", plan.When(plan.Field("a"), plan.Encrypt(plan.EQL())), plan.ErrInvalid, `table "2024_events"`}, + "digit in column": {"t", plan.When(plan.Field("a"), plan.Encrypt(plan.EQL()), plan.Column("2fa_secret")), plan.ErrInvalid, `column identity "2fa_secret"`}, + "b64 in column": {"t", plan.When(plan.Field("a"), plan.Encrypt(plan.EQL()), plan.Column("b64:x")), plan.ErrInvalid, "another descriptor form"}, + "paren in column": {"t", plan.When(plan.Field("a"), plan.Encrypt(plan.EQL()), plan.Column("a(b")), plan.ErrInvalid, "reserves"}, + "invisible in table": {"users\u200b", plan.When(plan.Field("a"), plan.Encrypt(plan.EQL())), plan.ErrInvalid, "invisible"}, "zero decision": {"t", plan.When(plan.Field("a"), plan.Decision{}), plan.ErrInvalid, "zero Decision"}, "empty context": {"t", plan.When(plan.Field("a"), plan.Encrypt(plan.Custom(""))), plan.ErrInvalid, "empty context"}, "nil policy": {"t", nil, plan.ErrUnmatched, ""}, @@ -474,3 +481,57 @@ func TestWhenRefusesANilMatcher(t *testing.T) { }() plan.When(nil, plan.Plaintext()) } + +// label is se.ParseLabel for a label the test knows to be valid. +func label(t testing.TB, s string) se.Label { + t.Helper() + l, err := se.ParseLabel(s) + if err != nil { + t.Fatal(err) + } + return l +} + +// With the identity pinned, the storage column is only the record key, as a +// Custom target's is: a '/' in it names a database column, not a context, so +// it is accepted and the context stays the pinned identity's label. +func TestASlashInARenamedStorageColumnIsOnlyARecordKey(t *testing.T) { + facts := []plan.Fact{{Field: "blob", GoField: "Blob", Annotations: []plan.Annotation{{Key: "k", Values: []string{"v"}}}}} + p, err := plan.ForMessage(nil, "t", plan.When(plan.Field("blob"), plan.Encrypt(plan.EQL()), plan.Column("blob/v1"), plan.Identity("blob"))).Build(facts) + if err != nil { + t.Fatal(err) + } + f := p.Fields()[0] + if f.Name != "blob/v1" || !f.Context.Equal(label(t, "t/blob").Context()) { + t.Fatalf("field = %+v, want record key blob/v1 under t/blob", f) + } +} + +// The label error survives the wrapping, so a caller can learn which half +// of the identifier was wrong rather than only that the decision is invalid. +func TestABadIdentifierKeepsItsLabelError(t *testing.T) { + facts := []plan.Fact{{Field: "a", Annotations: []plan.Annotation{{Key: "k", Values: []string{"v"}}}}} + _, err := plan.ForMessage(nil, "a/b", plan.When(plan.Field("a"), plan.Encrypt(plan.EQL()))).Build(facts) + var le *se.LabelError + if !errors.Is(err, plan.ErrInvalid) || !errors.As(err, &le) || le.Index != 0 { + t.Fatalf("err = %v; want ErrInvalid wrapping a LabelError for segment 0", err) + } + if !strings.Contains(err.Error(), `table "a/b"`) { + t.Errorf("err = %v; want the table named", err) + } +} + +// Only an EQL target's context is built from the table, so a message whose +// every encrypted field has a Custom target builds with a table name that +// would not be a plain segment. Recorded, not endorsed: the table is unused +// by such a field's context. +func TestACustomOnlyMessageTakesAnyTableName(t *testing.T) { + facts := []plan.Fact{{Field: "a", Annotations: []plan.Annotation{{Key: "k", Values: []string{"v"}}}}} + p, err := plan.ForMessage(nil, "a/b", plan.When(plan.Field("a"), plan.Encrypt(plan.Custom("ctx")))).Build(facts) + if err != nil { + t.Fatal(err) + } + if got := p.Fields()[0].Context; !got.Equal(se.MustContext("ctx")) { + t.Errorf("context = %v, want the custom part", got) + } +} diff --git a/languages/golang/stackencrypt/plan/policy.go b/languages/golang/stackencrypt/plan/policy.go index 5a802d508..e77608d8f 100644 --- a/languages/golang/stackencrypt/plan/policy.go +++ b/languages/golang/stackencrypt/plan/policy.go @@ -1,6 +1,7 @@ package plan import ( + "errors" "fmt" "slices" "strings" @@ -20,11 +21,21 @@ type Identifier struct { Column string } -// String is the context an EQL target binds: "
/", the -// shape the Rust derive gives a `#[stash(struct = T, context = "
")]` -// field. +// String joins the table and the column with '/', for messages. The context +// an EQL target binds, and the descriptor ZeroKMS logs for it, is +// [Identifier.Label], which refuses a name this joining would misrender. func (id Identifier) String() string { return id.Table + "/" + id.Column } +// Label is the identifier as the two-segment [stackencrypt.Label] an EQL +// target binds: the shape the Rust derive gives a +// `#[stash(struct = T, context = "
")]` field, and what EQL's own +// Identifier describes. It is refused when either half is not plain — +// contains '/', '(' or ')', a control character, or begins with "b64:", a +// digit or '-' — since such a name would not render as itself. +func (id Identifier) Label() (stackencrypt.Label, error) { + return stackencrypt.NewLabel(id.Table, id.Column) +} + // Target is what an encrypted field is stored as: the index terms derived // beside its ciphertext, and the context it binds. The context is the // AAD, the ZeroKMS data-key binding and the terms' PRF context at once. @@ -32,12 +43,12 @@ type Target interface { // Terms lists the index terms to derive, in order. Terms() []stackencrypt.TermKind // Context returns the field's context given its column identity. An - // EQL target returns id.String(); a custom target returns its own. - Context(id Identifier) string + // EQL target binds id.Label(); a custom target returns its own. + Context(id Identifier) (stackencrypt.Context, error) } // EQL is an EQL column target: the field binds its column identity -// ([Identifier.String]) as its context and derives the given terms. Typed +// ([Identifier.Label]) as its context and derives the given terms. Typed // EQL targets (a text-with-equality column, say) are this with the terms // filled in, and implement [Target] the same way. func EQL(terms ...stackencrypt.TermKind) Target { @@ -47,13 +58,30 @@ func EQL(terms ...stackencrypt.TermKind) Target { type eqlTarget struct{ terms []stackencrypt.TermKind } func (t eqlTarget) Terms() []stackencrypt.TermKind { return slices.Clone(t.terms) } -func (t eqlTarget) Context(id Identifier) string { return id.String() } -func (t eqlTarget) String() string { return "EQL(" + termList(t.terms) + ")" } +func (t eqlTarget) Context(id Identifier) (stackencrypt.Context, error) { + l, err := id.Label() + if err != nil { + // The label error names a segment index; the caller gave a table and + // a column identity, so say which of those it was. + half, name := "table", id.Table + var le *stackencrypt.LabelError + if errors.As(err, &le) && le.Index == 1 { + half, name = "column identity", id.Column + } + return stackencrypt.Context{}, fmt.Errorf("%s %q cannot name a context: %w", half, name, err) + } + return l.Context(), nil +} +func (t eqlTarget) String() string { return "EQL(" + termList(t.terms) + ")" } // Custom is a non-EQL target: the field binds context, whatever its -// column, and derives the given terms. The context need not be -// table/column shaped; it is the policy's to choose and, like any context, -// must never change once data is written under it. +// column, and derives the given terms. The context is the policy's to +// choose and, like any context, must never change once data is written +// under it. It is one arbitrary text part, exactly as written — what +// [stackencrypt.NewContext] makes and a Rust `#[stash(context = "..")]` +// literal binds — so a '/' in it is text, not a separator: "notes/v1" is +// one part, rendered escaped in the ZeroKMS log, never the table/column +// pair. A table and a column are an [EQL] target. func Custom(context string, terms ...stackencrypt.TermKind) Target { return customTarget{context: context, terms: slices.Clone(terms)} } @@ -64,7 +92,9 @@ type customTarget struct { } func (t customTarget) Terms() []stackencrypt.TermKind { return slices.Clone(t.terms) } -func (t customTarget) Context(Identifier) string { return t.context } +func (t customTarget) Context(Identifier) (stackencrypt.Context, error) { + return stackencrypt.NewContext(t.context) +} func (t customTarget) String() string { return fmt.Sprintf("Custom(%q%s)", t.context, prefixed(termList(t.terms))) } diff --git a/languages/golang/stackencrypt/policy_plan_test.go b/languages/golang/stackencrypt/policy_plan_test.go index 081ab64c1..4363a576c 100644 --- a/languages/golang/stackencrypt/policy_plan_test.go +++ b/languages/golang/stackencrypt/policy_plan_test.go @@ -14,7 +14,7 @@ import ( // Validate refuses a nil type as PlanFromTags does, for the zero plan // and a built one alike, rather than dereferencing it. func TestValidateRefusesANilType(t *testing.T) { - built, err := se.NewPlan(se.FieldPlan{Field: "A", Context: "t/a"}) + built, err := se.NewPlan(se.FieldPlan{Field: "A", Context: label(t, "t/a").Context()}) if err != nil { t.Fatal(err) } @@ -44,8 +44,8 @@ func TestPolicyPlanIsTheHandBuiltPlan(t *testing.T) { plan.When(category.Under("user"), plan.Encrypt(plan.EQL())), plan.When(category.Present(), plan.Plaintext()), ) - email := se.FieldPlan{Field: "Email", Name: "email", Context: "individuals/email", Terms: []se.TermKind{se.Equality, se.Match}} - name := se.FieldPlan{Field: "Name", Name: "name", Context: "individuals/name"} + email := se.FieldPlan{Field: "Email", Name: "email", Context: label(t, "individuals/email").Context(), Terms: []se.TermKind{se.Equality, se.Match}} + name := se.FieldPlan{Field: "Name", Name: "name", Context: label(t, "individuals/name").Context()} for label, tc := range map[string]struct { pins []plan.RuleOption medicare se.FieldPlan @@ -54,12 +54,12 @@ func TestPolicyPlanIsTheHandBuiltPlan(t *testing.T) { // database would have it; Column alone sets the identity too. "column": { []plan.RuleOption{plan.Column("medicare_number")}, - se.FieldPlan{Field: "MedicareNo", Name: "medicare_number", Context: "individuals/medicare_number", Terms: []se.TermKind{se.Equality, se.Ore}}, + se.FieldPlan{Field: "MedicareNo", Name: "medicare_number", Context: label(t, "individuals/medicare_number").Context(), Terms: []se.TermKind{se.Equality, se.Ore}}, }, // After a database rename: the new column, the old identity. "renamed column": { []plan.RuleOption{plan.Column("medicare_num"), plan.Identity("medicare_number")}, - se.FieldPlan{Field: "MedicareNo", Name: "medicare_num", Context: "individuals/medicare_number", Terms: []se.TermKind{se.Equality, se.Ore}}, + se.FieldPlan{Field: "MedicareNo", Name: "medicare_num", Context: label(t, "individuals/medicare_number").Context(), Terms: []se.TermKind{se.Equality, se.Ore}}, }, } { individuals := plan.ForMessage(&individual{}, "individuals", plan.FirstOf( @@ -86,3 +86,13 @@ func TestPolicyPlanIsTheHandBuiltPlan(t *testing.T) { } } } + +// label is se.ParseLabel for a label the test knows to be valid. +func label(t testing.TB, s string) se.Label { + t.Helper() + l, err := se.ParseLabel(s) + if err != nil { + t.Fatal(err) + } + return l +} diff --git a/languages/golang/stackencrypt/record.go b/languages/golang/stackencrypt/record.go index 2f95f1847..716a71a66 100644 --- a/languages/golang/stackencrypt/record.go +++ b/languages/golang/stackencrypt/record.go @@ -23,36 +23,44 @@ import ( // // type User struct { // ID int64 `stash:"-"` // not sent to the guest -// Age uint32 `stash:"context=users/age,index=eq;ore"` // sealed + equality and ORE terms -// Email string `stash:"context=users/email,index=eq;match"` // sealed + equality and match terms -// Notes string `stash:"context=users/notes"` // sealed only +// Age uint32 `stash:"label=users/age,index=eq;ore"` // sealed + equality and ORE terms +// Email string `stash:"label=users/email,index=eq;match"` // sealed + equality and match terms +// Notes string `stash:"label=users/notes"` // sealed only // } // -// Options are comma-separated: `context=` (required for a planned -// field — the field's own context, a string part), `index=[;]` +// Options are comma-separated: the field's own context as either +// `label=
/` (a [Label], parsed with [ParseLabel]) or +// `context=` (one arbitrary text part, as [NewContext] makes it, what +// a Rust `#[stash(context = "..")]` literal binds) — exactly one of the two, +// required for a planned field; see [FieldPlan.Context] — `index=[;]` // (eq, match, ore, ope), and `name=` (the record key; the Go // field name otherwise). A field tagged `-` or `plain`, or not tagged at // all, is not part of the record: it never crosses the boundary, and stays // the caller's to store. Unexported fields are ignored. // -// The same plan, built by hand: +// The same plan, built by hand. Each field's context is a [Label], the +// table and the column; [ParseLabel] refuses a name that would not render +// as itself, so check its error: // +// age, err := stackencrypt.ParseLabel("users/age") +// email, err := stackencrypt.ParseLabel("users/email") +// notes, err := stackencrypt.ParseLabel("users/notes") // plan, err := stackencrypt.NewPlan( // stackencrypt.FieldPlan{ // Field: "Age", -// Context: "users/age", +// Context: age.Context(), // Terms: []stackencrypt.TermKind{ // stackencrypt.Equality, stackencrypt.Ore, // }, // }, // stackencrypt.FieldPlan{ // Field: "Email", -// Context: "users/email", +// Context: email.Context(), // Terms: []stackencrypt.TermKind{ // stackencrypt.Equality, stackencrypt.Match, // }, // }, -// stackencrypt.FieldPlan{Field: "Notes", Context: "users/notes"}, +// stackencrypt.FieldPlan{Field: "Notes", Context: notes.Context()}, // ) // records, err := cipher.EncryptRecords( // ctx, users, stackencrypt.WithPlan(plan), @@ -131,8 +139,8 @@ func (e contextExtension) applyTerm(o *termOptions) { e.appendTo(&o.extension) } // ExtendContext extends every field's context by parts, in order, the way // the Rust derive extends a field's context by the caller's -// (encrypt_into_with_context): a field tagged context=users/age with -// ExtendContext(uint64(7)) binds ["users/age", 7]. On [Cipher.Term] it +// (encrypt_into_with_context): a field tagged label=users/age with +// ExtendContext(uint64(7)) binds [["users", "age"], 7]. On [Cipher.Term] it // extends the probe's context the same way, so a probe built under the // extension a record was written under compares against that record's // terms, and under any other extension, or none, against nothing. @@ -202,9 +210,18 @@ type FieldPlan struct { // Name is the record key the field's outputs are stored under: the // column name, in EQL terms. Field when empty. Name string - // Context is the field's own encryption context, a string part; the - // record call extends it by any ExtendContext parts. Required. - Context string + // Context is the field's own encryption context; the record call + // extends it by any ExtendContext parts. Required. + // + // A field stored in a database is named by a [Label] — a table and a + // column, ParseLabel("users/age") then .Context(): the pair ["users", "age"] + // a Rust `#[derive(EncryptFrom)]` with `struct = .., context = "users"` + // binds its `age` field under, rendering the ZeroKMS descriptor + // users/age. Any other context is one [NewContext] makes: an arbitrary + // part, what a Rust `#[stash(context = "..")]` literal binds, rendered + // escaped if it would read as something else. A probe for the field + // ([Cipher.Term]) takes the same Context, so the two cannot drift. + Context Context // Terms lists the terms to derive beside the ciphertext, in order. Terms []TermKind } @@ -227,7 +244,7 @@ type planData struct { type planField struct { field string name string - context string + context Context terms []TermKind } @@ -243,7 +260,7 @@ func (f planField) outputs() []string { } // NewPlan validates the fields and returns the plan. Every field needs a -// Field and a Context; Go field names must be unique, and so must record +// Field and a non-zero Context; Go field names must be unique, and so must record // names (Name, or Field); Terms must be kinds this package defines, each // at most once per field. A plan is built once and reused across calls, // like the type it describes. @@ -272,7 +289,7 @@ func newPlan(fields []FieldPlan) (Plan, error) { return Plan{}, fmt.Errorf("plan field %s: the Go field is planned twice", f.Field) } seenField[f.Field] = true - if f.Context == "" { + if f.Context.isZero() { return Plan{}, fmt.Errorf("plan field %s: a planned field needs a context", f.Field) } pf := planField{field: f.Field, name: f.Field, context: f.Context} @@ -337,11 +354,27 @@ func PlanFromTags(t reflect.Type) (Plan, error) { continue } pf := FieldPlan{Field: f.Name, Name: f.Name} + var ownKey string // the option that set pf.Context, for the message on a second one for _, opt := range strings.Split(tag, ",") { key, value, _ := strings.Cut(opt, "=") switch key { - case "context": - pf.Context = value + case "label", "context": + if ownKey != "" { + return Plan{}, fmt.Errorf("stackencrypt: field %s.%s: %s= and %s= both given; a field has one own context", t, f.Name, ownKey, key) + } + ownKey = key + var err error + if key == "label" { + var l Label + if l, err = ParseLabel(value); err == nil { + pf.Context = l.Context() + } + } else { + pf.Context, err = NewContext(value) + } + if err != nil { + return Plan{}, fmt.Errorf("stackencrypt: field %s.%s: %s=%q: %w", t, f.Name, key, value, err) + } case "name": if value == "" { return Plan{}, fmt.Errorf("stackencrypt: field %s.%s: name must not be empty", t, f.Name) @@ -375,9 +408,9 @@ func PlanFromTags(t reflect.Type) (Plan, error) { // fieldPlan is one planned field bound to a struct type: the plan's field // resolved to its index. type fieldPlan struct { - index int // struct field index - name string // wire name - context string // the field's own context part + index int // struct field index + name string // wire name + context Context // the field's own context outputs []string } @@ -435,13 +468,10 @@ func planFor(t reflect.Type, o recordOptions) ([]fieldPlan, error) { func planValue(plan []fieldPlan, opts recordOptions) (vcvalue.Object, error) { out := make(vcvalue.Object, 0, len(plan)) for _, f := range plan { - ctx, err := NewContext(f.context) + ctx, err := extend(f.context, opts.extension) if err != nil { return nil, err } - if ctx, err = extend(ctx, opts.extension); err != nil { - return nil, err - } outputs := make([]any, len(f.outputs)) for i, o := range f.outputs { outputs[i] = o diff --git a/languages/golang/stackencrypt/unit_test.go b/languages/golang/stackencrypt/unit_test.go index 5b3a206fd..27f02d413 100644 --- a/languages/golang/stackencrypt/unit_test.go +++ b/languages/golang/stackencrypt/unit_test.go @@ -24,8 +24,8 @@ import ( func TestCommitRecordsPreservesRowsAndIsAtomic(t *testing.T) { type row struct { ID int64 `stash:"-"` - Age uint8 `stash:"context=users/age"` - Email string `stash:"context=users/email"` + Age uint8 `stash:"label=users/age"` + Email string `stash:"label=users/email"` } plan, err := planFor(reflect.TypeOf(row{}), recordOptions{}) if err != nil { @@ -183,9 +183,9 @@ func TestContextNestsToTheLeft(t *testing.T) { type taggedUser struct { ID int64 `stash:"-"` - Age uint32 `stash:"context=users/age,index=eq;ore"` - Email string `stash:"context=users/email,index=eq;match,name=email"` - Notes string `stash:"context=users/notes"` + Age uint32 `stash:"label=users/age,index=eq;ore"` + Email string `stash:"label=users/email,index=eq;match,name=email"` + Notes string `stash:"label=users/notes"` Plain string `stash:"plain"` NoTag string hidden string `stash:"context=x"` //nolint:unused // proves unexported fields are skipped @@ -197,9 +197,9 @@ func TestPlanFromTags(t *testing.T) { t.Fatal(err) } want := []fieldPlan{ - {index: 1, name: "Age", context: "users/age", outputs: []string{"c", "eq", "ore"}}, - {index: 2, name: "email", context: "users/email", outputs: []string{"c", "eq", "match"}}, - {index: 3, name: "Notes", context: "users/notes", outputs: []string{"c"}}, + {index: 1, name: "Age", context: label(t, "users/age").Context(), outputs: []string{"c", "eq", "ore"}}, + {index: 2, name: "email", context: label(t, "users/email").Context(), outputs: []string{"c", "eq", "match"}}, + {index: 3, name: "Notes", context: label(t, "users/notes").Context(), outputs: []string{"c"}}, } if !reflect.DeepEqual(plan, want) { t.Fatalf("plan = %+v\nwant %+v", plan, want) @@ -210,7 +210,7 @@ func TestPlanFromTags(t *testing.T) { t.Fatal(err) } age := obj[0].Value.(vcvalue.Object) - if got := age[0].Value; !reflect.DeepEqual(got, []any{"users/age", uint64(7)}) { + if got := age[0].Value; !reflect.DeepEqual(got, []any{[]any{"users", "age"}, uint64(7)}) { t.Fatalf("extended context = %v", got) } if _, err := vcffi.Marshal(obj); err != nil { @@ -252,9 +252,9 @@ func TestPlanFromTags(t *testing.T) { func TestExplicitPlanIsTheTagPlan(t *testing.T) { typ := reflect.TypeOf(taggedUser{}) explicit, err := NewPlan( - FieldPlan{Field: "Age", Context: "users/age", Terms: []TermKind{Equality, Ore}}, - FieldPlan{Field: "Email", Name: "email", Context: "users/email", Terms: []TermKind{Equality, Match}}, - FieldPlan{Field: "Notes", Context: "users/notes"}, + FieldPlan{Field: "Age", Context: label(t, "users/age").Context(), Terms: []TermKind{Equality, Ore}}, + FieldPlan{Field: "Email", Name: "email", Context: label(t, "users/email").Context(), Terms: []TermKind{Equality, Match}}, + FieldPlan{Field: "Notes", Context: label(t, "users/notes").Context()}, ) if err != nil { t.Fatal(err) @@ -296,8 +296,8 @@ func TestExplicitPlanIsTheTagPlan(t *testing.T) { t.Fatalf("bound plans differ:\n%+v\n%+v", viaOption, viaTags) } // Fields returns a copy. - explicit.Fields()[0].Context = "changed" - if explicit.Fields()[0].Context != "users/age" { + explicit.Fields()[0].Context = MustContext("changed") + if !explicit.Fields()[0].Context.Equal(label(t, "users/age").Context()) { t.Fatal("Fields exposed the plan's own slice") } } @@ -309,7 +309,7 @@ func TestExplicitPlanIsTheTagPlan(t *testing.T) { // extension's, which is what makes the match tenant-specific. func TestTermExtensionMatchesRecordFieldContext(t *testing.T) { type row struct { - Email string `stash:"context=users/email,index=eq"` + Email string `stash:"label=users/email,index=eq"` } ext := []any{uint64(7), "eu"} o := applyOptions([]RecordOption{ExtendContext(ext...)}) @@ -329,17 +329,17 @@ func TestTermExtensionMatchesRecordFieldContext(t *testing.T) { var to termOptions ExtendContext(ext...).applyTerm(&to) - probe, err := extend(MustContext("users/email"), to.extension) + probe, err := extend(label(t, "users/email").Context(), to.extension) if err != nil { t.Fatal(err) } if !reflect.DeepEqual(probe.value(), fieldContext) { t.Fatalf("probe context %#v, record field context %#v", probe.value(), fieldContext) } - if reflect.DeepEqual(MustContext("users/email").value(), fieldContext) { + if reflect.DeepEqual(label(t, "users/email").Context().value(), fieldContext) { t.Fatal("the unextended probe context equals the extended field's") } - other, err := extend(MustContext("users/email"), []any{uint64(8), "eu"}) + other, err := extend(label(t, "users/email").Context(), []any{uint64(8), "eu"}) if err != nil { t.Fatal(err) } @@ -366,7 +366,7 @@ func TestTermExtensionMatchesRecordFieldContext(t *testing.T) { // and probes under different contexts with no error. func TestSeveralExtensionsJoinInOrder(t *testing.T) { type row struct { - Email string `stash:"context=users/email,index=eq"` + Email string `stash:"label=users/email,index=eq"` } typ := reflect.TypeOf(row{}) fieldContext := func(opts ...RecordOption) any { @@ -388,7 +388,7 @@ func TestSeveralExtensionsJoinInOrder(t *testing.T) { for _, opt := range opts { opt.applyTerm(&to) } - c, err := extend(MustContext("users/email"), to.extension) + c, err := extend(label(t, "users/email").Context(), to.extension) if err != nil { t.Fatal(err) } @@ -462,7 +462,7 @@ func TestPlanBindsByFieldName(t *testing.T) { if _, err := PlanFromTags(typ); err == nil { t.Fatal("untagged struct has a tag plan") } - ok, err := NewPlan(FieldPlan{Field: "Email", Context: "c"}) + ok, err := NewPlan(FieldPlan{Field: "Email", Context: MustContext("c")}) if err != nil { t.Fatal(err) } @@ -478,7 +478,7 @@ func TestPlanBindsByFieldName(t *testing.T) { "unexported": "hidden", "promoted": "Inner", } { - p, err := NewPlan(FieldPlan{Field: field, Context: "c"}) + p, err := NewPlan(FieldPlan{Field: field, Context: MustContext("c")}) if err != nil { t.Fatal(err) } @@ -494,12 +494,12 @@ func TestPlanBindsByFieldName(t *testing.T) { func TestNewPlanRefusesMalformedFields(t *testing.T) { for name, fields := range map[string][]FieldPlan{ "no fields": nil, - "no field name": {{Context: "c"}}, + "no field name": {{Context: MustContext("c")}}, "no context": {{Field: "A"}}, - "unknown kind": {{Field: "A", Context: "c", Terms: []TermKind{TermKind(9)}}}, - "duplicate name": {{Field: "A", Context: "c", Name: "x"}, {Field: "B", Context: "c", Name: "x"}}, - "field twice": {{Field: "A", Name: "x", Context: "c"}, {Field: "A", Name: "y", Context: "d"}}, - "term twice": {{Field: "A", Context: "c", Terms: []TermKind{Equality, Equality}}}, + "unknown kind": {{Field: "A", Context: MustContext("c"), Terms: []TermKind{TermKind(9)}}}, + "duplicate name": {{Field: "A", Context: MustContext("c"), Name: "x"}, {Field: "B", Context: MustContext("c"), Name: "x"}}, + "field twice": {{Field: "A", Name: "x", Context: MustContext("c")}, {Field: "A", Name: "y", Context: MustContext("d")}}, + "term twice": {{Field: "A", Context: MustContext("c"), Terms: []TermKind{Equality, Equality}}}, } { if _, err := NewPlan(fields...); err == nil { t.Errorf("%s: plan accepted", name) diff --git a/packages/stack-encrypt-derive/Cargo.toml b/packages/stack-encrypt-derive/Cargo.toml index c075a8436..223cd01bc 100644 --- a/packages/stack-encrypt-derive/Cargo.toml +++ b/packages/stack-encrypt-derive/Cargo.toml @@ -1,7 +1,7 @@ [package] name = "stack-encrypt-derive" description = "Derive macros for stack-encrypt's target-directed encryption" -version = "0.1.0" +version = "0.2.0" edition.workspace = true authors.workspace = true repository.workspace = true @@ -27,6 +27,9 @@ syn = { version = "3", features = ["full", "extra-traits"] } development = ["stack-encrypt", "tokio"] [dev-dependencies] +# The plain-segment rule is copied from stack-encrypt (a proc-macro crate cannot +# depend on the crate it serves); its test reads the fixture both suites share. +serde_json = { workspace = true } # The crate-level examples are real doctests, run against the fake key # source. A dev-dependency cycle back to `stack-encrypt` is the usual shape # for a derive crate (cf. serde_derive -> serde). The fake key source comes diff --git a/packages/stack-encrypt-derive/docs/attributes.md b/packages/stack-encrypt-derive/docs/attributes.md index 93d6fac9c..1948d0d8e 100644 --- a/packages/stack-encrypt-derive/docs/attributes.md +++ b/packages/stack-encrypt-derive/docs/attributes.md @@ -96,21 +96,30 @@ but each listed type only once). A `struct = ..` derive needs no attribute on its fields. With `#[stash(struct = User, context = "users")]`, a field `age` is derived from -`user.age` under the context `"users/age"`; a field `email` from `user.email` -under `"users/email"`; a tuple struct's `.0` under `"users/0"`. The first -half is the container's `context` and the second the *plaintext* field's -name, so `#[stash(from = email_address)] email: ..` is derived under -`"users/email_address"`: both halves name the stored field, not the -encrypted struct. Nothing is pluralised or otherwise guessed. `context = -".."` on a field is taken verbatim and replaces the inferred one; `nested` -on a field infers none — the field is handed the caller's context as it is, -which a nested `struct` derive (carrying its own contexts) composes with -them and a leaf accepts only as a `NonEmpty`. +`user.age` under the **pair** `("users", "age")` — two context parts, which +render the ZeroKMS descriptor `users/age`; a field `email` from `user.email` +under `("users", "email")`. The first part is the container's `context` and +the second the *plaintext* field's name. Both must be plain descriptor +segments (no `/`, `(`, `)`, control or invisible character; not beginning +with `b64:`, a digit or `-`), or the descriptor would render escaped and the +ZeroKMS log would not name the column: the derive refuses a prefix such as +`"public/users"` (write `"users"`), and a tuple field — whose index begins +with a digit — must carry its own `context = ".."`. So `#[stash(from = email_address)] email: ..` is derived under +`("users", "email_address")`: both parts name the stored field, not the +encrypted struct. Nothing is pluralised or otherwise guessed. The pair is +what `nonempty!("users").with("age")` spells at a call site, and what a +two-segment `Label` spells. A `context = ".."` literal on a field is **one** +text part, taken exactly as written, and replaces the inferred pair: the +literal `"users/age"` is not the pair, and renders escaped (`b64:…`) in the +descriptor because a `/` inside one part must never read as a separator. +`nested` on a field infers none — the field is handed the caller's context +as it is, which a nested `struct` derive (carrying its own contexts) composes +with them and a leaf accepts only as a `NonEmpty`. A context passed by the caller extends every field's: under `user.encrypt_into_with_context(&keyset, 7u64)` the `age` field is derived -under `("users/age", 7u64)`, and a query site probes it under -`nonempty!("users/age").with(7u64)`. This is how a field is bound to its +under `(("users", "age"), 7u64)`, and a query site probes it under +`nonempty!("users").with("age").with(7u64)`. This is how a field is bound to its record as well as its name — a record id, say — without the type having to know the id. Decryption takes the same extension. The extension may be borrowed at the call site: its Vitamin C encodings are owned by the declaration before execution. @@ -128,8 +137,10 @@ stored data stops decrypting — `Error::Kms` against ZeroKMS, which refuses the key retrieval under the changed descriptor before the AEAD runs, and `Error::Aead` under a key source that ignores descriptors, such as the fake one in tests — silently at the call site, with no compile-time signal. -Before such a rename, pin the old value with `context = ".."` on the fields -it reaches. +Before such a rename, keep the old plaintext field name in `from` on the +field it reaches (`#[stash(from = old_name)] new_name: ..`), which keeps the +inferred pair. A `context = ".."` literal cannot preserve it: a literal is one +part, and an inferred context is two. A `struct` derive has no field derived from the whole plaintext, and a `plaintext` record has none derived from a field of it: `from` and `nested` diff --git a/packages/stack-encrypt-derive/src/attrs.rs b/packages/stack-encrypt-derive/src/attrs.rs index 0caf6183b..0aea17c84 100644 --- a/packages/stack-encrypt-derive/src/attrs.rs +++ b/packages/stack-encrypt-derive/src/attrs.rs @@ -19,7 +19,7 @@ pub(crate) struct ContainerAttrs { /// field says otherwise. Exclusive with `plaintext`; requires `context`. pub(crate) by_field: Option, /// `#[stash(context = "...")]` on the container: the first half of every - /// field's inferred context — `"/"`. Names the stored + /// field's inferred context, the pair `("", "")`. Names the stored /// data, not the Rust type: it is part of the stored data's identity, so /// it is given explicitly rather than inferred from a name a refactor /// can change. Only meaningful with `struct`. @@ -149,7 +149,7 @@ impl ContainerAttrs { by_field, "`struct = ..` needs a `context = \"..\"` beside it naming the stored data \ (e.g. `#[stash(struct = User, context = \"users\")]`): each field is derived \ - under `\"/\"`, and the prefix is part of the stored data's \ + under the pair `(\"\", \"\")`, and the prefix is part of the stored data's \ identity, so it is given explicitly rather than inferred from the Rust \ type's name", )); @@ -172,6 +172,16 @@ impl ContainerAttrs { data (e.g. \"users\")", )); } + if !is_plain_segment(&context.value()) { + return Err(syn::Error::new( + context.span(), + "a container `context` is the first segment of every field's ZeroKMS \ + descriptor, so it must be plain: no `/`, `(`, `)`, control or invisible \ + character, and not beginning with `b64:`, a digit or `-`; otherwise it \ + would render escaped and the log would not name the table. For the table \ + `public.users` write `context = \"users\"`", + )); + } } if let (Some(by_field), Some(context_type)) = (&by_field, &context_type) { @@ -299,3 +309,57 @@ impl FieldAttrs { Ok(parsed) } } + +/// Whether `text` renders verbatim in a ZeroKMS descriptor: the plain-segment +/// rule of `stack_encrypt::Label`, copied here because a proc-macro crate +/// cannot depend on the crate it serves. The test below reads the fixture the +/// Rust and Go suites share, so this copy cannot drift from them. +pub(crate) fn is_plain_segment(text: &str) -> bool { + const INVISIBLE: &[char] = &[ + '\u{00AD}', '\u{061C}', '\u{180E}', '\u{200B}', '\u{200C}', '\u{200D}', '\u{200E}', + '\u{200F}', '\u{202A}', '\u{202B}', '\u{202C}', '\u{202D}', '\u{202E}', '\u{2060}', + '\u{2061}', '\u{2062}', '\u{2063}', '\u{2064}', '\u{2066}', '\u{2067}', '\u{2068}', + '\u{2069}', '\u{FEFF}', + ]; + !text.is_empty() + && !text.starts_with("b64:") + && !text.starts_with(|c: char| c.is_ascii_digit() || c == '-') + && !text + .chars() + .any(|c| c.is_control() || INVISIBLE.contains(&c) || matches!(c, '/' | '(' | ')')) +} + +#[cfg(test)] +mod plain_segment_tests { + use super::is_plain_segment; + + /// The derive is the third reader of `label_segments.json`, beside the + /// stack-encrypt and Go suites: one list of what is plain. + #[test] + fn the_copied_rule_matches_the_shared_fixture() { + let json: serde_json::Value = serde_json::from_str(include_str!( + "../../stack-encrypt/tests/fixtures/label_segments.json" + )) + .expect("a valid fixture"); + let list = |key: &str| { + json[key] + .as_array() + .expect("an array") + .iter() + .map(|v| v.as_str().expect("a string").to_owned()) + .collect::>() + }; + let (plain, not_plain) = (list("plain"), list("not_plain")); + assert!(!plain.is_empty() && !not_plain.is_empty()); + for text in &plain { + assert!(is_plain_segment(text), "{text:?}"); + } + for text in ¬_plain { + assert!(!is_plain_segment(text), "{text:?}"); + } + // The names the derive infers are Rust identifiers, which are plain + // unless raw; an index is not. + assert!(is_plain_segment("email_address")); + assert!(!is_plain_segment("0")); + } +} diff --git a/packages/stack-encrypt-derive/src/lib.rs b/packages/stack-encrypt-derive/src/lib.rs index e84243b81..fc394d3e1 100644 --- a/packages/stack-encrypt-derive/src/lib.rs +++ b/packages/stack-encrypt-derive/src/lib.rs @@ -35,7 +35,9 @@ //! `context_type = AeadContext` and accept a context type that implements //! `IntoAad` alone, as the ciphertext leaf itself does. //! `struct = User, context = "users"` selects plaintext fields -//! and binds them under `"users/"`. The storage envelope itself adds no +//! and binds each under the pair `("users", "")`, which renders +//! `users/` as its ZeroKMS descriptor. A `context = ".."` literal on a +//! field is one text part, exactly as written. The storage envelope itself adds no //! cryptographic map-entry context. Vitamin C still binds keys inside plaintext //! maps and preserves authenticated absence and empty-container markers. //! diff --git a/packages/stack-encrypt-derive/src/shape.rs b/packages/stack-encrypt-derive/src/shape.rs index f61a1e2a2..7c4ee4931 100644 --- a/packages/stack-encrypt-derive/src/shape.rs +++ b/packages/stack-encrypt-derive/src/shape.rs @@ -64,9 +64,9 @@ pub(crate) enum Kind { /// Derived from the source through the field type's own `EncryptFrom`. Derived { /// This field's own context, if it has one: a `#[stash(context = - /// "...")]` literal, or the `"/"` a `struct` derive + /// "...")]` literal, or the `(prefix, field)` pair a `struct` derive /// infers. A context the caller passes extends it either way. - context: Option, + context: Option, /// With `struct = ..`: the plaintext field this one is derived /// from — its own name, or the `#[stash(from = field)]` override. /// `None` for a `plaintext` record, whose fields are all derived @@ -96,7 +96,7 @@ impl Field { /// /// A field with a context of its own — a literal, or the one a `struct` /// derive infers — is derived under it as it is when the caller passes - /// `()`, and under it *extended* with the caller's (`("users/age", id)`) + /// `()`, and under it *extended* with the caller's (`(("users", "age"), id)`) /// when the caller passes a `NonEmpty<_>`. A field with none is handed /// the caller's context as it is, and its type decides what that means: /// a nested `struct` derive composes it with its own contexts; a leaf @@ -118,13 +118,37 @@ impl Field { } } +/// A derived field's own context. +#[cfg_attr(test, derive(Debug))] +pub(crate) enum OwnContext { + /// `#[stash(context = "...")]`: one text part, exactly as written. + Literal(LitStr), + /// What a `struct` derive infers: the pair (container `context` prefix, + /// plaintext field name), two parts, so it renders `prefix/field` without + /// the field name having to be joined into, or kept out of, a string. The + /// derive knows no tables (ADR-0003); a consumer whose prefix is a table + /// gets EQL's `(table, column)` shape from it. + Prefixed { prefix: LitStr, field: LitStr }, +} +impl OwnContext { + /// The `NonEmpty` the derive hands `under` / `extend`. + fn expr(&self, krate: &Path) -> TokenStream { + match self { + Self::Literal(lit) => quote!(#krate::nonempty!(#lit)), + Self::Prefixed { prefix, field } => { + quote!(#krate::nonempty!(#prefix).with(#field)) + } + } + } +} + /// Where a derived field's context comes from. See [`Field::field_context`]. #[cfg_attr(test, derive(Debug))] pub(crate) enum FieldContext<'a> { /// A context of the field's own — `#[stash(context = "...")]`, or the - /// `"/"` a `struct` derive infers: as it is under `()`, + /// `(prefix, field)` pair a `struct` derive infers: as it is under `()`, /// extended with the caller's context under `NonEmpty<_>`. - Own(&'a LitStr), + Own(&'a OwnContext), /// No context of its own: handed the caller's as it is — `()`, or the /// record's associated context. Caller, @@ -270,7 +294,7 @@ pub(crate) fn trait_impl( /// The fields, with what a `struct` derive (`prefix` is the container's /// `context`) fills in: `from` is the field's own name and `context` is -/// `"/"`, each unless the field gives its own. +/// the pair `("", "")`, each unless the field gives its own. /// `#[stash(nested)]` opts a field out of the inferred context — it is handed /// the caller's as it is, which a nested `struct` derive (a type carrying its /// own contexts) composes with them and a leaf accepts only as a @@ -311,8 +335,8 @@ fn collect(fields: &Fields, prefix: Option<&LitStr>) -> Result> { if context.value().is_empty() { let message = if prefix.is_some() { "an empty `context` is rejected when a value is encrypted: name the \ - field (e.g. \"users/email\"), or drop the attribute to use the inferred \ - `\"/\"`" + field (e.g. \"email\"), or drop the attribute to use the inferred \ + pair `(\"\", \"\")`" } else { "an empty `context` is rejected when a value is encrypted: name the \ field (e.g. \"users/email\"), or drop the attribute to hand the field \ @@ -359,14 +383,45 @@ fn collect(fields: &Fields, prefix: Option<&LitStr>) -> Result> { // is handed the caller's (`FieldContext::Caller`). None } else if let Some(lit) = attrs.context { - Some(lit) + Some(OwnContext::Literal(lit)) } else { - let column = match &from { + // The inferred second segment must render + // verbatim, or the descriptor would not name + // the field. A named field is a Rust identifier + // and plain unless raw; a tuple index begins with + // a digit, which the descriptor reserves. + let field = match &from { Member::Named(ident) => ident.to_string(), - Member::Unnamed(index) => index.index.to_string(), + Member::Unnamed(index) => { + return Err(syn::Error::new( + member.span(), + format!( + "a tuple field has no name to infer a context \ + from: its index `{0}` begins with a digit, \ + which a descriptor reserves, so `(\"{1}\", \ + \"{0}\")` would render escaped; give the \ + field `#[stash(context = \"..\")]`", + index.index, + prefix.value() + ), + )); + } }; - let prefix = prefix.value(); - Some(LitStr::new(&format!("{prefix}/{column}"), member.span())) + if !crate::attrs::is_plain_segment(&field) { + return Err(syn::Error::new( + member.span(), + format!( + "the field name `{field}` is not a plain descriptor \ + segment, so `(\"{}\", \"{field}\")` would render \ + escaped; give the field `#[stash(context = \"..\")]`", + prefix.value() + ), + )); + } + Some(OwnContext::Prefixed { + prefix: prefix.clone(), + field: LitStr::new(&field, member.span()), + }) }; Kind::Derived { context, @@ -374,7 +429,7 @@ fn collect(fields: &Fields, prefix: Option<&LitStr>) -> Result> { } } None => Kind::Derived { - context: attrs.context, + context: attrs.context.map(OwnContext::Literal), from: None, }, }, @@ -482,12 +537,13 @@ impl Record { let krate = &self.krate; let span = field.ty.span(); match field.field_context() { - FieldContext::Own(lit) => { + FieldContext::Own(own) => { + let own = respan(own.expr(krate), span); if self.declared_contexts() { - quote_spanned!(span=> .under(#krate::nonempty!(#lit))) + quote_spanned!(span=> .under(#own)) } else { let threaded = respan(self.threaded_context().into_token_stream(), span); - quote_spanned!(span=> .extend::<#threaded>(#krate::nonempty!(#lit))) + quote_spanned!(span=> .extend::<#threaded>(#own)) } } FieldContext::Caller => { @@ -530,13 +586,14 @@ impl Record { let krate = &self.krate; match field.field_context() { FieldContext::Caller => quote!(::core::clone::Clone::clone(&__context)), - FieldContext::Own(lit) => { + FieldContext::Own(own) => { let method = if self.declared_contexts() { quote!(under) } else { quote!(extend) }; - quote!(::core::clone::Clone::clone(&__context).#method(#krate::nonempty!(#lit))) + let own = own.expr(krate); + quote!(::core::clone::Clone::clone(&__context).#method(#own)) } } } @@ -584,7 +641,10 @@ mod tests { /// The field's own context, for assertions. fn own(field: &Field) -> String { match field.field_context() { - FieldContext::Own(lit) => lit.value(), + FieldContext::Own(OwnContext::Literal(lit)) => lit.value(), + FieldContext::Own(OwnContext::Prefixed { prefix, field }) => { + format!("({}, {})", prefix.value(), field.value()) + } other => panic!("expected a context of the field's own, got {other:?}"), } } @@ -824,10 +884,10 @@ mod tests { ); // Own name under the container's prefix. assert!(matches!(age.from(), Some(Member::Named(m)) if m == "age")); - assert_eq!(own(age), "user_profiles/age"); + assert_eq!(own(age), "(user_profiles, age)"); // `from` overrides the field; the context follows the plaintext field. assert!(matches!(email.from(), Some(Member::Named(m)) if m == "email_address")); - assert_eq!(own(email), "user_profiles/email_address"); + assert_eq!(own(email), "(user_profiles, email_address)"); // `context` is taken verbatim; like the inferred ones, the caller's // context extends it. assert!(matches!(name.from(), Some(Member::Named(m)) if m == "name")); @@ -839,15 +899,42 @@ mod tests { } #[test] - fn a_tuple_struct_is_reached_and_named_by_index() { - let record = parse(parse_quote! { + fn a_tuple_struct_is_reached_by_index_and_must_name_its_contexts() { + // An index is no name for a context: it begins with a digit, which a + // descriptor reserves, so the bare form is refused… + let err = parse(parse_quote! { #[stash(struct = Reading, context = "readings")] struct EncryptedReading(EncryptedAge, StackCipherText); }) + .unwrap_err(); + assert!( + err.to_string().contains("index `0` begins with a digit"), + "{err}" + ); + // …and each field names its own, still reached by index. + let record = parse(parse_quote! { + #[stash(struct = Reading, context = "readings")] + struct EncryptedReading( + #[stash(context = "reading_value")] EncryptedAge, + #[stash(context = "reading_unit")] StackCipherText, + ); + }) .unwrap(); assert!(matches!(record.fields[1].from(), Some(Member::Unnamed(i)) if i.index == 1)); - assert_eq!(own(&record.fields[0]), "readings/0"); - assert_eq!(own(&record.fields[1]), "readings/1"); + assert_eq!(own(&record.fields[0]), "reading_value"); + assert_eq!(own(&record.fields[1]), "reading_unit"); + } + + #[test] + fn a_container_prefix_that_is_not_plain_is_refused() { + let err = parse(parse_quote! { + #[stash(struct = User, context = "public/users")] + struct Encrypted { + email: StackCipherText, + } + }) + .unwrap_err(); + assert!(err.to_string().contains("must be plain"), "{err}"); } #[test] diff --git a/packages/stack-encrypt/CHANGELOG.md b/packages/stack-encrypt/CHANGELOG.md index 5bff1e910..0ad2ac34d 100644 --- a/packages/stack-encrypt/CHANGELOG.md +++ b/packages/stack-encrypt/CHANGELOG.md @@ -5,7 +5,37 @@ All notable changes to this project will be documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). -## [Unreleased] +## [0.2.0] - 2026-10-04 + +### Breaking + +- **A column has one encryption context, and its ZeroKMS descriptor renders + `users/email`.** `Descriptor::SEPARATOR` is `/` (was `|`); a text part that + contains `/` is escaped with URL-safe base64 (the standard alphabet contains + `/`). A data key minted through 0.1.0 was bound to the old rendering of the + same context and cannot be retrieved under this one. 0.1.0 had one known + consumer, aware of this; see ADR-0006. +- `#[derive(EncryptFrom)]` with `struct = User, context = "users"` binds each + field under the pair `("users", "")`, the same context EQL's + `Identifier` is. A `#[stash(context = "…")]` literal on a field stays one + text part, exactly as written. +- `Encryption::under` / `extend` and the context types accept any + `NonEmpty` as the own context, not only a static string. + +### Added + +- `Describe` and `Description`: a value whose parts are the descriptor of + the data it keys. `to_context` is what the type's `IntoContext` returns, so + the AAD and the ZeroKMS descriptor are one tree seen two ways. EQL's + `Identifier` (table, column) implements it in `eql-bindings`. +- `Label` and `LabelError`: a path of plain segments written and read as + `users/email`, for direct consumers. One segment is the same context as the + bare literal; two are the pair a `struct = ..` derive binds. + +## [0.1.0] - 2026-10-04 + +The first crates.io release. Everything below was in it; the heading was +added after the fact — this section said "Unreleased" when 0.1.0 shipped. ### Breaking diff --git a/packages/stack-encrypt/CONTEXT.md b/packages/stack-encrypt/CONTEXT.md index 3459d3bf6..166ee15e2 100644 --- a/packages/stack-encrypt/CONTEXT.md +++ b/packages/stack-encrypt/CONTEXT.md @@ -3,7 +3,7 @@ Client-side encryption of values under per-value ZeroKMS data keys, and the derivation of searchable index terms from the same values. Covers `stack-encrypt`, `stack-encrypt-derive`, and the WASI guest in -`bindings/go/stackencrypt/guest` that exposes them to Go. +`languages/golang/stackencrypt/guest` that exposes them to Go. ## Language @@ -46,15 +46,17 @@ leaf requires a nonempty context, validated by Vitamin C and owned in a record deriving terms threads to every field) or an `AeadContext` (the AAD encoding alone — what a ciphertext is sealed and opened under; a record deriving terms hands its ciphertext fields that half of its `CallerContext`); -a `nonempty!("users/email")` literal, a `NonEmpty::new(value)?` at runtime, -or a bare integer. It becomes the ciphertext's associated data, +a `nonempty!("users").with("email")` pair (a table and a column are two +parts, rendered `users/email`), a `NonEmpty::new(value)?` at runtime, or a +bare integer. It becomes the ciphertext's associated data, the term's PRF context, and the ZeroKMS descriptor of the data key. _Avoid_: AAD (that is one of its encodings, not the concept), lock context **Own context**: -The context a field carries itself: a `context = ".."` literal, or the one a -`struct = ..` derive infers as `/`. A caller's context -*extends* it (`("users/age", id)`); it is never discarded. A subtree of a +The context a field carries itself: a `context = ".."` literal (one text +part, exactly as written), or the pair a `struct = ..` derive infers, +`(, )`. A caller's context *extends* it +(`(("users", "age"), id)`); it is never discarded. A subtree of a declaration is given one with `under` (the caller's is then optional) or `extend` (the caller's stays required). _Avoid_: default context, field prefix @@ -72,13 +74,43 @@ _Avoid_: scope (that is a `Pending`'s), shared context, per-operation context **Descriptor**: The context, rendered as the string ZeroKMS binds into every data key and logs per retrieval, rendered from the context's parts: plain text verbatim, -integers by their width, sign-blind (`7u64`, and `7i64` is `7u64`), a -composite's parts joined by `|` (`users/email|7u64`); text that could read as -another form is `b64:`-escaped, and an empty part inside a list is the bare -`b64:`. Injective over encodings, and finer than them for a pre-encoded -`Aad` (opaque bytes) and for shapes that encode alike (`None` vs `0u64`): -seal and open must present the context in the same shape. -_Avoid_: key name, key id +integers by their width, sign-blind (`7u64`, and `7i64` is `7u64`), a list's +parts joined by `/` (`users/email`; a nested list is parenthesised, +`(users/email)/7u64`); text that could read as another form — containing +`/`, `(` or `)`, beginning with `b64:`, a digit or `-` — is `b64:`-escaped, +so one text part can never read as two, and an empty part inside a list is +the bare `b64:`. Rendered by one function, `Descriptor::from_piece`, and +**frozen**: a change re-keys everything. Finer than the encodings for a +pre-encoded `Aad` (opaque bytes) and coarser for shapes that render alike +(`7i64` and `7u64`): seal and open must present the context in the same +shape. The descriptor is derived, never authored: nothing takes a descriptor +string from a caller. +_Avoid_: key name, key id, path (that is a `Label`) + +**Describe**: +The trait of a value whose parts are a descriptor of its own — the identity +data is keyed under, as opposed to an arbitrary context. An implementor +returns its parts as a `Description`, built from a first part so it is never +empty, and never writes rendered text, so the one renderer keeps distinct +values apart whoever implements it. Open: a consumer's own column or +document type implements it; `Label` does, and EQL's `Identifier` does once +stack#971 lands. A `Describe` type is also a context, through the same parts +(`to_context` is what its `IntoContext` returns). +_Avoid_: descriptor trait, Descriptor (the rendered string), DescriptorBuilder + +**Label**: +The *name* of the data a value is sealed under (`users/email`, +`documents/v2/body`): the first-class `Describe` type, a path of plain +segments, each checked (non-empty; no `/`, `(`, `)`, control or invisible +format characters; not beginning with `b64:`, a digit or `-`), so its `Display` is its +descriptor and parses back losslessly. A context carries a name and, +optionally, a *scope* (a tenant, a row id), and each has one spelling: the +name is a `Label`, a flat list; the scope is `with`, which appends and nests +(`(users/email)/7u64`). Never build a name with `with` or put a scope in a +`Label`. A two-segment label is the context a `struct = ..` derive binds for +a field, so a label opens a row a derive wrote. An EQL consumer names its +data with an `Identifier`, a two-segment label. +_Avoid_: identifier (that is EQL's two-segment case), prefix, column context **Leaf**: An output type that authenticates or derives directly — a ciphertext or a diff --git a/packages/stack-encrypt/Cargo.toml b/packages/stack-encrypt/Cargo.toml index 509e86b3a..eaf3e1425 100644 --- a/packages/stack-encrypt/Cargo.toml +++ b/packages/stack-encrypt/Cargo.toml @@ -1,7 +1,7 @@ [package] name = "stack-encrypt" description = "Encrypt Rust values under per-value ZeroKMS data keys via the vitaminc cipher traits" -version = "0.1.0" +version = "0.2.0" edition.workspace = true authors.workspace = true repository.workspace = true @@ -22,7 +22,7 @@ stack-kms = { path = "../stack-kms", version = "0.1.0", default-features = false # return type names the auto-detected auth strategy. Only needed with `http`. stack-auth = { workspace = true, optional = true } # `#[derive(EncryptFrom)]` / `#[derive(DecryptInto)]`, re-exported from `target`. -stack-encrypt-derive = { path = "../stack-encrypt-derive", version = "0.1.0" } +stack-encrypt-derive = { path = "../stack-encrypt-derive", version = "0.2.0" } # The workspace vitaminc (0.5.0): one canonical context encoding shared by # the AEAD and PRF derivations (`IntoContext`, cipherstash/vitaminc#339), diff --git a/packages/stack-encrypt/docs/adr/0004-one-context-per-target-threaded-through-the-declaration-tree.md b/packages/stack-encrypt/docs/adr/0004-one-context-per-target-threaded-through-the-declaration-tree.md index 74cf00f66..8458d34c3 100644 --- a/packages/stack-encrypt/docs/adr/0004-one-context-per-target-threaded-through-the-declaration-tree.md +++ b/packages/stack-encrypt/docs/adr/0004-one-context-per-target-threaded-through-the-declaration-tree.md @@ -1,6 +1,7 @@ --- status: accepted date: 2026-09-13 +revised: 2026-10-04 extends: ADR-0003 --- @@ -90,7 +91,7 @@ A record gives its fields different contexts once each, visibly, instead of threading six arguments: ```rust -age.under(nonempty!("users/age")).zip(email.under(nonempty!("users/email"))) +age.under(nonempty!("users").with("age")).zip(email.under(nonempty!("users").with("email"))) ``` `under` gives a subtree a context of its own, which a caller's context @@ -101,8 +102,15 @@ extended by the caller's, which stays required. `under` is available wherever a `CallerContext` can become what the subtree needs, and `extend` wherever the caller's context — a `CallerContext`, or an `AeadContext` for a record that only seals — can; so a subtree may itself be a record whose own contexts -a caller's extends. An own context is a `NonEmpty<&'static str>`, so an empty -one is refused at compile time rather than at the first encryption. +a caller's extends. An own context is any `NonEmpty`: a +`nonempty!` literal, the pair `nonempty!("users").with("age")` a `struct` +derive infers for a field, or a `Label`. Each is nonempty by construction — +a literal at compile time, the others when they are built — so an empty one +is refused before the first encryption. (Revised 2026-10-04: this was +`NonEmpty<&'static str>`, and the examples spelled a column as the joined +literal `"users/age"`. A column is the pair now, which renders the ZeroKMS +descriptor `users/age`; the joined literal is one part and renders escaped. +See ADR-0006.) Two further combinators change nothing about *which* context reaches a subtree, only its type at the root. `accepting` converts the context a record @@ -119,7 +127,7 @@ not when it is built. - `ciphertext()` is `Encryption<.., AeadContext>` and `equality()` is `Encryption<.., CallerContext>` — each needs a real context -- `.under(nonempty!("users/age"))` yields `Encryption<.., DeclaredContext>` — +- `.under(nonempty!("users").with("age"))` yields `Encryption<.., DeclaredContext>` — now runnable under `()` or a caller's context - zipping a bare leaf with own-context fields is a type error, which is correct diff --git a/packages/stack-encrypt/docs/adr/0006-descriptors-render-with-a-slash-and-describe-is-open.md b/packages/stack-encrypt/docs/adr/0006-descriptors-render-with-a-slash-and-describe-is-open.md new file mode 100644 index 000000000..6b55b4e1f --- /dev/null +++ b/packages/stack-encrypt/docs/adr/0006-descriptors-render-with-a-slash-and-describe-is-open.md @@ -0,0 +1,110 @@ +--- +status: accepted +date: 2026-10-04 +extends: ADR-0004 +--- + +# Descriptors render with `/`, a column is a pair, and `Describe` is open + +The ZeroKMS descriptor is the string a data-key request carries and the one +ZeroKMS binds into the key tag and writes in its retrieval log. Stack Encrypt +renders it from a context's parts, and the module docs call the rendering +frozen, because a change strands every key issued under the old one. This +ADR records a change to that rendering, made on purpose, and the two +abstractions that came with it. + +## The problem + +Three spellings of "the `users.email` column" were in use at once: + +- The `#[derive(EncryptFrom)]` macro's `struct = User, context = "users"` + form joined the prefix and the field name into one text part, + `"users/email"`. +- The Go binding took the same joined string from a struct tag and sent it as + one part. +- EQL's `eql-bindings` (#971) bound the pair `("users", "email")`, because + EQL's `Identifier` *is* a table and a column. + +One part and two parts are different contexts, so the Rust derive, the Go +binding and EQL could not read each other's rows or match each other's +search terms, and nothing failed until someone tried. On the descriptor side +the pair rendered `users|email` with `|` as the list separator, which read +nothing like the column it named. + +Underneath, two concepts were being conflated: a *context* (any parts a +caller seals under, arbitrary) and an *identifier* (what the data is, with a +fixed shape). Issue #1049 and the review of #1050 named the split. + +## Options considered + +1. **Keep `|`, keep the joined string.** Make the derive and Go the standard + and have EQL join its identifier into one part. Rejected: EQL's + identifier is structurally two things, and joining them means a `/` in a + table or column name silently changes which column a value belongs to. +2. **The pair everywhere, `/` as the separator, no escaping.** Readable, but + a text part containing `/` would render exactly like two parts: the + literal `"users/email"` and the pair `("users", "email")` would derive + the same key. +3. **The pair everywhere, `/` as the separator, and escape any text part + that could read as another form.** A part containing `/`, `(`, `)`, a + control character or an invisible format character (a zero-width or + bidirectional mark, which would print as another name in the log), or + beginning with `b64:`, a digit or `-`, renders as + `b64:` plus URL-safe base64. The standard base64 alphabet contains `/`, + so it could not be used. + +## Decision + +Option 3. In detail: + +- `Descriptor::SEPARATOR` is `/`. Escaped parts use URL-safe base64. The + rendering is otherwise unchanged and remains frozen from this point. +- A column is the pair. The derive's `struct = .., context = ""` + form binds a field under `("", "")`. A `context = ".."` + literal on a field is one text part, exactly as written, and renders + escaped if it contains `/`. The derive still knows no tables (ADR-0003); a + consumer whose prefix is a table gets EQL's shape from it. +- **`Describe` is an open trait** for a value whose parts are a descriptor + of its own, the identity data is keyed under. An implementor returns a + `Description`, built from a first part so it cannot be empty, and never + writes rendered text; `Descriptor::from_piece` is the only renderer. That + is what makes an open trait safe as a key-derivation input: two + implementors render alike only when their parts are alike, and no + implementor can inject a separator. `to_context` is what the type's + `IntoContext` returns, so the AAD and the descriptor are one tree. It is + open rather than sealed because the renderer, not the implementor list, + carries the safety property, and a consumer's own column or document type + is the expected implementor. +- **`Label`** is the first-class `Describe`: a path of plain segments whose + `Display` is its descriptor and parses back losslessly. It is how a direct + consumer of the crate names its data; EQL's `Identifier` is a two-segment + `Label` in shape. The Go binding has the same type and the same segment + rule, held together by one fixture both test suites read. +- An own context in a declaration tree is any `NonEmpty`, + not only a `&'static str` (ADR-0004, revised in place). + +## Consequences + +- `stack-encrypt` 0.1.0 was published with the `|` rendering before this + landed. Any key minted through 0.1.0 renders its descriptor differently + from the same context under this rendering and cannot be retrieved by it. + A reader who finds `users|email` in a ZeroKMS log is looking at a 0.1.0 + key. +- **This ships as 0.2.0**, with `stack-encrypt-derive` in the same version + group. The first revision of #1050 deferred the bump — one known consumer, + aware of the change — on the assumption that the release tooling would + bump on the next release. It would not: the root release-plz line is + publish-only, and a stack-* version moves only when a pull request edits + `Cargo.toml`. The bump also has a consumer that needs it to be a specific + number: `eql-bindings`' `stack-encrypt` feature (#971) names the + stack-encrypt version it compiles against, and that version has to be one + that carries `Describe`, `Description` and `Label`, which 0.1.0 does not. + Cargo rejects a `version` requirement the in-tree path dependency does not + satisfy, so the bump has to land here, before #971 can name it. +- A `/` inside a name is legal but ugly: it renders escaped. `Label` refuses + such a segment instead, so a consumer who wants a readable log uses a + `Label` and finds out at construction. +- Text and bytes with the same content, and signed and unsigned integers of + one width, still render alike (documented coarseness, unchanged). The + rendering is one-to-one over part *trees* up to that coarseness, not over + every encoding. diff --git a/packages/stack-encrypt/examples/encrypted_record.rs b/packages/stack-encrypt/examples/encrypted_record.rs index c924c457c..d77e72940 100644 --- a/packages/stack-encrypt/examples/encrypted_record.rs +++ b/packages/stack-encrypt/examples/encrypted_record.rs @@ -55,7 +55,7 @@ struct User { } /// `User`, encrypted field by field: `age` from `user.age` under -/// `"users/age"`, `email` from `user.email` under `"users/email"`. The prefix +/// `("users", "age")`, `email` from `user.email` under `("users", "email")`. The prefix /// is named once, explicitly — it is part of the stored data's identity, so /// it is never inferred from a Rust type name — and the field half follows /// the plaintext field. @@ -108,7 +108,7 @@ async fn main() -> Result<(), Box> { // WHERE age = 34: compare equality terms. let probe: EqualityTerm = 34u32 - .encrypt_into_with_context(&keyset, nonempty!("users/age")) + .encrypt_into_with_context(&keyset, nonempty!("users").with("age")) .await?; let equal: Vec = (0..table.len()) .filter(|&i| table[i].age.eq == probe) @@ -117,7 +117,7 @@ async fn main() -> Result<(), Box> { // WHERE age > 40: compare ORE terms. let bound: OreTerm = 40u32 - .encrypt_into_with_context(&keyset, nonempty!("users/age")) + .encrypt_into_with_context(&keyset, nonempty!("users").with("age")) .await?; let over_40: Vec = (0..table.len()) .filter(|&i| table[i].age.ord > bound) @@ -153,10 +153,10 @@ async fn main() -> Result<(), Box> { // --- Binding a value to its record ---------------------------------------- // A context the caller passes *extends* every field's own: under the - // record's id, `age` is sealed under `("users/age", id)` and opens only + // record's id, `age` is sealed under `(("users", "age"), id)` and opens only // there — a ciphertext can no longer be moved between records of the // same table. The price is that its terms are scoped to that record too: - // a probe built under `"users/age"` alone never matches them, so extend + // a probe built under `("users", "age")` alone never matches them, so extend // where a value is read by id, not where it is searched across rows. let id = 42u64; let alice = User { @@ -174,7 +174,7 @@ async fn main() -> Result<(), Box> { !unscoped.is_empty() ); let scoped: EqualityTerm = 34u32 - .encrypt_into_with_context(&keyset, nonempty!("users/age").with(id)) + .encrypt_into_with_context(&keyset, nonempty!("users").with("age").with(id)) .await?; println!( " ...and a probe built under the same id: {}", diff --git a/packages/stack-encrypt/examples/mixed_user.rs b/packages/stack-encrypt/examples/mixed_user.rs index 40418e378..bb64bd84b 100644 --- a/packages/stack-encrypt/examples/mixed_user.rs +++ b/packages/stack-encrypt/examples/mixed_user.rs @@ -38,7 +38,7 @@ //! `zerokms_auth` example for the lookup order). use stack_encrypt::{ - Cipher, CipherText, Decipher, Decrypt, Encrypt, IntoAad, StackCipher, StackCipherText, + Cipher, CipherText, Decipher, Decrypt, Encrypt, IntoAad, Label, StackCipher, StackCipherText, Unspecified, }; use vitaminc_aead::{DecipherVisitor, MapAccess, MapCipher, Passthrough}; @@ -175,16 +175,25 @@ async fn main() -> Result<(), Box> { }, ]; + // The context names the data: a `Label` is the first-class spelling of a + // name, here the users record at schema version 1, rendering `users/v1` + // in the ZeroKMS log. (EQL's `Identifier`, a table and a column, is the + // same shape with two segments.) + let record = Label::new(["users", "v1"])?; + // One call, one batched generate_keys round-trip for every encrypted leaf // in the whole Vec (here: 3 rows x 2 encrypted fields = 6 data keys). - let ciphertext = cipher.default_keyset().encrypt(users, "users/v1").await?; + let ciphertext = cipher + .default_keyset() + .encrypt(users, record.clone()) + .await?; println!("what the stored ciphertext reveals:"); describe(&ciphertext, 1); // One batched retrieve_keys round-trip, then a crypto-free structural // decode back into the typed rows. The AAD must match the encrypt call. - let users: Vec = cipher.decrypt(ciphertext, "users/v1").await?; + let users: Vec = cipher.decrypt(ciphertext, record).await?; println!("\ndecrypted rows:"); for user in &users { diff --git a/packages/stack-encrypt/examples/search_terms.rs b/packages/stack-encrypt/examples/search_terms.rs index bf44ad5e1..74c556b96 100644 --- a/packages/stack-encrypt/examples/search_terms.rs +++ b/packages/stack-encrypt/examples/search_terms.rs @@ -39,21 +39,22 @@ async fn main() -> Result<(), Box> { // --- Equality: exact-match lookups -------------------------------------- // - // The context ("users/email") domain-separates terms per field: the same + // The context, the pair ("users", "email"), domain-separates terms per + // field: the same // value indexed under another field can never produce a colliding term. let stored: EqualityTerm = "alice@example.com" - .encrypt_into_with_context(&terms, nonempty!("users/email")) + .encrypt_into_with_context(&terms, nonempty!("users").with("email")) .await?; let hit: EqualityTerm = "alice@example.com" - .encrypt_into_with_context(&terms, nonempty!("users/email")) + .encrypt_into_with_context(&terms, nonempty!("users").with("email")) .await?; let miss: EqualityTerm = "bob@example.com" - .encrypt_into_with_context(&terms, nonempty!("users/email")) + .encrypt_into_with_context(&terms, nonempty!("users").with("email")) .await?; let wrong_field: EqualityTerm = "alice@example.com" - .encrypt_into_with_context(&terms, nonempty!("users/name")) + .encrypt_into_with_context(&terms, nonempty!("users").with("name")) .await?; println!("\nequality:"); @@ -73,13 +74,13 @@ async fn main() -> Result<(), Box> { let bio: MatchTerm = "alice, senior cryptography engineer" .to_string() - .encrypt_into_with_context(&terms, nonempty!("users/bio")) + .encrypt_into_with_context(&terms, nonempty!("users").with("bio")) .await?; for query in ["crypto", "engineer", "plumber"] { let probe: MatchTerm = query .to_string() - .encrypt_into_with_context(&terms, nonempty!("users/bio")) + .encrypt_into_with_context(&terms, nonempty!("users").with("bio")) .await?; println!("match: bio contains {query:?} => {}", bio.contains(&probe)); } @@ -97,13 +98,13 @@ async fn main() -> Result<(), Box> { // local. let age_30: OreTerm = 30u32 - .encrypt_into_with_context(&terms, nonempty!("users/age")) + .encrypt_into_with_context(&terms, nonempty!("users").with("age")) .await?; let age_45: OreTerm = 45u32 - .encrypt_into_with_context(&terms, nonempty!("users/age")) + .encrypt_into_with_context(&terms, nonempty!("users").with("age")) .await?; let query_40: OreTerm = 40u32 - .encrypt_into_with_context(&terms, nonempty!("users/age")) + .encrypt_into_with_context(&terms, nonempty!("users").with("age")) .await?; println!("\nore (WHERE age > 40):"); @@ -112,10 +113,10 @@ async fn main() -> Result<(), Box> { // Strings order lexicographically. let apple: OreTerm<&str> = "apple" - .encrypt_into_with_context(&terms, nonempty!("users/name")) + .encrypt_into_with_context(&terms, nonempty!("users").with("name")) .await?; let banana: OreTerm<&str> = "banana" - .encrypt_into_with_context(&terms, nonempty!("users/name")) + .encrypt_into_with_context(&terms, nonempty!("users").with("name")) .await?; println!(" \"apple\" < \"banana\" => {}", apple < banana); @@ -137,11 +138,11 @@ async fn main() -> Result<(), Box> { let record: SearchableEmail = "alice@example.com" .to_string() - .encrypt_into_with_context(&terms, nonempty!("users/email")) + .encrypt_into_with_context(&terms, nonempty!("users").with("email")) .await?; let probe: MatchTerm = "example" .to_string() - .encrypt_into_with_context(&terms, nonempty!("users/email")) + .encrypt_into_with_context(&terms, nonempty!("users").with("email")) .await?; println!("\nrecord:"); println!( @@ -157,7 +158,7 @@ async fn main() -> Result<(), Box> { record.ord > "alice" .to_string() - .encrypt_into_with_context(&terms, nonempty!("users/email")) + .encrypt_into_with_context(&terms, nonempty!("users").with("email")) .await? ); let _ = record.c; // the ciphertext, opened with `decrypt_into` under the same context diff --git a/packages/stack-encrypt/fuzz/Cargo.lock b/packages/stack-encrypt/fuzz/Cargo.lock index 30234dfae..9a2612b77 100644 --- a/packages/stack-encrypt/fuzz/Cargo.lock +++ b/packages/stack-encrypt/fuzz/Cargo.lock @@ -2020,7 +2020,7 @@ dependencies = [ [[package]] name = "stack-encrypt" -version = "0.1.0" +version = "0.2.0" dependencies = [ "base64ct", "cllw-ore", @@ -2040,7 +2040,7 @@ dependencies = [ [[package]] name = "stack-encrypt-derive" -version = "0.1.0" +version = "0.2.0" dependencies = [ "proc-macro2", "quote", diff --git a/packages/stack-encrypt/src/descriptor.rs b/packages/stack-encrypt/src/descriptor.rs index 55c0457d5..64a398de5 100644 --- a/packages/stack-encrypt/src/descriptor.rs +++ b/packages/stack-encrypt/src/descriptor.rs @@ -20,13 +20,24 @@ //! tag, so changing the rendering strands every key issued under the old //! one. //! +//! Two kinds of value sit under the renderer. A **context** is anything +//! [`IntoContext`] — a literal, a pair, an integer, a `NonEmpty` chain — and +//! is arbitrary: a direct consumer of this crate seals under whatever parts +//! name its data. A [`Describe`] value is one whose parts *are* a descriptor +//! of its own: the identity data is keyed under, returned as the parts of a +//! [`Description`] so the implementor never writes rendered text. +//! [`Label`] is the first-class one — a path of plain segments, written and +//! read as `users/email` — and EQL's identifier (a table and a column) is the +//! same shape. Both are contexts too, through the same parts, so what +//! ZeroKMS binds and what the AEAD seals under never disagree. +//! //! The descriptor follows the context's **parts**, not its encoded bytes, //! so it and the AEAD encoding can disagree about whether two contexts are //! one. They disagree in both directions, each in named cases: //! //! * The descriptor is *finer* for a pre-encoded //! [`Context`](vitaminc_aead::Context), which is one opaque bytes part. -//! `("tenant", 7u64)` renders `tenant|7u64`; the same tuple passed +//! `("tenant", 7u64)` renders `tenant/7u64`; the same tuple passed //! through `into_aad()` first encodes to the same AAD bytes but renders //! `b64:` + those bytes. ZeroKMS refuses what the AEAD would open. //! * The descriptor is *coarser* for shapes that render alike but encode @@ -42,10 +53,13 @@ //! both times, text or bytes as it was sealed — not merely one with the //! same bytes, and not merely one with the same descriptor. +use std::borrow::Cow; +use std::fmt::Write as _; use std::sync::Arc; -use base64ct::{Base64, Encoding}; +use base64ct::{Base64Url, Encoding}; use vitaminc_aead::{ContextPiece, IntoAad, IntoContext}; +use vitaminc_protected::{MaybeEmpty, NonEmpty}; /// A context rendered as the string sent to ZeroKMS with every data-key /// request. See the [module docs](self). @@ -68,7 +82,7 @@ impl Descriptor { pub const BASE64_PREFIX: &'static str = "b64:"; /// The separator between the parts of a list. - pub const SEPARATOR: char = '|'; + pub const SEPARATOR: char = '/'; /// The longest descriptor ZeroKMS accepts, in bytes of the rendered /// string: the protocol's [`MAX_DESCRIPTOR_LEN`](crate::kms::MAX_DESCRIPTOR_LEN). @@ -92,12 +106,14 @@ impl Descriptor { /// ``` /// use stack_encrypt::{nonempty, Descriptor}; /// - /// // A textual context is its own descriptor. - /// assert_eq!(Descriptor::of("users/email").as_str(), "users/email"); + /// // A table and a column are two parts, joined by `/`. + /// let column = nonempty!("users").with("email"); + /// assert_eq!(Descriptor::of(column).as_str(), "users/email"); /// - /// // A composite renders its parts in order: a field bound to a row id. - /// let row = nonempty!("users/email").with(7u64); - /// assert_eq!(Descriptor::of(row).as_str(), "users/email|7u64"); + /// // A composite renders its parts in order: that column bound to a row + /// // id. The nested pair is parenthesised. + /// let row = column.with(7u64); + /// assert_eq!(Descriptor::of(row).as_str(), "(users/email)/7u64"); /// assert!(Descriptor::of(row).fits()); /// /// // Rendered from the parts, so it follows the encoding: integers are @@ -105,7 +121,7 @@ impl Descriptor { /// // could read as another form is escaped. /// assert_eq!(Descriptor::of(7i64), Descriptor::of(7u64)); /// assert_eq!(Descriptor::of(Some("")).as_str(), "(b64:)"); - /// assert_eq!(Descriptor::of("a|b").as_str(), "b64:YXxi"); + /// assert_eq!(Descriptor::of("users/email").as_str(), "b64:dXNlcnMvZW1haWw="); /// ``` pub fn of<'a>(context: impl IntoContext<'a>) -> Self { Self::from_piece(&context.into_context()) @@ -116,13 +132,17 @@ impl Descriptor { /// # Frozen rendering /// /// * A **text** part, or a **bytes** part that is UTF-8, renders - /// **verbatim** when it is *plain*: non-empty, no control characters, - /// none of `|`, `(`, `)`, not beginning with + /// **verbatim** when it is *plain*: non-empty, no control characters + /// and no invisible format characters (zero-width and bidirectional + /// marks, which would print as another name), none of `/`, `(`, `)`, + /// not beginning with /// [`b64:`](Self::BASE64_PREFIX), and not beginning with an ASCII digit - /// or `-`. So a `&str` context — `users/email` — is its own - /// descriptor, readable in the ZeroKMS log. Any other text or bytes - /// part renders as `b64:` followed by the standard (padded) base64 of - /// its bytes; an **empty** part is therefore the bare prefix, `b64:`, + /// or `-`. So a table and a column are two parts — the pair + /// `("users", "email")` renders `users/email`, readable in the ZeroKMS + /// log — and a single text part containing `/` is escaped, so it can + /// never be mistaken for one. Any other text or bytes part renders as + /// `b64:` followed by the URL-safe (padded) base64 of its bytes: the + /// standard alphabet's `/` would read as a separator; an **empty** part is therefore the bare prefix, `b64:`, /// so `Some("")` is `(b64:)` and `None` is `()`. Text and bytes with /// the same bytes render the same, though since vitaminc 0.5 they /// encode differently: the rendering is of the parts, not the bytes. @@ -133,11 +153,12 @@ impl Descriptor { /// leaf's type tag carries the signedness, so `7i64` and `7u64` are /// two contexts to the AEAD; the rendering, frozen before that, does /// not follow. - /// * A **list** renders its parts joined by [`|`](Self::SEPARATOR). At + /// * A **list** renders its parts joined by [`/`](Self::SEPARATOR). At /// the root, a list of two or more parts has no delimiters — - /// `nonempty!("users/email").with(7u64)` is `users/email|7u64` — and - /// any other list, nested or of fewer than two parts, is parenthesised: - /// `(users/email)`, `()`, `a|(b|c)`. + /// `nonempty!("users").with("email")` is `users/email` — and any other + /// list, nested or of fewer than two parts, is parenthesised: + /// `(users/email)/7u64` for that pair extended with a row id, `()`, + /// `a/(b/c)`. /// * At the root, the empty text or bytes part — the `()` AAD, or `""` — /// renders as the empty string, which is what ZeroKMS receives when a /// caller opts out of descriptors. @@ -208,26 +229,21 @@ impl Descriptor { Ok(text) if Self::is_plain(text) => out.push_str(text), _ => { out.push_str(Self::BASE64_PREFIX); - out.push_str(&Base64::encode_string(bytes)); + out.push_str(&Base64Url::encode_string(bytes)); } } } fn render_int(value: &impl std::fmt::Display, suffix: &str, out: &mut String) { - use std::fmt::Write as _; // Writing to a `String` cannot fail. let _ = write!(out, "{value}{suffix}"); } /// Text that renders verbatim: non-empty, and nothing another form - /// begins with or contains. + /// begins with or contains — exactly what a [`Label`] segment may be. + /// One definition serves both, so a label always renders verbatim. fn is_plain(text: &str) -> bool { - !text.is_empty() - && !text.starts_with(Self::BASE64_PREFIX) - && !text.starts_with(|c: char| c.is_ascii_digit() || c == '-') - && !text - .chars() - .any(|c| c.is_control() || matches!(c, '|' | '(' | ')')) + Label::check_segment(0, text).is_ok() } /// The rendered string, as sent to ZeroKMS. @@ -263,6 +279,372 @@ impl Descriptor { } } +/// A value with a ZeroKMS descriptor of its own: the identity data is keyed +/// under. See the [module docs](self). +/// +/// Implement it for the type that names where a value lives — a table and a +/// column, a document path, a tenant's record kind — and that name is what +/// ZeroKMS binds into the data key and logs on every retrieval. An +/// implementor returns the **parts** of its name as a [`Description`]; it +/// never writes the rendered string. The one renderer, +/// [`Descriptor::from_piece`], turns the parts into the string, so two +/// implementors render alike only when their parts are alike, and a part +/// that contains the separator is escaped rather than read as two. That is +/// what keeps an open trait safe as a key-derivation input: the implementor +/// chooses *what* the identity is, this crate chooses how it is spelled. A +/// `Description` is built from its first part, so a part cannot be forgotten. +/// The parts themselves are not checked: `Description::text("")` is the +/// empty text part and, alone, renders the empty descriptor, which binds no +/// identity, and `Description::part(None::<&str>)` renders `()`. A name +/// taken from a runtime string belongs in a [`Label`], which refuses what +/// would not render as itself; `Description` is for a type whose parts are +/// fixed by its definition. +/// +/// A `Describe` type is sealed under as a context through the same parts: +/// [`to_context`](Self::to_context) is the [`ContextPiece`] the type's +/// [`IntoContext`] must return, so the descriptor ZeroKMS binds and the AAD +/// the ciphertext is sealed under are one value seen two ways. [`Label`] is +/// the ready-made implementor, a path of plain segments; EQL's identifier +/// (a table and a column) is the same shape with two. +/// +/// ``` +/// use stack_encrypt::{ContextPiece, Describe, Description, IntoContext}; +/// +/// /// A column of a database table. +/// struct Column { +/// table: &'static str, +/// name: &'static str, +/// } +/// +/// impl Describe for Column { +/// fn describe(&self) -> Description { +/// Description::text(self.table).then_text(self.name) +/// } +/// } +/// +/// // Sealed under as a context through the same two parts. +/// impl<'a> IntoContext<'a> for Column { +/// fn into_context(self) -> ContextPiece<'a> { +/// self.to_context() +/// } +/// } +/// +/// let email = Column { table: "users", name: "email" }; +/// assert_eq!(email.descriptor().as_str(), "users/email"); +/// // A part containing the separator is one part, escaped — never a pair. +/// let odd = Column { table: "users/email", name: "x" }; +/// assert_eq!(odd.descriptor().as_str(), "b64:dXNlcnMvZW1haWw=/x"); +/// ``` +pub trait Describe { + /// The parts of this value's descriptor, in order, starting from the + /// first: [`Description::text`] or [`Description::part`], then + /// [`then_text`](Description::then_text) / [`then`](Description::then). + fn describe(&self) -> Description; + + /// The parts as one context piece: the single part, or the list of the + /// parts. What the type's [`IntoContext`] returns, so the AAD and the + /// descriptor are derived from one tree. + fn to_context(&self) -> ContextPiece<'static> { + self.describe().into_context() + } + + /// The descriptor ZeroKMS binds and logs: [`to_context`](Self::to_context) + /// rendered by [`Descriptor::from_piece`]. + fn descriptor(&self) -> Descriptor { + Descriptor::from_piece(&self.to_context()) + } +} + +/// The parts of a [`Describe`] value's descriptor: a first part and any +/// number after it. +/// +/// It holds parts, never rendered text, so an implementor cannot write a +/// separator, an escape prefix or a parenthesis into the descriptor: each +/// part is rendered by [`Descriptor::from_piece`] under the frozen rules, +/// and text that would read as another form is escaped there. It is built +/// from its first part, so there is no empty description. +/// +/// As a context ([`IntoContext`]), one part is that part — a one-segment +/// name is the same context as the bare literal — and two or more are a +/// flat list of them. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Description { + first: ContextPiece<'static>, + rest: Vec>, +} + +impl Description { + /// A description whose first part is text. Plain text (see [`Label`]) + /// renders verbatim; any other text renders escaped. + pub fn text(first: impl Into) -> Self { + Self::part(ContextPiece::Text(Cow::Owned(first.into()))) + } + + /// A description whose first part is any context part — an integer, a + /// bytes part, a nested list — as the context encoding sees it. + pub fn part<'a>(first: impl IntoContext<'a>) -> Self { + Self { + first: first.into_context().into_owned(), + rest: Vec::new(), + } + } + + /// Append a text part. + pub fn then_text(self, text: impl Into) -> Self { + self.then(ContextPiece::Text(Cow::Owned(text.into()))) + } + + /// Append any context part. + pub fn then<'a>(mut self, part: impl IntoContext<'a>) -> Self { + self.rest.push(part.into_context().into_owned()); + self + } +} + +impl<'a> IntoContext<'a> for Description { + fn into_context(self) -> ContextPiece<'a> { + if self.rest.is_empty() { + self.first + } else { + let mut parts = Vec::with_capacity(1 + self.rest.len()); + parts.push(self.first); + parts.extend(self.rest); + ContextPiece::List(parts) + } + } +} + +/// The name of the data a value is sealed under: a table and a column +/// (`users/email`), a document path (`documents/v2/body`), any name a direct +/// consumer of this crate chooses. ZeroKMS binds the data key to that name +/// and writes it in its log, spelled exactly as given. EQL's identifier, a +/// table and a column, is a `Label` of two segments. +/// +/// # Naming and scoping +/// +/// A context carries two kinds of information, and each has one spelling: +/// +/// * A **name** says *what* the data is. Spell it as a `Label`. +/// * A **scope** says *which* slice of that data: a tenant, a row. Spell it +/// by extending the name with [`NonEmpty::with`] (or, on a derived record, +/// by the caller's context, which extends every field's own). +/// +/// | What you mean | Spelling | ZeroKMS log | +/// |---|---|---| +/// | the `users.email` column | `Label::parse("users/email")?` | `users/email` | +/// | that column, tenant 7 | `NonEmpty::from(label).with(7u64)` | `(users/email)/7u64` | +/// | a deeper name | `Label::parse("documents/v2/body")?` | `documents/v2/body` | +/// | a one-part name | `Label::parse("users")?`, the same as `nonempty!("users")` | `users` | +/// +/// Do not build a name with `with`, and do not put a scope into a `Label`. +/// The renderer keeps the two apart: a name is one flat list, a scope nests. +/// So `(users/email)/7u64` is never read as a three-segment name, and +/// `documents/v2/body` is never read as a scoped column. +/// +/// A two-segment `Label` binds the same context a +/// `#[stash(struct = .., context = "
")]` derive gives a field, which +/// the derive spells `nonempty!("users").with("email")`. That is what lets a +/// label open a row a derive wrote, and a probe built from the label match +/// the terms the derive produced. +/// +/// ``` +/// use stack_encrypt::{nonempty, Descriptor, Label, NonEmpty}; +/// +/// // A name. +/// let email = Label::parse("users/email")?; +/// assert_eq!(email.to_string(), "users/email"); +/// assert_eq!(Descriptor::of(&email).as_str(), "users/email"); +/// assert_eq!(Label::new(["users", "email"])?, email); +/// +/// // The same column, scoped to tenant 7. +/// let tenant_7 = NonEmpty::from(email.clone()).with(7u64); +/// assert_eq!(Descriptor::of(tenant_7).as_str(), "(users/email)/7u64"); +/// +/// // What a `struct = .., context = "users"` derive binds its `email` field under. +/// assert_eq!(Descriptor::of(&email), Descriptor::of(nonempty!("users").with("email"))); +/// +/// // Not a label: the separator inside a segment, and an empty segment. +/// assert!(Label::new(["users/email"]).is_err()); +/// assert!(Label::parse("users//email").is_err()); +/// # Ok::<(), stack_encrypt::LabelError>(()) +/// ``` +/// +/// # Segments +/// +/// Every segment is **plain** — non-empty, no control or invisible format +/// characters (zero-width and bidirectional marks), none of +/// `/`, `(`, `)`, not beginning with `b64:`, a digit or `-` — which is +/// exactly the text [`Descriptor::from_piece`] renders verbatim. So a +/// `Label` renders as its segments joined by [`/`](Descriptor::SEPARATOR), +/// its [`Display`](std::fmt::Display) *is* its descriptor, and +/// [`parse`](Self::parse) reads that string back losslessly: no segment can +/// contain the separator, so the split is unambiguous. A string that is not +/// a label is refused with a [`LabelError`] naming the segment, never +/// escaped silently. +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +pub struct Label(Box<[Box]>); + +impl Label { + /// A label from its segments, each checked to be plain. + pub fn new(segments: I) -> Result + where + I: IntoIterator, + I::Item: AsRef, + { + let segments = segments + .into_iter() + .enumerate() + .map(|(index, segment)| { + let segment = segment.as_ref(); + Self::check_segment(index, segment)?; + Ok(Box::from(segment)) + }) + .collect::]>, LabelError>>()?; + if segments.is_empty() { + return Err(LabelError::Empty); + } + Ok(Self(segments)) + } + + /// A label from its rendered form: segments separated by + /// [`/`](Descriptor::SEPARATOR). The inverse of + /// [`Display`](std::fmt::Display). + pub fn parse(text: &str) -> Result { + Self::new(text.split(Descriptor::SEPARATOR)) + } + + /// The segments, in order; at least one. + pub fn segments(&self) -> impl ExactSizeIterator + '_ { + self.0.iter().map(|s| &**s) + } + + /// Whether `segment` is plain, as the error that says why not. This + /// is the one definition of plain text: [`Descriptor::from_piece`] + /// renders verbatim exactly what passes here. + fn check_segment(index: usize, segment: &str) -> Result<(), LabelError> { + if segment.is_empty() { + return Err(LabelError::EmptySegment { index }); + } + if segment.starts_with(Descriptor::BASE64_PREFIX) + || segment.starts_with(|c: char| c.is_ascii_digit() || c == '-') + { + return Err(LabelError::ReservedPrefix { index }); + } + for found in segment.chars() { + if found == Descriptor::SEPARATOR { + return Err(LabelError::Separator { index }); + } + if found.is_control() || Self::INVISIBLE.contains(&found) || matches!(found, '(' | ')') + { + return Err(LabelError::Reserved { index, found }); + } + } + Ok(()) + } + + /// Format characters with no glyph of their own: the soft hyphen, the + /// Arabic letter mark, the Mongolian vowel separator, the zero-width + /// characters, the bidirectional embeddings, overrides and isolates, and + /// the byte-order mark. `char::is_control` covers only `Cc`; these are + /// `Cf`. A name containing one prints like another name in the ZeroKMS + /// log, which is what a plain segment exists to prevent, so they are + /// refused beside the control characters. The Go binding carries the + /// same list, and the shared fixture holds the two together. + const INVISIBLE: &'static [char] = &[ + '\u{00AD}', '\u{061C}', '\u{180E}', '\u{200B}', '\u{200C}', '\u{200D}', '\u{200E}', + '\u{200F}', '\u{202A}', '\u{202B}', '\u{202C}', '\u{202D}', '\u{202E}', '\u{2060}', + '\u{2061}', '\u{2062}', '\u{2063}', '\u{2064}', '\u{2066}', '\u{2067}', '\u{2068}', + '\u{2069}', '\u{FEFF}', + ]; +} + +impl Describe for Label { + fn describe(&self) -> Description { + let mut segments = self.segments(); + // A label has at least one segment by construction. + let first = segments.next().unwrap_or(""); + segments.fold(Description::text(first), Description::then_text) + } +} + +impl<'a> IntoContext<'a> for Label { + fn into_context(self) -> ContextPiece<'a> { + self.to_context() + } +} + +impl<'a> IntoContext<'a> for &'a Label { + fn into_context(self) -> ContextPiece<'a> { + self.to_context() + } +} + +/// Never empty: a label has at least one non-empty segment. +impl MaybeEmpty for Label { + fn is_empty(&self) -> bool { + false + } +} + +/// A label is nonempty by construction, so it needs no runtime check to be +/// the context a target-directed leaf takes. +impl From