diff --git a/CHANGELOG.md b/CHANGELOG.md index 4b3570f..5806ba7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -269,6 +269,11 @@ subgroup), `TestPinPathMatchesDecode` and `TestChainHashFormula`; ### Fixed +- `datekey.Parse` rejects invalid UTF-8 in the `dk1_` JSON at step 2, as + §19 requires. `encoding/json` replaced it with U+FFFD, so a member that a + repeated name overwrites passed steps 2 and 3 and ended as + `ERR_DATEKEY_NON_CANONICAL` instead of `ERR_DATEKEY_INVALID`. Found by the + second implementation's differential; new `dk1.json` vector. - `extension.CheckDisjoint` is a linear merge of the two sorted arrays; a PUBLIC_HEADER with 40 000 + 40 000 extensions took 8.3 s in the pairwise check (§76, case 6). diff --git a/datekey/datekey.go b/datekey/datekey.go index 702dd0d..57def5a 100644 --- a/datekey/datekey.go +++ b/datekey/datekey.go @@ -15,6 +15,7 @@ import ( "strconv" "strings" "time" + "unicode/utf8" datekeys "g.activething.com/go/DateKeys" "g.activething.com/go/DateKeys/profile" @@ -194,6 +195,12 @@ func parseJSON(raw []byte) (DateKey, error) { invalid := func(format string, a ...any) error { return fmt.Errorf("datekey: "+format+": %w", append(a, datekeys.ErrDateKeyInvalid)...) } + // Spec §19 step 2: invalid UTF-8 fails the step. encoding/json would + // replace it with U+FFFD inside strings, so a member later overwritten + // by a repeated name could otherwise reach the canonical comparison. + if !utf8.Valid(raw) { + return DateKey{}, invalid("payload is not valid UTF-8") + } dec := json.NewDecoder(bytes.NewReader(raw)) dec.UseNumber() var obj map[string]any diff --git a/datekey/datekey_test.go b/datekey/datekey_test.go index 71995ab..7f3350f 100644 --- a/datekey/datekey_test.go +++ b/datekey/datekey_test.go @@ -212,6 +212,8 @@ func TestReadingRules(t *testing.T) { {"both alphabets mixed", datekey.Prefix + "-+" + payload[2:], datekeys.ErrDateKeyInvalid}, {"byte order mark", enc("\ufeff" + `{"version":1,"network":"datekeys:quicknet:v1","round":66884212}`), datekeys.ErrDateKeyInvalid}, {"JSON white space before the object", enc(" \t\r\n" + `{"version":1,"network":"datekeys:quicknet:v1","round":66884212}`), datekeys.ErrDateKeyNonCanonical}, + {"invalid UTF-8 in a member a repeated name overwrites", enc(`{"version":1,"network":"` + "\xff" + `","network":"datekeys:quicknet:v1","round":66884212}`), datekeys.ErrDateKeyInvalid}, + {"invalid UTF-8 in a member name", enc(`{"version":1,"netw` + "\xff" + `ork":"x","network":"datekeys:quicknet:v1","round":66884212}`), datekeys.ErrDateKeyInvalid}, {"a JSON array", enc(`[{"version":1,"network":"datekeys:quicknet:v1","round":66884212}]`), datekeys.ErrDateKeyInvalid}, {"version 1.0000000000000001, 1 as a double", enc(`{"version":1.0000000000000001,"network":"datekeys:quicknet:v1","round":66884212}`), datekeys.ErrDateKeyInvalid}, {"round 66884212.00000000000001", enc(`{"version":1,"network":"datekeys:quicknet:v1","round":66884212.00000000000001}`), datekeys.ErrDateKeyInvalid}, diff --git a/docs/traceability.md b/docs/traceability.md index e4ab844..fb81083 100644 --- a/docs/traceability.md +++ b/docs/traceability.md @@ -27,7 +27,7 @@ Paths are relative to the repository root. `§` numbers refer to | 16 | Normative round vector | — | `datekey.TestNormativeRoundVector`; `testdata/vectors/quicknet_rounds.json` | | 17 | Past-round attack; at step 10 the release round is compared before the signature | `provider.Verify` (round equality first), `capsule.Encrypt` (round time ≥ requested), `agewrap.CheckTimeStanzas` | `provider.TestVerifyRejects` (*another round and a short signature*); mutations *DateKey A + release of round B*, *tlock stanza round differs from DateKey.round* | | 18 | `dk1_` representation; the canonical JSON has no escapes, the `profile_id` alphabet needs none | `DateKey.CanonicalJSON`, `DateKey.Compact` | `datekey.TestGoldenDK1Vectors`, `TestNormativeRoundVector` | -| 19 | `dk1_` canonicality; steps 1 to 3 `ERR_DATEKEY_INVALID`, step 6 `ERR_DATEKEY_NON_CANONICAL`; step 1 accepts either Base64 alphabet, padding and non-zero trailing bits but no other character (CR and LF included); step 2 one RFC 8259 JSON object, no byte order mark; JSON numbers by their exact decimal value; round in 1..2^53−1 without the profile | `datekey.Parse` (`decodeBase64`, which rejects CR and LF before the Go decoders, `parseJSON`, `jsonUint`) | `datekey.TestGoldenDK1Vectors`, `TestReadingRules`, `TestNumberSpellings`, `FuzzParse`; mutation *non-canonical dk1_ JSON*; `testdata/vectors/dk1.json` (*byte order mark*, *line feed inside the Base64*, *carriage return and line feed after the Base64*, *version 1.0000000000000001: its exact value, not a double*) | +| 19 | `dk1_` canonicality; steps 1 to 3 `ERR_DATEKEY_INVALID`, step 6 `ERR_DATEKEY_NON_CANONICAL`; step 1 accepts either Base64 alphabet, padding and non-zero trailing bits but no other character (CR and LF included); step 2 one RFC 8259 JSON object, no byte order mark; JSON numbers by their exact decimal value; round in 1..2^53−1 without the profile | `datekey.Parse` (`decodeBase64`, which rejects CR and LF before the Go decoders, `parseJSON`, which rejects invalid UTF-8 before `encoding/json` can replace it, `jsonUint`) | `datekey.TestGoldenDK1Vectors`, `TestReadingRules`, `TestNumberSpellings`, `FuzzParse`; mutation *non-canonical dk1_ JSON*; `testdata/vectors/dk1.json` (*byte order mark*, *line feed inside the Base64*, *carriage return and line feed after the Base64*, *version 1.0000000000000001: its exact value, not a double*, *invalid UTF-8 in a member a repeated name overwrites*) | | 20 | File extensions and magic | magic checks in `capsule.ParsePrelude`, `accesskey.Decode` | mutation *a .dkk offered as a .dkc*; `accesskey.TestDecodeRejects` *a .dkc* | | 21 | `capsule_id` | `capsule.Encrypt` (16 bytes from `crypto/rand`), `capsule.DecodeHeader` | `capsule.TestPortableKeysAreNeverReused` | | 22 | `.dkc` framing; `PUBLIC_HEADER_LEN` in 1..1 MiB, `SEALED_CONTROL_LEN` in 1..64 MiB; PAYLOAD_AGE to EOF, at least an age header | `capsule.Prelude`, `capsule.ParsePrelude` | mutations *version changed*, *flags != 0*, *reserved != 0*, *magic*, length limits; `capsule.TestFrameLengthLowerBounds`, `FuzzParsePrelude` | diff --git a/internal/testkit/vectors.go b/internal/testkit/vectors.go index 341c5df..b6b7820 100644 --- a/internal/testkit/vectors.go +++ b/internal/testkit/vectors.go @@ -177,6 +177,9 @@ func DK1Vectors() DK1VectorFile { {"line feed inside the Base64", datekey.Prefix + canon.Compact()[len(datekey.Prefix):][:8] + "\n" + canon.Compact()[len(datekey.Prefix)+8:]}, {"carriage return and line feed after the Base64", canon.Compact() + "\r\n"}, {"version 1.0000000000000001: its exact value, not a double", enc(`{"version":1.0000000000000001,"network":"datekeys:quicknet:v1","round":66884212}`)}, + // Spec §19 step 2: invalid UTF-8 fails step 2 even inside a member + // that a repeated name overwrites. + {"invalid UTF-8 in a member a repeated name overwrites", enc(`{"version":1,"network":"` + "\xff" + `","network":"datekeys:quicknet:v1","round":66884212}`)}, } for _, b := range bad { _, err := datekey.Parse(b.input) diff --git a/testdata/vectors/dk1.json b/testdata/vectors/dk1.json index 066c375..8489097 100644 --- a/testdata/vectors/dk1.json +++ b/testdata/vectors/dk1.json @@ -163,6 +163,11 @@ "name": "version 1.0000000000000001: its exact value, not a double", "input": "dk1_eyJ2ZXJzaW9uIjoxLjAwMDAwMDAwMDAwMDAwMDEsIm5ldHdvcmsiOiJkYXRla2V5czpxdWlja25ldDp2MSIsInJvdW5kIjo2Njg4NDIxMn0", "error": "ERR_DATEKEY_INVALID" + }, + { + "name": "invalid UTF-8 in a member a repeated name overwrites", + "input": "dk1_eyJ2ZXJzaW9uIjoxLCJuZXR3b3JrIjoi_yIsIm5ldHdvcmsiOiJkYXRla2V5czpxdWlja25ldDp2MSIsInJvdW5kIjo2Njg4NDIxMn0", + "error": "ERR_DATEKEY_INVALID" } ] }