From e2c70f50c16d7a36fe3af9e42512360630fd8098 Mon Sep 17 00:00:00 2001 From: dev Date: Mon, 5 Oct 2026 15:09:20 +0200 Subject: [PATCH] Stage 1: the CBOR profile of section 58, as package codec of Go lib/src/cbor.dart is the port of codec/codec.go: CborEncoder, CborDecoder, unmarshalCbor, peekSchema, checkSchema and walkCbor, with the same reads, the same checks in the same order and the same error texts, such as "codec: offset 0: 23 is not in its shortest form (initial byte 0x18): ERR_NON_CANONICAL_CBOR". Integers are exact on the VM and on the web, where an int is a double and the bit operators work on 32 bits. An argument of eight bytes is read as two halves of 32 bits, and is an int up to 2^53-1 and a BigInt above, map keys and the numbers of the error texts included. uint returns an int, since every schema bounds its integers at 2^53-1, and uint64 a BigInt. The map that peekSchema reads is bounded at 2^63-1, Go's math.MaxInt, on the web too. Two kinds of text are of Dart only. CborEncoder.uint refuses an int outside 0..2^53-1 and uint64 a BigInt outside 0..2^64-1, with the text of datekeys-ts, where Go's uint64 cannot hold such a value. And the only invalid text that a Dart String holds is a lone surrogate: the error quotes it as Go quotes its bytes in generalized UTF-8. The tests port codec_test.go, internal_test.go and vectors_test.go, with the texts that Go prints, and cbor.test.ts. The fuzz targets are properties over seeded inputs, checked against a reference encoder and decoder written apart, as internal/cbortest. cbor.json runs its 36 accept and 67 reject vectors with the walk limits and the values; its 172 schema vectors are read and wait for the schema decoders of stage 4. Co-Authored-By: Claude Opus 5.5 --- lib/datekeys.dart | 5 +- lib/src/cbor.dart | 682 +++++++++++++ test/cbor_reference.dart | 223 +++++ test/cbor_test.dart | 1806 +++++++++++++++++++++++++++++++++++ test/cbor_vectors_test.dart | 120 +++ 5 files changed, 2834 insertions(+), 2 deletions(-) create mode 100644 lib/src/cbor.dart create mode 100644 test/cbor_reference.dart create mode 100644 test/cbor_test.dart create mode 100644 test/cbor_vectors_test.dart diff --git a/lib/datekeys.dart b/lib/datekeys.dart index 4771584..774fc94 100644 --- a/lib/datekeys.dart +++ b/lib/datekeys.dart @@ -3,11 +3,12 @@ /// them, checked against the same test data. /// /// Stage 0 of docs/PLAN_dart.md, the package and its test data, and from -/// stage 1 the normative errors of spec §69 and byte helpers. The protocol -/// arrives stage by stage. +/// stage 1 the normative errors of spec §69, byte helpers and the CBOR +/// profile of spec §58. The protocol arrives stage by stage. library; export 'src/bytes.dart' show compareBytes, concatBytes, decodeUtf8, equalBytes, fromHex, toHex; +export 'src/cbor.dart'; export 'src/errors.dart'; export 'src/version.dart'; diff --git a/lib/src/cbor.dart b/lib/src/cbor.dart new file mode 100644 index 0000000..1fdde5b --- /dev/null +++ b/lib/src/cbor.dart @@ -0,0 +1,682 @@ +/// The CBOR profile of the DateKeys protocol (spec §58, §58.1), as package +/// codec of datekeys-go: the same reads, the same checks in the same order +/// and the same error texts. +/// +/// The profile is Deterministic CBOR (RFC 8949 §4.2.1) restricted to major +/// types 0 (unsigned integer), 2 (byte string), 3 (text string), 4 (array) +/// and 5 (map), with unsigned integer map keys in strictly ascending order, +/// integers and lengths in their shortest form, definite lengths only and +/// valid UTF-8 text. Negative integers, tags, floats, simple values (false, +/// true, null, undefined), indefinite lengths and every other map key are +/// rejected with ERR_NON_CANONICAL_CBOR. +/// +/// Each schema writes its own encoding with a [CborEncoder] and reads it with +/// a [CborDecoder], a strict cursor that reads exactly what the schema asks +/// for. [unmarshalCbor] runs the decoder of a schema and then re-encodes what +/// it decoded: the input must be reproduced byte for byte, or it is +/// ERR_NON_CANONICAL_CBOR. The same principle as dk1_ canonicality (spec +/// §19): canonicality does not depend on the decoder rejecting every +/// non-canonical form. +/// +/// [peekSchema] reads the type tag and the schema version of an object +/// before strict decoding (spec §70). [walkCbor] checks that bytes are one +/// data item of the profile; it is a helper for vectors and diagnostics, and +/// never decides whether an object of the protocol is valid. +/// +/// Integers. An unsigned integer of CBOR goes up to 2^64-1, and Dart's int +/// does not hold that much on every platform: on the VM it has 64 bits with a +/// sign, and on the web it is a double, exact up to 2^53, whose bit operators +/// work on 32 bits. So this library never relies on an int above 2^53-1, nor +/// on a shift or a bit operation beyond 31 bits: +/// +/// - the decoder reads an argument of eight bytes as two halves of 32 bits; +/// it is an [int] up to 2^53-1 ([maxSafeUint]) and a [BigInt] above, as +/// datekeys-ts reads a `number | bigint`. Comparisons with the bounds of +/// the caller, the order of the map keys and the numbers in the error texts +/// are exact on both platforms; +/// - every schema of the protocol bounds its integers at 2^53-1 (spec §58), +/// so [CborDecoder.uint] returns an int, and [CborEncoder.uint] takes one; +/// - what may be larger is exact: [CborDecoder.uint64] returns a BigInt, +/// [CborEncoder.uint64] takes one, and [CborDecoder.key] returns a map key +/// as an int up to 2^53-1 and as a BigInt above, so that a schema compares +/// it with its keys in a switch and prints any other exactly. +library; + +import 'dart:typed_data'; + +import 'bytes.dart'; +import 'errors.dart'; + +/// 2^53-1, the largest unsigned integer any schema of the protocol allows, +/// so that every integer is exact as an IEEE 754 double (spec §58), and as +/// an int of Dart on every platform. +const int maxSafeUint = 9007199254740991; + +/// The bound of the type tag that [peekSchema] reads, in bytes. Every type +/// tag of V1 is at most 25 bytes, so a longer one is of no known schema; the +/// bound also keeps an input-sized tag out of the errors. It is an +/// implementation limit (spec §74). +const int maxTypeTagLen = 64; + +/// 2^64-1, the largest unsigned integer of CBOR. +final BigInt maxUint64 = (BigInt.one << 64) - BigInt.one; + +// Go's math.MaxInt, 2^63-1: the bound of the map that Peek reads. +final BigInt _maxInt64 = (BigInt.one << 63) - BigInt.one; + +// 2^32, the weight of the high half of an argument of eight bytes. +const _twoPow32 = 0x100000000; + +// The low half of an argument of eight bytes, as a mask of a BigInt. +final BigInt _low32 = BigInt.from(0xffffffff); + +// The largest high half of an argument that is at most 2^53-1. +const _maxSafeHigh = 0x1fffff; + +// Major types of the profile (spec §58). +const _majorUint = 0; +const _majorBytes = 2; +const _majorText = 3; +const _majorArray = 4; +const _majorMap = 5; + +const _majorNames = [ + 'an unsigned integer', + 'a negative integer', + 'a byte string', + 'a text string', + 'an array', + 'a map', + 'a tag', + 'a float or simple value', +]; + +/// The ERR_NON_CANONICAL_CBOR error `codec: `, as errorf of the +/// reference. +DateKeysException _cborError(String detail) => + DateKeysException(ErrorCode.nonCanonicalCbor, 'codec: $detail'); + +// Go's %#02x of an initial byte. +String _initialByte(int b) => '0x${b.toRadixString(16).padLeft(2, '0')}'; + +// Whether a > b, for two unsigned integers that are each an int or a BigInt. +bool _above(Object a, Object b) => + a is int && b is int ? a > b : _bigOf(a) > _bigOf(b); + +BigInt _bigOf(Object v) => v is BigInt ? v : BigInt.from(v as int); + +// --------------------------------------------------------------------------- +// Encoder + +/// Writes the deterministic encoding of data items of the profile: every +/// integer and length in its shortest form, definite lengths only. The first +/// error is kept and later calls do nothing; [out] throws it. +/// +/// An encoder does not know the schema: the caller writes the map keys, as +/// unsigned integers in ascending order, and as many entries and items as it +/// announced. The decoder of the schema checks both on the output (spec §72). +/// +/// An encoding may hold secrets, such as I_PAYLOAD or access_material: the +/// encoder wipes every buffer it outgrows, so that the output is the only +/// copy, and the caller wipes the output. +final class CborEncoder { + /// An encoder whose buffer has room for [capacity] bytes before it grows. + CborEncoder({int capacity = 0}) : _buf = Uint8List(capacity); + + Uint8List _buf; + int _len = 0; + Exception? _err; + + // Makes room for n more bytes, wiping the buffer it outgrows. + void _grow(int n) { + if (_buf.length - _len >= n) return; + final b = Uint8List(2 * _buf.length + n)..setRange(0, _len, _buf); + _buf.fillRange(0, _len, 0); + _buf = b; + } + + // Appends the head of a data item: its major type and the argument + // hi * 2^32 + lo, each half below 2^32. + void _head(int major, int hi, int lo) { + if (_err != null) return; + final ib = major << 5; + if (hi == 0 && lo < 24) { + _grow(1); + _buf[_len++] = ib | lo; + } else if (hi == 0 && lo <= 0xff) { + _grow(2); + _buf[_len++] = ib | 24; + _buf[_len++] = lo; + } else if (hi == 0 && lo <= 0xffff) { + _grow(3); + _buf[_len++] = ib | 25; + _buf[_len++] = lo >> 8; + _buf[_len++] = lo & 0xff; + } else if (hi == 0) { + _grow(5); + _buf[_len++] = ib | 26; + _put32(lo); + } else { + _grow(9); + _buf[_len++] = ib | 27; + _put32(hi); + _put32(lo); + } + } + + // Appends v, below 2^32, in four bytes, big-endian, without a bit + // operation on more than 31 bits. + void _put32(int v) { + final top = v ~/ 0x1000000; + final rest = v - top * 0x1000000; + _buf[_len++] = top; + _buf[_len++] = rest >> 16; + _buf[_len++] = rest >> 8 & 0xff; + _buf[_len++] = rest & 0xff; + } + + // Appends the head of an argument that is a non-negative int. + void _headInt(int major, int v) { + if (v < _twoPow32) { + _head(major, 0, v); + } else { + final hi = v ~/ _twoPow32; + _head(major, hi, v - hi * _twoPow32); + } + } + + void _append(List b) { + if (_err != null) return; + _grow(b.length); + _buf.setRange(_len, _len + b.length, b); + _len += b.length; + } + + /// Records [error] as the error of the encoding unless one is recorded + /// already. Every later call does nothing, and [out] throws the first + /// error. The encoder of a schema calls it when its value breaks a rule of + /// the schema, so that bytes the decoder rejects are never returned. + void fail(Exception error) { + _err ??= error; + } + + /// Writes the head of a map of [pairs] entries. The caller then writes each + /// key, with [uint], followed by its value. + void map(int pairs) { + if (pairs < 0) { + fail(_cborError('map of $pairs entries')); + return; + } + _headInt(_majorMap, pairs); + } + + /// Writes the head of an array of [items] items. The caller then writes + /// each item. + void array(int items) { + if (items < 0) { + fail(_cborError('array of $items items')); + return; + } + _headInt(_majorArray, items); + } + + /// Writes the unsigned integer [v], which must be in 0..2^53-1, the range of + /// every schema of the protocol and of an exact int on every platform. + /// [uint64] writes any unsigned integer of CBOR. + void uint(int v) { + if (v < 0 || v > maxSafeUint) { + fail(_cborError('$v is not an unsigned integer in 0..2^53-1')); + return; + } + _headInt(_majorUint, v); + } + + /// Writes the unsigned integer [v], which must be in 0..2^64-1. + void uint64(BigInt v) { + if (v.isNegative || v > maxUint64) { + fail(_cborError('$v is not an unsigned integer in 0..2^64-1')); + return; + } + _head(_majorUint, (v >> 32).toInt(), (v & _low32).toInt()); + } + + /// Writes a byte string. + void bstr(List b) { + _headInt(_majorBytes, b.length); + _append(b); + } + + /// Writes a text string, which must be valid Unicode: a lone surrogate, + /// which a Dart String may hold, has no UTF-8 encoding. The error quotes the + /// string as Go's `%q` quotes the bytes of the generalized UTF-8 of such a + /// string, as [utf8Bytes] writes them. + void text(String s) { + final b = utf8Bytes(s); + if (!isWellFormedUtf16(s)) { + fail(_cborError('text string ${goQuote(b)} is not valid UTF-8')); + return; + } + _headInt(_majorText, b.length); + _append(b); + } + + /// Returns the encoding, or throws the first error. The encoding is a view + /// of the buffer of the encoder, not a copy, so that no other copy of it + /// remains: the caller wipes it, and writes nothing more with this encoder. + /// On error the partial output is wiped. + Uint8List out() { + final err = _err; + if (err != null) { + _buf.fillRange(0, _len, 0); + _buf = Uint8List(0); + _len = 0; + throw err; + } + return Uint8List.sublistView(_buf, 0, _len); + } +} + +// --------------------------------------------------------------------------- +// Decoder + +/// A strict cursor over the encoding of one data item of the profile. Each +/// method reads one data item, or one head, and throws ERR_NON_CANONICAL_CBOR +/// for a major type outside the profile, a major type other than the one +/// asked for, an indefinite length, an integer or length not in its shortest +/// form, a length beyond the remaining input and a value outside the bounds +/// the caller gives. Within each open map the keys are unsigned integers in +/// strictly ascending order. Every error names the offset of the input where +/// it was found: `codec: offset 3: …: ERR_NON_CANONICAL_CBOR`. +/// +/// The first error is kept: every later call throws it again. +final class CborDecoder { + /// A decoder positioned at the start of [input]. + CborDecoder(List input) + : _in = input is Uint8List ? input : Uint8List.fromList(input); + + final Uint8List _in; + int _off = 0; + final _maps = <_OpenMap>[]; + DateKeysException? _err; + + int get _remaining => _in.length - _off; + + // Records the first error, at the current offset, and returns it. + DateKeysException _fail(String detail) => + _err ??= _cborError('offset $_off: $detail'); + + void _throwIfFailed() { + final err = _err; + if (err != null) throw err; + } + + // Reads the head of the next data item and returns its major type and + // argument, an int up to 2^53-1 and a BigInt above. It rejects the major + // types outside the profile, reserved values, indefinite lengths, + // arguments not in their shortest form and truncation. + (int, Object) _head() { + _throwIfFailed(); + if (_off >= _in.length) throw _fail('truncated input'); + final b = _in[_off]; + final major = b >> 5; + final info = b & 0x1f; + if (major == 1 || major == 6 || major == 7) { + throw _fail( + '${_majorNames[major]} (initial byte ${_initialByte(b)}) ' + 'is outside the CBOR profile', + ); + } + if (info < 24) { + _off++; + return (major, info); + } + if (info == 31) { + throw _fail('indefinite length (initial byte ${_initialByte(b)})'); + } + if (info > 27) { + throw _fail( + 'reserved additional information (initial byte ${_initialByte(b)})', + ); + } + final n = 1 << (info - 24); + if (_remaining < 1 + n) throw _fail('truncated input'); + final p = _off + 1; + final Object arg; + final int min; + switch (n) { + case 1: + arg = _in[p]; + min = 24; + case 2: + arg = _in[p] << 8 | _in[p + 1]; + min = 0x100; + case 4: + arg = _uint32At(p); + min = 0x10000; + default: + final hi = _uint32At(p); + final lo = _uint32At(p + 4); + arg = hi <= _maxSafeHigh + ? hi * _twoPow32 + lo + : BigInt.from(hi) << 32 | BigInt.from(lo); + min = _twoPow32; + } + // A BigInt is above 2^53-1, and so in its shortest form. + if (arg is int && arg < min) { + throw _fail( + '$arg is not in its shortest form (initial byte ${_initialByte(b)})', + ); + } + _off = p + n; + return (major, arg); + } + + // The big-endian uint32 at p, without a bit operation on more than 31 bits. + int _uint32At(int p) => + _in[p] * 0x1000000 + (_in[p + 1] << 16 | _in[p + 2] << 8 | _in[p + 3]); + + // Reads the head of a data item of major type want. + Object _expect(int want) { + final start = _off; + final (major, arg) = _head(); + if (major != want) { + _off = start; + throw _fail( + '${_majorNames[major]} where ${_majorNames[want]} was expected', + ); + } + return arg; + } + + /// Reads the head of a map of at most [max] entries and returns the number + /// of entries. The caller reads each entry with [key] and a value, then + /// calls [endMap]. + int map(int max) => _map(max); + + // map with a bound that is an int or a BigInt, as Go's MaxInt of Peek. + int _map(Object max) { + final n = _expect(_majorMap); + if ((max is int && max < 0) || _above(n, max)) { + throw _fail('map of $n entries, at most $max'); + } + if (_above(n, _remaining ~/ 2)) { + throw _fail('truncated input: map of $n entries'); + } + // At most half the remaining input, so an int. + final pairs = n as int; + _maps.add(_OpenMap(pairs)); + return pairs; + } + + /// Reads the key of the next entry of the innermost open map: an unsigned + /// integer greater than the previous key of that map. The key is exact: an + /// [int] up to 2^53-1 and a [BigInt] above, which no schema defines. + Object key() { + _throwIfFailed(); + if (_maps.isEmpty) throw _fail('map key outside a map'); + final m = _maps.last; + if (m.left == 0) throw _fail('map key after the last entry'); + final start = _off; + final k = _expect(_majorUint); + if (m.started && !_above(k, m.last)) { + _off = start; + throw _fail( + 'map key $k after key ${m.last}: keys must be strictly ascending', + ); + } + m.left--; + m.last = k; + m.started = true; + return k; + } + + /// Closes the innermost open map, all of whose entries must have been + /// read. + void endMap() { + _throwIfFailed(); + if (_maps.isEmpty) throw _fail('end of a map outside a map'); + final left = _maps.last.left; + if (left != 0) throw _fail('$left map entries not read'); + _maps.removeLast(); + } + + /// Reads the head of an array of at most [max] items and returns the number + /// of items, which the caller then reads. + int array(int max) { + final n = _expect(_majorArray); + if (max < 0 || _above(n, max)) { + throw _fail('array of $n items, at most $max'); + } + if (_above(n, _remaining)) { + throw _fail('truncated input: array of $n items'); + } + return n as int; + } + + /// Reads an unsigned integer of at most [max], which must be in 0..2^53-1, + /// as the bounds of every schema of the protocol: the integer is then an + /// exact int on every platform. [uint64] reads up to 2^64-1. + int uint([int max = maxSafeUint]) { + if (max < 0 || max > maxSafeUint) { + throw ArgumentError.value(max, 'max', 'not in 0..2^53-1'); + } + final v = _expect(_majorUint); + if (_above(v, max)) throw _fail('unsigned integer $v above $max'); + return v as int; + } + + /// Reads an unsigned integer of at most [max], 2^64-1 when absent. + BigInt uint64([BigInt? max]) { + final bound = max ?? maxUint64; + if (bound.isNegative || bound > maxUint64) { + throw ArgumentError.value(max, 'max', 'not in 0..2^64-1'); + } + final v = _expect(_majorUint); + if (_above(v, bound)) throw _fail('unsigned integer $v above $bound'); + return _bigOf(v); + } + + // Reads a string of major type want and returns its content, a view of the + // input. The length is checked against the remaining input and then + // against min and max. + Uint8List _content(int want, int min, int max) { + final n = _expect(want); + if (_above(n, _remaining)) { + throw _fail('truncated input: ${_majorNames[want]} of $n bytes'); + } + final length = n as int; + if (length < min || length > max) { + throw _fail('${_majorNames[want]} of $length bytes outside $min..$max'); + } + final b = Uint8List.sublistView(_in, _off, _off + length); + _off += length; + return b; + } + + /// Reads a byte string of [min] to [max] bytes and returns a copy of its + /// content. The length is checked before anything is copied. + Uint8List bstr(int min, int max) => + Uint8List.fromList(_content(_majorBytes, min, max)); + + /// Reads a text string of at most [max] bytes of valid UTF-8. A leading + /// U+FEFF is part of the text. + String text(int max) { + final start = _off; + final s = decodeUtf8(_content(_majorText, 0, max)); + if (s == null) { + _off = start; + throw _fail('text string is not valid UTF-8'); + } + return s; + } + + /// Checks that every map was closed and that no byte follows the data + /// item. + void done() { + _throwIfFailed(); + if (_maps.isNotEmpty) throw _fail('${_maps.length} maps not closed'); + if (_off != _in.length) throw _fail('${_in.length - _off} trailing bytes'); + } + + // The major type of the next data item, or an unsigned integer at the end + // of the input, where reading it reports the truncation. + int _next() => _off >= _in.length ? _majorUint : _in[_off] >> 5; + + // Reads one data item for walkCbor: a scalar, or the head of a container, + // which it pushes on open. + void _walkItem(List<_WalkLevel> open, int maxDepth, int maxLen) { + switch (_next()) { + case final major when major == _majorMap || major == _majorArray: + if (open.length >= maxDepth) { + throw _fail('containers nested deeper than $maxDepth'); + } + final isMap = major == _majorMap; + open.add(_WalkLevel(isMap ? map(maxLen) : array(maxLen), isMap)); + case _majorBytes: + _content(_majorBytes, 0, maxLen); + case _majorText: + text(maxLen); + default: + // An unsigned integer, any up to 2^64-1, or the error of whatever is + // there. + _expect(_majorUint); + } + } +} + +// The state of a map between map and endMap. +final class _OpenMap { + _OpenMap(this.left); + + // Entries not read yet. + int left; + + // The last key read, an int or a BigInt. + Object last = 0; + + // Whether at least one key was read. + bool started = false; +} + +// An open container of walkCbor. +final class _WalkLevel { + _WalkLevel(this.left, this.isMap); + + // Entries of a map or items of an array not read yet. + int left; + final bool isMap; +} + +// --------------------------------------------------------------------------- +// Objects + +/// Decodes one object from [input] with [decode], checks that the whole +/// input was read, and re-encodes the decoded value with [encode]: the result +/// must reproduce [input] byte for byte. [decode] and [encode] are the two +/// halves of one schema and work on the same value. +/// +/// Errors of the decoder and of the re-encoding are ERR_NON_CANONICAL_CBOR; +/// any other exception of [decode] passes as it is. +/// +/// The re-encoding equals [input] on success, so it may hold secrets such as +/// I_PAYLOAD or access_material; it is wiped on every path, and so is every +/// buffer the encoder outgrows. +void unmarshalCbor( + List input, + void Function(CborDecoder d) decode, + void Function(CborEncoder e) encode, +) { + final d = CborDecoder(input); + decode(d); + d.done(); + final e = CborEncoder(capacity: input.length); + encode(e); + Uint8List? re; + try { + re = e.out(); + } on Exception { + // The error of the encoder is not the error of the input: the input is + // not what the encoder writes, whatever the reason. + } + try { + if (re == null || !equalBytes(re, input)) { + throw _cborError('input is not the deterministic encoding of its value'); + } + } finally { + re?.fillRange(0, re.length, 0); + } +} + +/// Reads the type tag (key 0, a text string of at most [maxTypeTagLen] +/// bytes) and the schema version (key 1, an unsigned integer of at most +/// 2^53-1) of the map at the start of [input], before strict decoding, so +/// that an unknown schema version is reported as such (spec §70). The map +/// must start with keys 0 and 1, in the profile; nothing after them is read. +/// The result must never be used as the decoded object. +({String typeTag, int version}) peekSchema(List input) { + final d = CborDecoder(input); + // The input bounds the map. + final pairs = d._map(_maxInt64); + if (pairs < 2) { + throw d._fail('map without a type tag and a schema version'); + } + var typeTag = ''; + var version = 0; + for (var want = 0; want < 2; want++) { + final k = d.key(); + if (k != want) throw d._fail('map key $k where key $want was expected'); + if (want == 0) { + typeTag = d.text(maxTypeTagLen); + } else { + version = d.uint(); + } + } + return (typeTag: typeTag, version: version); +} + +/// Reads the type tag and the schema version of an object with [peekSchema] +/// and requires [typeTag] and [version]. A different type tag is +/// ERR_NON_CANONICAL_CBOR; a different version is ERR_UNSUPPORTED_VERSION. +void checkSchema(List input, String typeTag, int version) { + final (typeTag: tag, version: v) = peekSchema(input); + if (tag != typeTag) { + throw _cborError( + 'type ${goQuote(utf8Bytes(tag))}, want ${goQuote(utf8Bytes(typeTag))}', + ); + } + if (v != version) { + throw DateKeysException( + ErrorCode.unsupportedVersion, + 'codec: $typeTag schema version $v, want $version', + ); + } +} + +/// Checks that [input] is exactly one data item of the profile, with +/// containers nested at most [maxDepth] deep (a scalar has depth 0) and every +/// string and container at most [maxLen] long: bytes of a string, items of +/// an array, entries of a map. Unsigned integers take any value up to +/// 2^64-1. It reads iteratively, so deep input cannot exhaust the stack. +/// +/// A helper for vectors and diagnostics, and for registered extensions whose +/// data is CBOR (spec §72). It never decides whether an object of the +/// protocol is valid: the decoder of its schema does. +void walkCbor(List input, int maxDepth, int maxLen) { + final d = CborDecoder(input); + final open = <_WalkLevel>[]; + for (var first = true; first || open.isNotEmpty; first = false) { + final top = open.isEmpty ? null : open.last; + if (top != null && top.left == 0) { + // The innermost container is complete. + if (top.isMap) d.endMap(); + open.removeLast(); + continue; + } + if (top != null) { + top.left--; + if (top.isMap) d.key(); + } + d._walkItem(open, maxDepth, maxLen); + } + d.done(); +} diff --git a/test/cbor_reference.dart b/test/cbor_reference.dart new file mode 100644 index 0000000..b9015ca --- /dev/null +++ b/test/cbor_reference.dart @@ -0,0 +1,223 @@ +// An encoder and a decoder of generic CBOR values for the tests, as +// internal/cbortest of datekeys-go. They are written independently of +// lib/src/cbor.dart on purpose, with BigInt arithmetic and dart:convert: +// tests use them to build inputs that the codec cannot write, such as null, +// negative integers or a float inside an otherwise valid object, and as a +// second reading of the CBOR profile of spec §58 to check the codec against. + +import 'dart:convert'; +import 'dart:typed_data'; + +/// An encoded data item that [marshal] writes verbatim. +final class Raw { + const Raw(this.bytes); + + final List bytes; +} + +/// A text string held as its bytes, as [unmarshal] returns it: dart:convert +/// would drop a leading U+FEFF. +final class Text { + const Text(this.bytes); + + final Uint8List bytes; +} + +/// The deterministic encoding of [v], which is one of: an int (a negative one +/// as major type 1) or a BigInt (unsigned), a String or a [Text], a +/// Uint8List, a bool, null, a List of values, a Map whose keys are ints or +/// BigInts (written in ascending order), or a [Raw]. +Uint8List marshal(Object? v) { + final out = []; + _append(out, v); + return Uint8List.fromList(out); +} + +void _append(List out, Object? v) { + switch (v) { + case null: + out.add(0xf6); + case final bool b: + out.add(b ? 0xf5 : 0xf4); + case final int i when i < 0: + _head(out, 1, BigInt.from(-(i + 1))); + case final int i: + _head(out, 0, BigInt.from(i)); + case final BigInt i: + _head(out, 0, i); + case final Uint8List b: + _head(out, 2, BigInt.from(b.length)); + out.addAll(b); + case final String s: + final b = utf8.encode(s); + _head(out, 3, BigInt.from(b.length)); + out.addAll(b); + case final Text t: + _head(out, 3, BigInt.from(t.bytes.length)); + out.addAll(t.bytes); + case final Raw r: + out.addAll(r.bytes); + case final List l: + _head(out, 4, BigInt.from(l.length)); + for (final x in l) { + _append(out, x); + } + case final Map m: + final keys = m.keys.toList()..sort((a, b) => _big(a).compareTo(_big(b))); + _head(out, 5, BigInt.from(m.length)); + for (final k in keys) { + _head(out, 0, _big(k)); + _append(out, m[k]); + } + default: + throw ArgumentError('cannot encode ${v.runtimeType}'); + } +} + +BigInt _big(Object k) => k is BigInt ? k : BigInt.from(k as int); + +void _head(List out, int major, BigInt arg) { + final m = major << 5; + if (arg < BigInt.from(24)) { + out.add(m | arg.toInt()); + return; + } + final int n; + if (arg < BigInt.from(0x100)) { + out.add(m | 24); + n = 1; + } else if (arg < BigInt.from(0x10000)) { + out.add(m | 25); + n = 2; + } else if (arg < BigInt.one << 32) { + out.add(m | 26); + n = 4; + } else { + out.add(m | 27); + n = 8; + } + for (var i = n - 1; i >= 0; i--) { + out.add(((arg >> (8 * i)) & BigInt.from(0xff)).toInt()); + } +} + +/// The nesting that [unmarshal] follows at most. +const maxDepth = 1000; + +/// Decodes exactly one data item of the CBOR profile of spec §58 into BigInt, +/// Uint8List, [Text], List and Map with BigInt keys in their order. It throws +/// a FormatException for every other major type, an indefinite length, a +/// head not in its shortest form, a map key that is not an unsigned integer +/// greater than the previous one, invalid UTF-8, truncation, trailing bytes +/// and nesting deeper than [maxDepth]. +Object? unmarshal(List b) { + final r = _Reader(b); + final v = r.value(0); + if (r.off != b.length) { + throw FormatException('${b.length - r.off} trailing bytes'); + } + return v; +} + +final class _Reader { + _Reader(this.b); + + final List b; + int off = 0; + + (int, BigInt) head() { + if (off >= b.length) throw const FormatException('truncated'); + final ib = b[off++]; + final major = ib >> 5; + final info = ib & 0x1f; + if (major == 1 || major >= 6) throw FormatException('major type $major'); + if (info < 24) return (major, BigInt.from(info)); + if (info > 27) throw FormatException('additional information $info'); + final n = 1 << (info - 24); + if (b.length - off < n) throw const FormatException('truncated'); + var arg = BigInt.zero; + for (var i = 0; i < n; i++) { + arg = arg << 8 | BigInt.from(b[off + i]); + } + off += n; + if ((n == 1 && arg < BigInt.from(24)) || + (n > 1 && arg >> (4 * n) == BigInt.zero)) { + throw FormatException('$arg not in its shortest form'); + } + return (major, arg); + } + + Uint8List bytes(BigInt n) { + if (n > BigInt.from(b.length - off)) { + throw const FormatException('truncated'); + } + final s = Uint8List.fromList(b.sublist(off, off + n.toInt())); + off += n.toInt(); + return s; + } + + Object? value(int depth) { + final (major, arg) = head(); + switch (major) { + case 0: + return arg; + case 2: + return bytes(arg); + case 3: + final s = bytes(arg); + // dart:convert is strict about everything but a leading U+FEFF, + // which it drops: the text keeps its bytes. + const Utf8Decoder().convert(s); + return Text(s); + } + if (depth >= maxDepth) throw const FormatException('nested too deep'); + if (arg > BigInt.from(b.length - off)) { + throw const FormatException('truncated'); + } + final count = arg.toInt(); + if (major == 4) { + return [for (var i = 0; i < count; i++) value(depth + 1)]; + } + final out = {}; + BigInt? last; + for (var i = 0; i < count; i++) { + final (km, k) = head(); + if (km != 0) throw FormatException('map key of major type $km'); + if (last != null && k <= last) { + throw FormatException('map key $k after $last'); + } + last = k; + out[k] = value(depth + 1); + } + return out; + } +} + +/// The nesting depth of [v], a value of [marshal] or [unmarshal], and the +/// length of its longest string or container: bytes of a string, items of an +/// array, entries of a map. +(int, int) shape(Object? v) { + final Iterable items; + switch (v) { + case final Uint8List b: + return (0, b.length); + case final String s: + return (0, utf8.encode(s).length); + case final Text t: + return (0, t.bytes.length); + case final List l: + items = l; + case final Map m: + items = m.values; + default: + return (0, 0); + } + var depth = 0; + var length = items.length; + for (final x in items) { + final (d, l) = shape(x); + if (d > depth) depth = d; + if (l > length) length = l; + } + return (depth + 1, length); +} diff --git a/test/cbor_test.dart b/test/cbor_test.dart new file mode 100644 index 0000000..d12b4cf --- /dev/null +++ b/test/cbor_test.dart @@ -0,0 +1,1806 @@ +// Tests of the CBOR profile of spec §58, ported from the tests of package +// codec of datekeys-go (codec_test.go and internal_test.go, the fuzz targets +// as properties over seeded inputs) and from cbor.test.ts of datekeys-ts. +// Every error text is the one that the Go reference prints for the same +// input (Go 1.26.8, datekeys-go at 601e6d2). + +import 'dart:convert'; +import 'dart:math'; +import 'dart:typed_data'; + +import 'package:datekeys/datekeys.dart'; +import 'package:test/test.dart'; + +import 'cbor_reference.dart' as ref; + +const nc = 'ERR_NON_CANONICAL_CBOR'; + +// One backslash, for the texts of Go's %q. +const bs = '\\'; + +Uint8List h(String hex) => fromHex(hex.replaceAll(' ', '')); + +String hexOf(String s) => toHex(utf8.encode(s)); + +final maxSafe = BigInt.from(maxSafeUint); +final twoPow53 = BigInt.two.pow(53); +final twoPow63 = BigInt.two.pow(63); + +/// The DateKeysException that [f] throws. +DateKeysException thrown(void Function() f) { + try { + f(); + } on DateKeysException catch (e) { + return e; + } + fail('no DateKeysException'); +} + +/// The message of the DateKeysException that [f] throws. +String messageOf(void Function() f) => thrown(f).message; + +/// The code of what [f] throws, or '' when it returns. +String codeOf(void Function() f) { + try { + f(); + } on DateKeysException catch (e) { + return e.code.code; + } + return ''; +} + +String encoded(void Function(CborEncoder e) write) { + final e = CborEncoder(); + write(e); + return toHex(e.out()); +} + +/// Runs [prog], a space-separated list of decoder calls, on the hexadecimal +/// [input] and returns the first error, after checking that every later call +/// throws the same exception (the run of the Go tests): +/// +/// m map k key e endMap a array u[] uint64 +/// b, bstr t text d done +DateKeysException? run(String input, String prog) { + final d = CborDecoder(h(input)); + DateKeysException? first; + for (final op in prog.split(' ').where((s) => s.isNotEmpty)) { + final arg = op.substring(1); + DateKeysException? err; + try { + switch (op[0]) { + case 'm': + d.map(int.parse(arg)); + case 'k': + d.key(); + case 'e': + d.endMap(); + case 'a': + d.array(int.parse(arg)); + case 'u': + d.uint64(arg.isEmpty ? null : BigInt.parse(arg)); + case 'b': + final [lo, hi] = arg.split(',').map(int.parse).toList(); + d.bstr(lo, hi); + case 't': + d.text(int.parse(arg)); + case 'd': + d.done(); + default: + throw ArgumentError('bad op $op'); + } + } on DateKeysException catch (e) { + err = e; + } + if (first == null) { + first = err; + } else if (!identical(err, first)) { + fail('error not sticky: $first, then $err'); + } + } + return first; +} + +/// The sample schema of the Go tests: {0: tstr, 1: uint, 2: bstr, ? 10: +/// [* uint]}. Its decoder accepts an empty list at key 10 and its encoder +/// omits an empty list, so an empty list that is present is left to the +/// re-encoding check. +final class Sample { + String type = ''; + BigInt n = BigInt.zero; + Uint8List? bytes; + List list = []; + + void decode(CborDecoder d) { + final pairs = d.map(4); + for (var i = 0; i < pairs; i++) { + switch (d.key()) { + case 0: + type = d.text(16); + case 1: + n = d.uint64(); + case 2: + bytes = d.bstr(0, 64); + case 10: + list = [for (var j = d.array(8); j > 0; j--) d.uint64()]; + case final k: + throw DateKeysException(ErrorCode.nonCanonicalCbor, 'unknown key $k'); + } + } + if (type.isEmpty || bytes == null) { + throw DateKeysException(ErrorCode.nonCanonicalCbor, 'missing key'); + } + d.endMap(); + } + + void encode(CborEncoder e) { + e + ..map(list.isEmpty ? 3 : 4) + ..uint(0) + ..text(type) + ..uint(1) + ..uint64(n) + ..uint(2) + ..bstr(bytes ?? Uint8List(0)); + if (list.isNotEmpty) { + e + ..uint(10) + ..array(list.length); + for (final v in list) { + e.uint64(v); + } + } + } + + Uint8List marshal() { + final e = CborEncoder(); + encode(e); + return e.out(); + } +} + +Sample unmarshalSample(String hex) { + final s = Sample(); + unmarshalCbor(h(hex), s.decode, s.encode); + return s; +} + +/// A random unsigned integer of 64 bits shifted right by 0 to 63 bits, as +/// `r.Uint64() >> r.IntN(64)` of the Go tests. +BigInt randomUint64(Random r) => + (BigInt.from(r.nextInt(0x100000000)) << 32 | + BigInt.from(r.nextInt(0x100000000))) >> + r.nextInt(64); + +/// A random value of the profile, in the types of cbor_reference.dart, as +/// randomValue of the Go tests. +Object? randomValue(Random r, int depth) { + final k = r.nextInt(6); + if (k == 0 && depth < 4) { + return [for (var i = r.nextInt(4); i > 0; i--) randomValue(r, depth + 1)]; + } + if (k == 1 && depth < 4) { + final m = {}; + for (var i = r.nextInt(4); i > 0; i--) { + m[randomUint64(r)] = randomValue(r, depth + 1); + } + return m; + } + if (k == 2) return Uint8List(r.nextInt(30)); + if (k == 3) return String.fromCharCode(0xe9) * r.nextInt(20); + return randomUint64(r); +} + +/// Writes [v], a value of [randomValue], with the encoder: integers up to +/// 2^53-1 through uint and the larger ones through uint64. +void encodeValue(CborEncoder e, Object? v) { + void writeUint(BigInt i) => i <= maxSafe ? e.uint(i.toInt()) : e.uint64(i); + switch (v) { + case final BigInt i: + writeUint(i); + case final Uint8List b: + e.bstr(b); + case final String s: + e.text(s); + case final List l: + e.array(l.length); + for (final x in l) { + encodeValue(e, x); + } + case final Map m: + e.map(m.length); + for (final k in m.keys.toList()..sort()) { + writeUint(k); + encodeValue(e, m[k]); + } + default: + throw ArgumentError('${v.runtimeType}'); + } +} + +/// [b] with one to three random edits, as the mutations of a fuzzer. +Uint8List mutate(Random r, List b) { + final out = List.of(b); + for (var n = 1 + r.nextInt(3); n > 0; n--) { + final k = r.nextInt(6); + if (k == 0 && out.isNotEmpty) { + out[r.nextInt(out.length)] ^= 1 << r.nextInt(8); + } else if (k == 1 && out.isNotEmpty) { + out[r.nextInt(out.length)] = r.nextInt(256); + } else if (k == 2 && out.isNotEmpty) { + out.length = r.nextInt(out.length); + } else if (k == 3) { + out.insert(r.nextInt(out.length + 1), r.nextInt(256)); + } else if (k == 4 && out.length > 1) { + out.removeAt(r.nextInt(out.length)); + } else { + out.add(r.nextInt(256)); + } + } + return Uint8List.fromList(out); +} + +/// The seeds of the fuzz targets of the Go tests, and encodings of random +/// values. +List fuzzSeeds(Random r) => [ + for (final s in [ + 'a400617801170241010a82011901f4', + 'a30061780117024101', + 'a3006178011702f6', + '9f01ff', + 'a200010101', + 'a1008181a10040', + '64efbbbf61', + '1bffffffffffffffff', + 'a2006a646174656b657963617001' + '01', + ]) + h(s), + for (var i = 0; i < 100; i++) ref.marshal(randomValue(r, 0)), +]; + +bool containsBytes(List haystack, List needle) { + for (var i = 0; i + needle.length <= haystack.length; i++) { + var j = 0; + while (j < needle.length && haystack[i + j] == needle[j]) { + j++; + } + if (j == needle.length) return true; + } + return false; +} + +void main() { + group('CborEncoder', () { + test('writes every integer and length in its shortest form', () { + final cases = <(int, String)>[ + (0, '00'), + (23, '17'), + (24, '1818'), + (255, '18ff'), + (256, '190100'), + (65535, '19ffff'), + (65536, '1a00010000'), + (0xffffffff, '1affffffff'), + (0x100000000, '1b0000000100000000'), + (maxSafeUint, '1b001fffffffffffff'), + ]; + for (final (v, want) in cases) { + expect(encoded((e) => e.uint(v)), want, reason: '$v'); + expect(encoded((e) => e.uint64(BigInt.from(v))), want, reason: '$v'); + } + expect(encoded((e) => e.uint64(maxUint64)), '1bffffffffffffffff'); + // {0: "x", 1: 23, 2: h'01', 10: [1, 500], 11: h'', 12: ""} + final map = encoded( + (e) => e + ..map(6) + ..uint(0) + ..text('x') + ..uint(1) + ..uint(23) + ..uint(2) + ..bstr([1]) + ..uint(10) + ..array(2) + ..uint(1) + ..uint(500) + ..uint(11) + ..bstr([]) + ..uint(12) + ..text(''), + ); + expect(map, 'a600617801170241010a82011901f40b400c60'); + final long = 'a' * 24; + final s = hexOf(long); + expect( + encoded( + (e) => e + ..text(long) + ..bstr(utf8.encode(long)) + ..map(24) + ..array(256), + ), + '7818${s}5818${s}b818990100', + ); + expect( + encoded( + (e) => e + ..map(0x100000000) + ..array(0x100000000), + ), + 'bb00000001000000009b0000000100000000', + ); + }); + + test('writes text as UTF-8, a leading U+FEFF included', () { + expect( + encoded((e) => e.text(String.fromCharCodes([0xfeff, 0x61]))), + '64efbbbf61', + ); + expect( + encoded((e) => e.text(String.fromCharCode(0x10000))), + '64f0908080', + ); + expect( + encoded((e) => e.text(String.fromCharCode(0x10ffff))), + '64f48fbfbf', + ); + }); + + test('keeps the first error: later calls do nothing and out throws it', () { + final first = DateKeysException(ErrorCode.nonCanonicalCbor, 'first'); + final failures = { + 'negative map': ( + (e) => e.map(-1), + 'codec: map of -1 entries: ERR_NON_CANONICAL_CBOR', + ), + 'negative array': ( + (e) => e.array(-2), + 'codec: array of -2 items: ERR_NON_CANONICAL_CBOR', + ), + // A Dart String cannot hold the invalid UTF-8 of the Go tests, "\xff" + // and the overlong "\xc0\x80", but it can hold a lone surrogate. + 'lone surrogate': ( + (e) => e.text(String.fromCharCode(0xd800)), + 'codec: text string "${bs}xed${bs}xa0${bs}x80" is not valid UTF-8: ' + 'ERR_NON_CANONICAL_CBOR', + ), + 'lone surrogates': ( + (e) => e.text(String.fromCharCodes([0x61, 0xdc00, 0x62, 0xd800])), + 'codec: text string "a${bs}xed${bs}xb0${bs}x80b${bs}xed${bs}xa0${bs}x80"' + ' is not valid UTF-8: ERR_NON_CANONICAL_CBOR', + ), + // Texts of Dart only: Go's Uint takes a uint64, which is never + // negative nor above 2^64-1, and has no limit of 2^53-1. + 'negative uint': ( + (e) => e.uint(-1), + 'codec: -1 is not an unsigned integer in 0..2^53-1: ' + 'ERR_NON_CANONICAL_CBOR', + ), + 'uint above 2^53-1': ( + (e) => e.uint(maxSafeUint + 1), + 'codec: 9007199254740992 is not an unsigned integer in 0..2^53-1: ' + 'ERR_NON_CANONICAL_CBOR', + ), + 'negative uint64': ( + (e) => e.uint64(BigInt.from(-1)), + 'codec: -1 is not an unsigned integer in 0..2^64-1: ' + 'ERR_NON_CANONICAL_CBOR', + ), + 'uint64 above 2^64-1': ( + (e) => e.uint64(maxUint64 + BigInt.one), + 'codec: 18446744073709551616 is not an unsigned integer in ' + '0..2^64-1: ERR_NON_CANONICAL_CBOR', + ), + 'fail': ((e) => e.fail(first), 'first: ERR_NON_CANONICAL_CBOR'), + 'fail twice': ( + (e) => e + ..fail(first) + ..fail(FormatException('second')), + 'first: ERR_NON_CANONICAL_CBOR', + ), + }; + for (final MapEntry(key: name, value: (failure, text)) + in failures.entries) { + final e = CborEncoder()..bstr(utf8.encode('secret')); + failure(e); + e + ..uint(1) + ..bstr([1]) + ..text('ok') + ..map(1) + ..array(1); + final err = thrown(e.out); + expect(err.message, text, reason: name); + expect(err.code, ErrorCode.nonCanonicalCbor, reason: name); + expect(identical(thrown(e.out), err), isTrue, reason: name); + } + // Go's Fail(nil), which records nothing, has no Dart counterpart: fail + // takes an exception. + }); + + test('wipes every buffer it outgrows', () { + // The internal test of the reference, through the view that out + // returns: no stale copy of a secret is left behind. + final secret = Uint8List(32)..fillRange(0, 32, 0xab); + final e = CborEncoder()..bstr(secret); + final old = e.out().buffer.asUint8List(); + e.bstr(Uint8List(2 * old.length)); + expect(containsBytes(old, secret.sublist(0, 8)), isFalse); + expect(containsBytes(e.out(), secret), isTrue); + // A buffer with room is not replaced. + final f = CborEncoder(capacity: 64)..text('fits'); + expect(f.out().buffer.lengthInBytes, 64); + }); + + test('wipes the partial output when it fails', () { + final e = CborEncoder()..bstr(Uint8List(8)..fillRange(0, 8, 0xab)); + final partial = e.out(); + e.map(-1); + expect(e.out, throwsA(isA())); + expect(partial, everyElement(0)); + }); + }); + + group('CborDecoder', () { + test('accepts the items of the profile', () { + final cases = <(String, String, String)>[ + ('uint 23 inline', '17', 'u23 d'), + ('uint 24 one byte', '1818', 'u24 d'), + ('uint 256 two bytes', '190100', 'u d'), + ('uint 65536 four bytes', '1a00010000', 'u d'), + ('uint 2^32 eight bytes', '1b0000000100000000', 'u d'), + ('uint 2^64-1', '1bffffffffffffffff', 'u d'), + ('empty bstr', '40', 'b0,0 d'), + ('bstr at its bounds', '420102', 'b2,2 d'), + ('bstr 24 bytes', '5818${'00' * 24}', 'b0,24 d'), + ('empty text', '60', 't0 d'), + ('text with leading BOM', '64efbbbf61', 't4 d'), + ('text U+10FFFF', '64f48fbfbf', 't4 d'), + ('empty map', 'a0', 'm0 e d'), + ('map two sorted keys', 'a200010101', 'm2 k u k u e d'), + ( + 'map keys 0 and 2^64-1', + 'a200001bffffffffffffffff00', + 'm2 k u k u e d', + ), + ('empty array', '80', 'a0 d'), + ('array of maps', '82a10000a10101', 'a2 m1 k u e m1 k u e d'), + ('nested maps', 'a100a10000', 'm1 k m1 k u e e d'), + ]; + for (final (name, hex, prog) in cases) { + expect(run(hex, prog), isNull, reason: name); + } + }); + + test('rejects everything else with the text of the reference', () { + final cases = <(String, String, String, String)>[ + // Outside the profile of spec §58. + ( + 'negative int', + '20', + 'u', + 'offset 0: a negative integer (initial byte 0x20) is outside the CBOR profile', + ), + ( + 'tag', + 'c101', + 'u', + 'offset 0: a tag (initial byte 0xc1) is outside the CBOR profile', + ), + ( + 'tag on a byte string', + 'c24101', + 'b0,9', + 'offset 0: a tag (initial byte 0xc2) is outside the CBOR profile', + ), + ( + 'half float', + 'f97e00', + 'u', + 'offset 0: a float or simple value (initial byte 0xf9) is outside the CBOR profile', + ), + ( + 'single float', + 'fa3f800000', + 'u', + 'offset 0: a float or simple value (initial byte 0xfa) is outside the CBOR profile', + ), + ( + 'double float', + 'fb3ff0000000000000', + 'u', + 'offset 0: a float or simple value (initial byte 0xfb) is outside the CBOR profile', + ), + ( + 'false', + 'f4', + 'u', + 'offset 0: a float or simple value (initial byte 0xf4) is outside the CBOR profile', + ), + ( + 'true', + 'f5', + 'u', + 'offset 0: a float or simple value (initial byte 0xf5) is outside the CBOR profile', + ), + ( + 'null', + 'f6', + 'b0,9', + 'offset 0: a float or simple value (initial byte 0xf6) is outside the CBOR profile', + ), + ( + 'undefined', + 'f7', + 't9', + 'offset 0: a float or simple value (initial byte 0xf7) is outside the CBOR profile', + ), + ( + 'break', + 'ff', + 'u', + 'offset 0: a float or simple value (initial byte 0xff) is outside the CBOR profile', + ), + ( + 'indefinite array', + '9f01ff', + 'a9', + 'offset 0: indefinite length (initial byte 0x9f)', + ), + ( + 'indefinite map', + 'bf0001ff', + 'm9', + 'offset 0: indefinite length (initial byte 0xbf)', + ), + ( + 'indefinite byte string', + '5f4101ff', + 'b0,9', + 'offset 0: indefinite length (initial byte 0x5f)', + ), + ( + 'indefinite text', + '7f6161ff', + 't9', + 'offset 0: indefinite length (initial byte 0x7f)', + ), + ( + 'reserved 28', + '1c', + 'u', + 'offset 0: reserved additional information (initial byte 0x1c)', + ), + ( + 'reserved 29', + '1d', + 'u', + 'offset 0: reserved additional information (initial byte 0x1d)', + ), + ( + 'reserved 30', + '1e', + 'u', + 'offset 0: reserved additional information (initial byte 0x1e)', + ), + // Shortest form. + ( + 'uint 23 with one extra byte', + '1817', + 'u', + 'offset 0: 23 is not in its shortest form (initial byte 0x18)', + ), + ( + 'uint 255 in two bytes', + '1900ff', + 'u', + 'offset 0: 255 is not in its shortest form (initial byte 0x19)', + ), + ( + 'uint 65535 in four bytes', + '1a0000ffff', + 'u', + 'offset 0: 65535 is not in its shortest form (initial byte 0x1a)', + ), + ( + 'uint 2^32-1 in eight bytes', + '1b00000000ffffffff', + 'u', + 'offset 0: 4294967295 is not in its shortest form (initial byte 0x1b)', + ), + ( + 'bstr length not shortest', + '5800', + 'b0,9', + 'offset 0: 0 is not in its shortest form (initial byte 0x58)', + ), + ( + 'map length not shortest', + 'b800', + 'm9', + 'offset 0: 0 is not in its shortest form (initial byte 0xb8)', + ), + ( + 'key not shortest', + 'a1180000', + 'm1 k', + 'offset 1: 0 is not in its shortest form (initial byte 0x18)', + ), + // Truncation. + ('empty input', '', 'u', 'offset 0: truncated input'), + ('truncated uint', '1901', 'u', 'offset 0: truncated input'), + ( + 'truncated uint 8', + '1b00000000000000', + 'u', + 'offset 0: truncated input', + ), + ( + 'length beyond input', + '5affffffff', + 'b0,9', + 'offset 5: truncated input: a byte string of 4294967295 bytes', + ), + ( + 'bstr shorter than its length', + '4300', + 'b0,9', + 'offset 1: truncated input: a byte string of 3 bytes', + ), + ( + 'text shorter than its length', + '6361', + 't9', + 'offset 1: truncated input: a text string of 3 bytes', + ), + ( + 'map beyond input', + 'a300', + 'm9', + 'offset 1: truncated input: map of 3 entries', + ), + ( + 'huge map', + 'bbffffffffffffffff', + 'm100', + 'offset 9: map of 18446744073709551615 entries, at most 100', + ), + ( + 'array beyond input', + '8300', + 'a9', + 'offset 1: truncated input: array of 3 items', + ), + ( + 'huge array', + '9b7fffffffffffffff', + 'a100', + 'offset 9: array of 9223372036854775807 items, at most 100', + ), + ( + 'truncated inside a map', + 'a200', + 'm2 k u', + 'offset 1: truncated input: map of 2 entries', + ), + // Types and bounds. + ( + 'bstr where uint', + '4100', + 'u', + 'offset 0: a byte string where an unsigned integer was expected', + ), + ( + 'uint where bstr', + '00', + 'b0,9', + 'offset 0: an unsigned integer where a byte string was expected', + ), + ( + 'text where bstr', + '6161', + 'b0,9', + 'offset 0: a text string where a byte string was expected', + ), + ( + 'bstr where text', + '4161', + 't9', + 'offset 0: a byte string where a text string was expected', + ), + ( + 'array where map', + '80', + 'm9', + 'offset 0: an array where a map was expected', + ), + ( + 'map where array', + 'a0', + 'a9', + 'offset 0: a map where an array was expected', + ), + ( + 'uint above max', + '1818', + 'u23', + 'offset 2: unsigned integer 24 above 23', + ), + ( + 'bstr below min', + '4101', + 'b2,9', + 'offset 1: a byte string of 1 bytes outside 2..9', + ), + ( + 'bstr above max', + '420102', + 'b0,1', + 'offset 1: a byte string of 2 bytes outside 0..1', + ), + ( + 'text above max', + '626161', + 't1', + 'offset 1: a text string of 2 bytes outside 0..1', + ), + ( + 'map above max', + 'a200010101', + 'm1', + 'offset 1: map of 2 entries, at most 1', + ), + ( + 'negative map max', + 'a0', + 'm-1', + 'offset 1: map of 0 entries, at most -1', + ), + ( + 'array above max', + '820101', + 'a1', + 'offset 1: array of 2 items, at most 1', + ), + ( + 'negative array max', + '80', + 'a-1', + 'offset 1: array of 0 items, at most -1', + ), + // Keys. + ( + 'keys out of order', + 'a201000001', + 'm2 k u k', + 'offset 3: map key 0 after key 1: keys must be strictly ascending', + ), + ( + 'duplicate key', + 'a200000001', + 'm2 k u k', + 'offset 3: map key 0 after key 0: keys must be strictly ascending', + ), + ( + 'text key', + 'a1616100', + 'm1 k', + 'offset 1: a text string where an unsigned integer was expected', + ), + ( + 'bstr key', + 'a1416100', + 'm1 k', + 'offset 1: a byte string where an unsigned integer was expected', + ), + ( + 'negative key', + 'a12000', + 'm1 k', + 'offset 1: a negative integer (initial byte 0x20) is outside the CBOR profile', + ), + ('key outside a map', '00', 'k', 'offset 0: map key outside a map'), + ( + 'key after the last entry', + 'a10000', + 'm1 k u k', + 'offset 3: map key after the last entry', + ), + ( + 'key after the map closed', + 'a0', + 'm0 e k', + 'offset 1: map key outside a map', + ), + ( + 'end outside a map', + '00', + 'e', + 'offset 0: end of a map outside a map', + ), + ( + 'end with entries left', + 'a10000', + 'm1 e', + 'offset 1: 1 map entries not read', + ), + ('done with a map open', 'a0', 'm0 d', 'offset 1: 1 maps not closed'), + // Trailing bytes and UTF-8. + ('trailing byte', '0100', 'u d', 'offset 1: 1 trailing bytes'), + ( + 'invalid UTF-8', + '61ff', + 't9', + 'offset 0: text string is not valid UTF-8', + ), + ( + 'overlong UTF-8', + '62c080', + 't9', + 'offset 0: text string is not valid UTF-8', + ), + ( + 'UTF-8 surrogate', + '63eda080', + 't9', + 'offset 0: text string is not valid UTF-8', + ), + ( + 'above U+10FFFF', + '64f4908080', + 't9', + 'offset 0: text string is not valid UTF-8', + ), + ( + 'truncated UTF-8', + '62e282', + 't9', + 'offset 0: text string is not valid UTF-8', + ), + // The first error stays. + ( + 'sticky after a failed read', + 'f600', + 'u u d', + 'offset 0: a float or simple value (initial byte 0xf6) is outside the CBOR profile', + ), + ]; + for (final (name, hex, prog, text) in cases) { + final err = run(hex, prog); + expect(err?.message, 'codec: $text: $nc', reason: name); + expect(err?.code, ErrorCode.nonCanonicalCbor, reason: name); + } + }); + + test('names the offset of the item in its errors', () { + expect( + run('a2 00 01 00 02', 'm2 k u k')?.message, + 'codec: offset 3: map key 0 after key 0: keys must be strictly ' + 'ascending: ERR_NON_CANONICAL_CBOR', + ); + }); + + test('copies byte strings', () { + final input = h('43010203'); + final d = CborDecoder(input); + final b = d.bstr(3, 3); + d.done(); + input[1] = 9; + expect(b, [1, 2, 3]); + expect(CborDecoder(h('40')).bstr(0, 0), isEmpty); + }); + + test('keeps a leading U+FEFF of a text', () { + expect(CborDecoder(h('64efbbbf61')).text(4).codeUnits, [0xfeff, 0x61]); + expect(CborDecoder(h('66efbbbfefbbbf')).text(6).codeUnits, [ + 0xfeff, + 0xfeff, + ]); + expect(CborDecoder(h('63ed9fbf')).text(3).codeUnits, [0xd7ff]); + }); + + test( + 'takes the bounds of uint in 0..2^53-1 and of uint64 in 0..2^64-1', + () { + expect(CborDecoder(h('04')).uint(4), 4); + expect(() => CborDecoder(h('00')).uint(-1), throwsArgumentError); + expect( + () => CborDecoder(h('00')).uint(maxSafeUint + 1), + throwsArgumentError, + ); + expect( + () => CborDecoder(h('00')).uint64(BigInt.from(-1)), + throwsArgumentError, + ); + expect( + () => CborDecoder(h('00')).uint64(maxUint64 + BigInt.one), + throwsArgumentError, + ); + }, + ); + }); + + group('integers at 2^53-1, 2^53, 2^63 and 2^64-1', () { + test('are read exactly', () { + final d = CborDecoder( + h( + '1b001fffffffffffff 1b0020000000000000 1b8000000000000000 ' + '1bffffffffffffffff', + ), + ); + expect(d.uint64(), maxSafe); + expect(d.uint64(), twoPow53); + expect(d.uint64(), twoPow63); + expect(d.uint64(), maxUint64); + d.done(); + expect('$maxUint64', '18446744073709551615'); + expect(CborDecoder(h('1b001fffffffffffff')).uint(), maxSafeUint); + // A map key is an int up to 2^53-1 and a BigInt above. + final k = CborDecoder( + h( + 'a4 1b001fffffffffffff 00 1b0020000000000000 00 ' + '1b8000000000000000 00 1bffffffffffffffff 00', + ), + ); + expect(k.map(4), 4); + final keys = []; + for (var i = 0; i < 4; i++) { + keys.add(k.key()); + k.uint(0); + } + k + ..endMap() + ..done(); + expect(keys[0], isA()); + expect(keys[0], maxSafeUint); + expect(keys.skip(1), everyElement(isA())); + expect(keys.skip(1), [twoPow53, twoPow63, maxUint64]); + // Keys above 2^53 compare exactly: 2^53 and 2^53+1 ascend. + expect( + run('a2 1b0020000000000000 00 1b0020000000000001 00', 'm2 k u k u e d'), + isNull, + ); + }); + + test('are printed exactly in errors', () { + final cases = <(String, String, String, String)>[ + ( + 'uint 2^53 above 2^53-1', + '1b0020000000000000', + 'u9007199254740991', + 'offset 9: unsigned integer 9007199254740992 above 9007199254740991', + ), + ( + 'uint 2^63+1 above 2^63', + '1b8000000000000001', + 'u9223372036854775808', + 'offset 9: unsigned integer 9223372036854775809 above 9223372036854775808', + ), + ( + 'uint 2^64-1 above 2^64-2', + '1bffffffffffffffff', + 'u18446744073709551614', + 'offset 9: unsigned integer 18446744073709551615 above 18446744073709551614', + ), + ( + 'keys 2^53+1 then 2^53', + 'a2 1b0020000000000001 00 1b0020000000000000 00', + 'm2 k u k', + 'offset 11: map key 9007199254740992 after key 9007199254740993: keys must be strictly ascending', + ), + ( + 'keys 2^64-1 then 2^64-2', + 'a2 1bffffffffffffffff 00 1bfffffffffffffffe 00', + 'm2 k u k', + 'offset 11: map key 18446744073709551614 after key 18446744073709551615: keys must be strictly ascending', + ), + ( + 'keys 2^63 twice', + 'a2 1b8000000000000000 00 1b8000000000000000 00', + 'm2 k u k', + 'offset 11: map key 9223372036854775808 after key 9223372036854775808: keys must be strictly ascending', + ), + ( + 'keys 2^53-1 twice', + 'a2 1b001fffffffffffff 00 1b001fffffffffffff 00', + 'm2 k u k', + 'offset 11: map key 9007199254740991 after key 9007199254740991: keys must be strictly ascending', + ), + ( + 'bstr of 2^53 bytes', + '5b0020000000000000', + 'b0,9', + 'offset 9: truncated input: a byte string of 9007199254740992 bytes', + ), + ( + 'text of 2^64-1 bytes', + '7bffffffffffffffff', + 't9', + 'offset 9: truncated input: a text string of 18446744073709551615 bytes', + ), + ( + 'array of 2^63 items', + '9b8000000000000000', + 'a100', + 'offset 9: array of 9223372036854775808 items, at most 100', + ), + ( + 'map of 2^53-1 entries', + 'bb001fffffffffffff', + 'm9007199254740991', + 'offset 9: truncated input: map of 9007199254740991 entries', + ), + ( + 'array of 2^53 items', + '9b0020000000000000', + 'a9007199254740991', + 'offset 9: array of 9007199254740992 items, at most 9007199254740991', + ), + ]; + for (final (name, hex, prog, text) in cases) { + expect(run(hex, prog)?.message, 'codec: $text: $nc', reason: name); + } + expect( + messageOf(() => CborDecoder(h('1b0020000000000000')).uint()), + 'codec: offset 9: unsigned integer 9007199254740992 above ' + '9007199254740991: ERR_NON_CANONICAL_CBOR', + ); + }); + + test('are written exactly', () { + expect(encoded((e) => e.uint(maxSafeUint)), '1b001fffffffffffff'); + expect(encoded((e) => e.uint64(twoPow53)), '1b0020000000000000'); + expect(encoded((e) => e.uint64(twoPow63)), '1b8000000000000000'); + expect(encoded((e) => e.uint64(maxUint64)), '1bffffffffffffffff'); + }); + + test('bound the map that peekSchema reads at 2^63-1, as Go', () { + final cases = <(String, String)>[ + ( + 'bb8000000000000000', + 'offset 9: map of 9223372036854775808 entries, at most 9223372036854775807', + ), + ( + 'bbffffffffffffffff', + 'offset 9: map of 18446744073709551615 entries, at most 9223372036854775807', + ), + ( + 'bb7fffffffffffffff', + 'offset 9: truncated input: map of 9223372036854775807 entries', + ), + ( + 'bb0020000000000005', + 'offset 9: truncated input: map of 9007199254740997 entries', + ), + ( + 'a2 1bffffffffffffffff 00 01 01', + 'offset 10: map key 18446744073709551615 where key 0 was expected', + ), + ( + 'a2 00 6161 1b0020000000000000 01', + 'offset 13: map key 9007199254740992 where key 1 was expected', + ), + ]; + for (final (hex, text) in cases) { + expect(messageOf(() => peekSchema(h(hex))), 'codec: $text: $nc'); + } + }); + }); + + group('unmarshalCbor', () { + test('decodes the deterministic encoding of a value', () { + final s = unmarshalSample('a400617801170241010a82011901f4'); + expect(s.type, 'x'); + expect(s.n, BigInt.from(23)); + expect(s.bytes, [1]); + expect(s.list, [BigInt.one, BigInt.from(500)]); + }); + + test('rejects every non-canonical or invalid encoding', () { + final cases = <(String, String)>[ + ( + 'integer not in shortest form', + 'a300617801181702410' + '1', + ), + ( + 'keys out of order', + 'a301170061780241' + '01', + ), + ( + 'duplicate key', + 'a4006178006179011702' + '4101', + ), + ( + 'indefinite-length map', + 'bf00617801170241' + '01ff', + ), + ( + 'indefinite-length byte string', + 'a3006178011702' + '5f4101ff', + ), + ( + 'tag', + 'a3006178011702' + 'c24101', + ), + ( + 'unknown key', + 'a4006178011702410103' + '00', + ), + ( + 'missing key', + 'a2006178011' + '7', + ), + ( + 'trailing byte', + 'a30061780117024101' + '00', + ), + ('invalid UTF-8', 'a30061ff0117024101'), + ( + 'empty optional array present', + 'a400617801170241010a' + '80', + ), + ('wrong type', 'a300617801617a024101'), + ( + 'not a map', + '83006178' + '01', + ), + ('empty input', ''), + // Outside the CBOR profile of spec §58. + ( + 'negative integer', + 'a3006178012002' + '4101', + ), + ( + 'float', + 'a300617801f9400002' + '4101', + ), + ( + 'true', + 'a300617801f502' + '4101', + ), + ( + 'null byte string', + 'a30061780117' + '02f6', + ), + ( + 'undefined byte string', + 'a30061780117' + '02f7', + ), + ( + 'null text', + 'a300f60117' + '024101', + ), + ( + 'text map key', + 'a300617801176162' + '4101', + ), + ]; + for (final (name, hex) in cases) { + expect(codeOf(() => unmarshalSample(hex)), nc, reason: name); + } + expect( + messageOf(() => unmarshalSample('a400617801170241010a80')), + 'codec: input is not the deterministic encoding of its value: ' + 'ERR_NON_CANONICAL_CBOR', + ); + expect( + messageOf(() => unmarshalSample('a3006178011702410100')), + 'codec: offset 9: 1 trailing bytes: ERR_NON_CANONICAL_CBOR', + ); + }); + + test('checks the re-encoding, whatever the decoder and the encoder do', () { + final input = h('a30061780117024101'); + // An error of the schema passes as it is. + final own = DateKeysException(ErrorCode.dateKeyInvalid, 'schema'); + expect( + () => unmarshalCbor(input, (_) => throw own, (_) {}), + throwsA(same(own)), + ); + final other = StateError('not a DateKeys error'); + expect( + () => unmarshalCbor(input, (_) => throw other, (_) {}), + throwsA(same(other)), + ); + // A decoder that stops early leaves bytes behind. + expect( + messageOf(() => unmarshalCbor(input, (d) => d.map(3), (_) {})), + 'codec: offset 1: 1 maps not closed: ERR_NON_CANONICAL_CBOR', + ); + // A decoder that ignores an error of the decoder fails anyway. + void ignore(CborDecoder d) { + try { + d.uint(0); + } on DateKeysException { + // Ignored on purpose: done throws it again. + } + } + + expect( + messageOf(() => unmarshalCbor(input, ignore, (_) {})), + 'codec: offset 0: a map where an unsigned integer was expected: ' + 'ERR_NON_CANONICAL_CBOR', + ); + // An encoder that fails, and one that writes something else. + const notDeterministic = + 'codec: input is not the deterministic encoding of its value: ' + 'ERR_NON_CANONICAL_CBOR'; + final s = Sample(); + expect( + messageOf(() => unmarshalCbor(input, s.decode, (e) => e.map(-1))), + notDeterministic, + ); + expect( + messageOf( + () => unmarshalCbor(input, s.decode, (e) { + s.encode(e); + e.uint(0); + }), + ), + notDeterministic, + ); + }); + + test('wipes the re-encoding, on success and on failure', () { + final input = h('a30061780117024101'); + late CborEncoder used; + final s = Sample(); + unmarshalCbor(input, s.decode, (e) => s.encode(used = e)); + expect(used.out(), hasLength(input.length)); + expect(used.out(), everyElement(0)); + expect( + codeOf( + () => unmarshalCbor(input, s.decode, (e) { + s.encode(used = e); + e.uint(0); + }), + ), + nc, + ); + expect(used.out(), hasLength(input.length + 1)); + expect(used.out(), everyElement(0)); + }); + + test('round-trips random values of the sample schema', () { + final r = Random(2); + for (var i = 0; i < 500; i++) { + final s = Sample() + ..type = String.fromCharCode(0x61 + r.nextInt(26)) + ..n = randomUint64(r) + ..bytes = Uint8List(r.nextInt(40)) + ..list = [for (var j = r.nextInt(4); j > 0; j--) randomUint64(r)]; + final b = Uint8List.fromList(s.marshal()); + final back = Sample(); + unmarshalCbor(b, back.decode, back.encode); + expect(toHex(back.marshal()), toHex(b)); + } + }); + }); + + group('peekSchema and checkSchema', () { + final tag = '6a${hexOf('datekeycap')}'; + + test('read the type tag and the version, and nothing after them', () { + // A future version may use anything after key 1. + final future = ref.marshal({ + 0: 'datekeycap', + 1: 2, + 99: 'new', + 100: const ref.Raw([0xf9, 0x7e, 0x00]), + }); + expect(peekSchema(future), (typeTag: 'datekeycap', version: 2)); + final long = 'a' * maxTypeTagLen; + expect(peekSchema(h('a200 7840 ${hexOf(long)} 0101')).typeTag, long); + expect( + peekSchema(h('a200 $tag 01 1b001fffffffffffff')).version, + maxSafeUint, + ); + }); + + test('reject any other start with the text of the reference', () { + final cases = <(String, String, String)>[ + ('empty', '', 'offset 0: truncated input'), + ('not a map', '8200', 'offset 0: an array where a map was expected'), + ( + 'one entry', + 'a1006161', + 'offset 1: map without a type tag and a schema version', + ), + ( + 'first key not 0', + 'a2016161' + '0201', + 'offset 2: map key 1 where key 0 was expected', + ), + ( + 'second key not 1', + 'a2006161' + '0201', + 'offset 5: map key 2 where key 1 was expected', + ), + ( + 'text key', + 'a2616100' + '0101', + 'offset 1: a text string where an unsigned integer was expected', + ), + ( + 'type tag not text', + 'a2004161' + '0101', + 'offset 2: a byte string where a text string was expected', + ), + ( + 'version not uint', + 'a2006161' + '0120', + 'offset 5: a negative integer (initial byte 0x20) is outside the CBOR profile', + ), + ( + 'version above 2^53-1', + 'a2006161' + '011b0020000000000000', + 'offset 14: unsigned integer 9007199254740992 above 9007199254740991', + ), + ( + 'keys swapped', + 'a2010100' + '6161', + 'offset 2: map key 1 where key 0 was expected', + ), + ( + 'truncated type tag', + 'a2006361', + 'offset 1: truncated input: map of 2 entries', + ), + ( + 'map beyond input', + 'a5006161' + '0101', + 'offset 1: truncated input: map of 5 entries', + ), + ( + 'type tag above maxTypeTagLen', + 'a200 7841 ${'61' * 65} 0101', + 'offset 4: a text string of 65 bytes outside 0..64', + ), + ( + 'type tag not valid UTF-8', + 'a20061ff0101', + 'offset 2: text string is not valid UTF-8', + ), + ( + 'map head not shortest', + 'b802 00 $tag 0102', + 'offset 0: 2 is not in its shortest form (initial byte 0xb8)', + ), + ]; + for (final (name, hex, text) in cases) { + expect( + messageOf(() => peekSchema(h(hex))), + 'codec: $text: $nc', + reason: name, + ); + } + }); + + test('check the type tag first, and only then the version', () { + final b = + (Sample() + ..type = 'datekeycap' + ..n = BigInt.one + ..bytes = Uint8List(0)) + .marshal(); + checkSchema(b, 'datekeycap', 1); + expect( + messageOf(() => checkSchema(b, 'datekeys-control', 1)), + 'codec: type "datekeycap", want "datekeys-control": ' + 'ERR_NON_CANONICAL_CBOR', + ); + expect( + messageOf( + () => checkSchema(h('a200 $tag 0102'), 'datekeys-control', 1), + ), + 'codec: type "datekeycap", want "datekeys-control": ' + 'ERR_NON_CANONICAL_CBOR', + ); + final version = thrown(() => checkSchema(b, 'datekeycap', 2)); + expect(version.code, ErrorCode.unsupportedVersion); + expect( + version.message, + 'codec: datekeycap schema version 1, want 2: ERR_UNSUPPORTED_VERSION', + ); + expect( + messageOf( + () => checkSchema( + h('a200 $tag 01 1b001fffffffffffff'), + 'datekeycap', + 1, + ), + ), + 'codec: datekeycap schema version 9007199254740991, want 1: ' + 'ERR_UNSUPPORTED_VERSION', + ); + // A future version with unknown keys still reports the version. + final future = ref.marshal({0: 'datekeycap', 1: 2, 99: -7}); + expect( + codeOf(() => checkSchema(future, 'datekeycap', 1)), + 'ERR_UNSUPPORTED_VERSION', + ); + for (final hex in ['', 'ff', '8301', 'a10061']) { + expect(codeOf(() => checkSchema(h(hex), 'datekeycap', 1)), nc); + } + }); + + test('read a version only as the second key, after a type tag', () { + // Spec §70: every other form of the version is ERR_NON_CANONICAL_CBOR, + // whatever its value. + const uv = 'ERR_UNSUPPORTED_VERSION'; + final text = hexOf('datekeycap'); + final cases = <(String, String, String)>[ + ('version 2', 'a200 $tag 0102', uv), + ('version 2, rest malformed', 'a300 $tag 0102 02ff', uv), + ('version 2^53-1', 'a200 $tag 011b001fffffffffffff', uv), + ('version 2 not in shortest form', 'a200 $tag 011802', nc), + ('version 2^53', 'a200 $tag 011b0020000000000000', nc), + ('version 2^64-1', 'a200 $tag 011bffffffffffffffff', nc), + ('version before key 0', 'a2 0102 00 $tag', nc), + ('version after key 2', 'a3 00 $tag 0200 0102', nc), + ('map head not in shortest form', 'b802 00 $tag 0102', nc), + ('type tag head not in shortest form', 'a200 780a $text 0102', nc), + ('version missing', 'a100 $tag', nc), + ('version null', 'a200 $tag 01f6', nc), + ('version undefined', 'a200 $tag 01f7', nc), + ('version true', 'a200 $tag 01f5', nc), + ('version false', 'a200 $tag 01f4', nc), + ]; + for (final (name, hex, want) in cases) { + expect( + codeOf(() => checkSchema(h(hex), 'datekeycap', 1)), + want, + reason: name, + ); + } + }); + + test('quote a type tag as Go %q, with the tables of Go 1.26', () { + final cases = <(String, String)>[ + ('67 64617465e280a8', '"date${bs}u2028"'), + ('65 efbbbf2261', '"${bs}ufeff$bs"a"'), + ('62 0a7f', '"${bs}n${bs}x7f"'), + ]; + for (final (tagHex, quoted) in cases) { + expect( + messageOf(() => checkSchema(h('a200 $tagHex 0101'), 'datekeycap', 1)), + 'codec: type $quoted, want "datekeycap": ERR_NON_CANONICAL_CBOR', + ); + } + }); + }); + + group('walkCbor', () { + test('bounds depth and length as the reference', () { + final cases = <(String, String, int, int, String?)>[ + ('uint', '17', 0, 0, null), + ('empty bstr', '40', 0, 0, null), + ('text', '626161', 0, 2, null), + ( + 'text above max', + '626161', + 0, + 1, + 'offset 1: a text string of 2 bytes outside 0..1', + ), + ( + 'bstr above max', + '420000', + 0, + 1, + 'offset 1: a byte string of 2 bytes outside 0..1', + ), + ('map two sorted keys', 'a200010101', 1, 2, null), + ( + 'map at depth 0', + 'a0', + 0, + 0, + 'offset 0: containers nested deeper than 0', + ), + ( + 'map above max', + 'a200010101', + 1, + 1, + 'offset 1: map of 2 entries, at most 1', + ), + ( + 'array above max', + '83010203', + 1, + 2, + 'offset 1: array of 3 items, at most 2', + ), + ('four levels', 'a1008181a10040', 4, 1, null), + ( + 'four levels, depth 3', + 'a1008181a10040', + 3, + 1, + 'offset 4: containers nested deeper than 3', + ), + ('empty containers', '82a080', 2, 2, null), + ( + 'keys out of order', + 'a201000001', + 1, + 2, + 'offset 3: map key 0 after key 1: keys must be strictly ascending', + ), + ( + 'text key', + 'a1616100', + 1, + 1, + 'offset 1: a text string where an unsigned integer was expected', + ), + ( + 'float inside', + '8201f97e00', + 1, + 2, + 'offset 2: a float or simple value (initial byte 0xf9) is outside the CBOR profile', + ), + ( + 'truncated map', + 'a20001', + 1, + 2, + 'offset 1: truncated input: map of 2 entries', + ), + ( + 'truncated array', + '8201', + 1, + 2, + 'offset 1: truncated input: array of 2 items', + ), + ('trailing byte', '8000', 1, 0, 'offset 1: 1 trailing bytes'), + ('empty input', '', 1, 1, 'offset 0: truncated input'), + ( + 'invalid UTF-8 inside', + 'a10061ff', + 1, + 1, + 'offset 2: text string is not valid UTF-8', + ), + ( + 'null', + 'f6', + 1, + 1, + 'offset 0: a float or simple value (initial byte 0xf6) is outside the CBOR profile', + ), + ('2^64-1 inside', 'a1001bffffffffffffffff', 1, 1, null), + ]; + for (final (name, hex, maxDepth, maxLen, text) in cases) { + void walk() => walkCbor(h(hex), maxDepth, maxLen); + if (text == null) { + expect(walk, returnsNormally, reason: name); + } else { + expect(messageOf(walk), 'codec: $text: $nc', reason: name); + } + } + }); + + test('reads deep input iteratively', () { + const n = 1 << 20; + final deep = Uint8List(n + 1)..fillRange(0, n, 0x81); + walkCbor(deep, n, 1); + expect( + messageOf(() => walkCbor(deep, n - 1, 1)), + 'codec: offset 1048575: containers nested deeper than 1048575: ' + 'ERR_NON_CANONICAL_CBOR', + ); + }); + }); + + // The property tests and the fuzz targets of the Go tests, over seeded + // random values and seeded mutations of the seeds of the fuzz targets. + group('properties', () { + test('the encoder writes what the reference writes, and walk accepts it ' + 'within its exact shape and rejects it one level or one byte ' + 'tighter', () { + final r = Random(1); + for (var i = 0; i < 1000; i++) { + final v = randomValue(r, 0); + final e = CborEncoder(); + encodeValue(e, v); + final b = e.out(); + expect(toHex(b), toHex(ref.marshal(v))); + final (depth, length) = ref.shape(v); + walkCbor(b, depth, length); + if (depth > 0) { + expect(codeOf(() => walkCbor(b, depth - 1, length)), nc); + } + if (length > 0) { + expect(codeOf(() => walkCbor(b, depth, length - 1)), nc); + } + } + }); + + test( + 'walk accepts exactly the items of the profile that fit its bounds', + () { + final r = Random(3); + final seeds = fuzzSeeds(r); + for (var i = 0; i < 4000; i++) { + final input = mutate(r, seeds[r.nextInt(seeds.length)]); + final code = codeOf(() => walkCbor(input, 8, 64)); + expect(code, anyOf('', nc)); + final Object? v; + try { + v = ref.unmarshal(input); + } on FormatException { + expect( + code, + nc, + reason: '${toHex(input)}: the reference rejects it', + ); + continue; + } + final (depth, length) = ref.shape(v); + expect( + code == '', + depth <= 8 && length <= 64, + reason: '${toHex(input)} of depth $depth and length $length', + ); + } + }, + ); + + test('peekSchema reads what the reference reads', () { + final r = Random(4); + final seeds = fuzzSeeds(r); + for (var i = 0; i < 4000; i++) { + final input = mutate(r, seeds[r.nextInt(seeds.length)]); + ({String typeTag, int version})? got; + try { + got = peekSchema(input); + } on DateKeysException catch (e) { + expect(e.code, ErrorCode.nonCanonicalCbor); + } + final Object? m; + try { + m = ref.unmarshal(input); + } on FormatException { + continue; + } + if (m is! Map) continue; + // Keys ascend, so keys 0 and 1, when present, come first. + final tag = m[BigInt.zero]; + final version = m[BigInt.one]; + final want = + tag is ref.Text && + tag.bytes.length <= maxTypeTagLen && + version is BigInt && + version <= maxSafe; + expect(got != null, want, reason: toHex(input)); + if (got != null && tag is ref.Text && version is BigInt) { + expect(utf8.encode(got.typeTag), tag.bytes); + expect(BigInt.from(got.version), version); + } + } + }); + + test( + 'the decoder keeps its bounds and its first error, of the profile', + () { + final r = Random(5); + final seeds = fuzzSeeds(r); + for (var i = 0; i < 4000; i++) { + final input = mutate(r, seeds[r.nextInt(seeds.length)]); + final prog = [for (var j = r.nextInt(16); j > 0; j--) r.nextInt(256)]; + final d = CborDecoder(input); + DateKeysException? first; + for (var p = 0; p < prog.length; p++) { + final bound = p + 1 < prog.length ? prog[p + 1] : 0; + DateKeysException? err; + try { + switch (prog[p] % 8) { + case 0: + expect(d.map(bound), lessThanOrEqualTo(bound)); + p++; + case 1: + d.key(); + case 2: + d.endMap(); + case 3: + expect(d.array(bound), lessThanOrEqualTo(bound)); + p++; + case 4: + expect(d.uint(bound), lessThanOrEqualTo(bound)); + p++; + case 5: + final lo = bound % 16; + expect( + d.bstr(lo, bound).length, + allOf(greaterThanOrEqualTo(lo), lessThanOrEqualTo(bound)), + ); + p++; + case 6: + expect( + utf8.encode(d.text(bound)).length, + lessThanOrEqualTo(bound), + ); + p++; + case 7: + d.done(); + } + } on DateKeysException catch (e) { + err = e; + expect(e.code, ErrorCode.nonCanonicalCbor); + } + if (first == null) { + first = err; + } else { + expect(identical(err, first), isTrue, reason: 'not sticky'); + } + } + } + }, + ); + + test( + 'unmarshalCbor accepts only deterministic encodings of the profile', + () { + final r = Random(6); + final seeds = [ + ...fuzzSeeds(r), + for (var i = 0; i < 50; i++) + (Sample() + ..type = 'x' + ..n = randomUint64(r) + ..bytes = Uint8List(r.nextInt(8)) + ..list = [ + for (var j = r.nextInt(3); j > 0; j--) randomUint64(r), + ]) + .marshal(), + ]; + var accepted = 0; + for (var i = 0; i < 4000; i++) { + final input = mutate(r, seeds[r.nextInt(seeds.length)]); + final s = Sample(); + try { + unmarshalCbor(input, s.decode, s.encode); + } on DateKeysException catch (e) { + expect(e.code, ErrorCode.nonCanonicalCbor); + continue; + } + accepted++; + expect(toHex(s.marshal()), toHex(input)); + walkCbor(input, 2, 64); + } + expect(accepted, greaterThan(0)); + }, + ); + }); +} diff --git a/test/cbor_vectors_test.dart b/test/cbor_vectors_test.dart new file mode 100644 index 0000000..8bf0077 --- /dev/null +++ b/test/cbor_vectors_test.dart @@ -0,0 +1,120 @@ +// The shared vectors of the CBOR profile of spec §58, +// testdata/vectors/cbor.json, as vectors_test.go of package codec of +// datekeys-go: walkCbor accepts exactly the accept list, with the value +// recorded for unsigned integers, and rejects the reject list with the +// recorded code. +@TestOn('vm') +library; + +import 'dart:convert'; +import 'dart:io'; + +import 'package:datekeys/datekeys.dart'; +import 'package:test/test.dart'; + +void main() { + final file = jsonDecode( + File('testdata/vectors/cbor.json').readAsStringSync(), + ) as Map; + final walk = file['walk']! as Map; + final maxDepth = walk['max_depth']! as int; + final maxLen = walk['max_len']! as int; + List> list(String key) => + (file[key]! as List).cast>(); + final accept = list('accept'); + final reject = list('reject'); + final schemas = list('schemas'); + + test('the vector file is complete', () { + expect(file['spec'], specVersion); + expect(accept, isNotEmpty); + expect(reject, isNotEmpty); + expect(maxDepth, isPositive); + expect(maxLen, isPositive); + }); + + group('accept', () { + for (final v in accept) { + test(v['name'], () { + final input = fromHex(v['hex']! as String); + walkCbor(input, maxDepth, maxLen); + // An unsigned integer records its value: a JSON number up to 2^53-1 + // and a decimal string above, so that no reader loses precision. + final isUint = input.isNotEmpty && input[0] >> 5 == 0; + expect(v.containsKey('value'), isUint); + if (!isUint) return; + final n = CborDecoder(input).uint64(); + if (n > BigInt.from(maxSafeUint)) { + expect(v['value'], '$n'); + } else { + expect(v['value'], n.toInt()); + } + }); + } + }); + + group('reject', () { + for (final v in reject) { + test(v['name'], () { + final input = fromHex(v['hex']! as String); + var code = ''; + try { + walkCbor(input, maxDepth, maxLen); + } on DateKeysException catch (e) { + code = e.code.code; + } + expect(code, v['error']); + }); + } + }); + + // The schemas block is decoded with the decoder of each schema: Provider + // Profile, PUBLIC_HEADER, CONTROL_CBOR and the body of a .dkk, which arrive + // with stage 4 of docs/PLAN_dart.md. Until then the block is only read: its + // shape is checked, and the objects it holds as valid are items of the + // profile whose type tag peekSchema reads. + group('schemas', () { + const typeTags = { + 'provider_profile': 'datekeys-provider-profile', + 'public_header': 'datekeycap', + 'control_cbor': 'datekeys-control', + 'dkk_body': 'datekeys-access-key', + }; + final codes = {for (final c in ErrorCode.values) c.code}; + + test('the block is well formed', () { + expect(schemas, isNotEmpty); + for (final v in schemas) { + expect(typeTags.keys, contains(v['schema']), reason: '${v['name']}'); + expect(v['block'], isA()); + expect(v['name'], isA()); + expect(fromHex(v['hex']! as String), isNotEmpty); + expect( + v['result'] == 'ok' || codes.contains(v['result']), + isTrue, + reason: '${v['name']}: ${v['result']}', + ); + } + }); + + test('the valid objects are items of the profile with their type tag', () { + final valid = schemas.where((v) => v['result'] == 'ok').toList(); + expect(valid, isNotEmpty); + for (final v in valid) { + final input = fromHex(v['hex']! as String); + walkCbor(input, 16, input.length); + expect( + peekSchema(input).typeTag, + typeTags[v['schema']], + reason: '${v['name']}', + ); + } + }); + + test( + 'the vectors decode with the decoders of their schemas', + () {}, + skip: 'the decoders of the schemas arrive with stage 4', + ); + }); +}