You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
dateKeys-dart/tool/wordkey_go_vectors.go

442 lines
18 KiB

//go:build ignore
// Writes test/vectors/wordkey_vectors.json and the same JSON as a Dart
// constant, wordkey_vectors.g.dart: the results of package wordkey of the Go
// reference (spec §38.1) for lib/src/wordkey.dart. Every expected value is
// computed here by the Go reference; none is written by hand.
//
// - normalize: texts and the words of wordkey.Normalize: the cases of the
// tests of wordkey and of wordkey.test.ts of datekeys-ts, the examples of
// §38.1, every white space of the list of §38.1 and the look-alikes that
// are not, marks inside and outside U+0300 to U+036F, the lower case of
// Unicode 18.0.0, bytes that are not valid UTF-8, and texts drawn from a
// fixed seed.
// - check: lists of words and the result of wordkey.Check: those of the
// tests, edges of the count, every kind of refused code point, and the
// words of the texts of normalize.
// - keys: words, a chain hash, a round and a capsule_id; the salt S of
// §38.1, and the password P, the words joined by one U+0020, both
// checked against wordkey.Key; wordkey.Key, 600 000 iterations of
// PBKDF2-HMAC-SHA256; the same PBKDF2 with 1000 iterations, for the tests
// compiled to JavaScript; and the recipient of wordkey.Identity. The
// first is the vector of spec §38.1.
//
// Binary values are lower-case hexadecimal, and the JSON is ASCII, so that
// the Dart constant is too. The seed is fixed: every run writes the same
// bytes. It needs the module context of datekeys-go, whose wordkey and
// profile packages it imports, and changes nothing there:
//
// cd ../datekeys-go && go run ../datekeys-dart/tool/wordkey_go_vectors.go -out ../datekeys-dart/test/vectors
package main
import (
"bytes"
"crypto/pbkdf2"
"crypto/sha256"
"encoding/hex"
"encoding/json"
"flag"
"fmt"
"log"
"math/rand/v2"
"os"
"path/filepath"
"runtime"
"strings"
"g.activething.com/go/DateKeys/profile"
"g.activething.com/go/DateKeys/wordkey"
)
var rng = rand.New(rand.NewChaCha8(sha256.Sum256([]byte("datekeys-dart stage 4a: wordkey"))))
func h(s string) string { return hex.EncodeToString([]byte(s)) }
func hexAll(ss []string) []string {
out := make([]string, len(ss))
for i, s := range ss {
out[i] = h(s)
}
return out
}
func pick[T any](xs []T) T { return xs[rng.IntN(len(xs))] }
func chance(p float64) bool { return rng.Float64() < p }
// The white space of §38.1, Go's unicode.IsSpace, and code points that look
// like it and are not.
var (
spaces = []string{"\t", "\n", "\v", "\f", "\r", " ", "\u0085", "\u00a0", "\u1680", "\u2000", "\u2001", "\u2002", "\u2003", "\u2004", "\u2005", "\u2006", "\u2007", "\u2008", "\u2009", "\u200a", "\u2028", "\u2029", "\u202f", "\u205f", "\u3000"}
notSpaces = []string{"\u200b", "\u180e", "\u2060", "\ufeff", "\u00ad", "_", "-", "\u3164", "\u115f"}
vocabulary = []string{
"perro", "luna", "casa", "verde", "tren", "mar", "\u00e1rbol", "\u00c1RBOL", "ni\u00f1o", "NI\u00d1O", "canci\u00f3n", "\u00c1baco",
"\u00d1and\u00fa", "\u00fcber", "STRA\u00dfE", "stra\u00dfe", "\u03a3\u03bf\u03c6\u03af\u03b1", "\u03a3\u0391\u03a3", "\u1f40\u03b4\u03c5\u03c3\u03c3\u03b5\u03cd\u03c2", "\u041c\u043e\u0441\u043a\u0432\u0430", "\u041c\u041e\u0421\u041a\u0412\u0410", "\ud55c\uad6d\uc5b4", "\u65e5\u672c\u8a9e",
"\U0001F600", "x", "de", "la", "al", "a\u00f1o", "caf\u00e9", "CAF\u00c9", "na\u00efve", "r\u00e9sum\u00e9", "\u0130stanbul", "\u01c5emal",
"\uab70\uab71\uab72", "\u13a0\u13a1\u13a2", "\uff26\uff35\uff2c\uff2c", "\uff46\uff55\uff4c\uff4c", "e\u0301", "a\u0323\u0301", "a\u0301\u0323", "\u212aelvin",
"\u1e9e", "\ua7da", "\ua7cb", "\ufb00", "perro,", "\u00a1hola!", "\u0250", "\u023a", "\ua78c", "\u0531", "\u0561", "\u10a0", "\U00010400", "\U0001E900",
"\u0483x", "x\u1ab0", "x\u20d0", "x\ufe20", "x\u0345", "\u0e01\u0e34", "\u05d0\u05b0",
}
noise = []string{"\x01", "\u009f", "\u0378", "\ufffe", "\U000E0001", "\ufe0f", "\xff", "\xed\xa0\x80", "\xc0\x80", "\ue000", "\ufffd", "\U0010fffd"}
)
func randCase(s string) string {
switch rng.IntN(4) {
case 0:
return strings.ToUpper(s)
case 1:
return strings.ToLower(s)
}
return s
}
// text is words of the vocabulary, sometimes with noise, between runs of
// white space, with white space or not at the ends.
func text() string {
var b strings.Builder
if chance(0.3) {
b.WriteString(pick(spaces))
}
for n := rng.IntN(10); n > 0; n-- {
w := randCase(pick(vocabulary))
if chance(0.08) {
i := rng.IntN(len(w) + 1)
w = w[:i] + pick(append(noise, notSpaces...)) + w[i:]
}
b.WriteString(w)
for k := 1 + rng.IntN(3)*rng.IntN(2); k > 0; k-- {
b.WriteString(pick(spaces))
}
}
return b.String()
}
type normalizeCase struct {
Name string `json:"name,omitempty"`
In string `json:"in"`
Words []string `json:"words"`
}
func normalizeOf(name, text string) normalizeCase {
words := wordkey.Normalize(text)
if words == nil {
words = []string{}
}
return normalizeCase{name, h(text), hexAll(words)}
}
type checkCase struct {
Name string `json:"name,omitempty"`
Words []string `json:"words"`
Result string `json:"result"`
}
func checkOf(name string, words []string) checkCase {
result := "ok"
if err := wordkey.Check(words); err != nil {
result = err.Error()
}
if words == nil {
words = []string{}
}
return checkCase{name, hexAll(words), result}
}
type keyCase struct {
Name string `json:"name"`
Words []string `json:"words"`
ChainHash string `json:"chain_hash"`
Round uint64 `json:"round"`
CapsuleID string `json:"capsule_id"`
Password string `json:"password"`
Salt string `json:"salt"`
Key string `json:"key"`
Key1000 string `json:"key_1000"`
Recipient string `json:"recipient"`
}
func keyOf(name string, words []string, chain []byte, round uint64, capsuleID []byte) keyCase {
key, err := wordkey.Key(words, chain, round, capsuleID)
if err != nil {
log.Fatal(err)
}
// The salt of §38.1, as wordkey.Key writes it: the same key from it
// proves that it is the one Key uses.
salt := fmt.Sprintf("DateKeys llave de palabras v2|%x|%d|%x", chain, round, capsuleID)
again, err := pbkdf2.Key(sha256.New, strings.Join(words, " "), []byte(salt), wordkey.Rounds, 32)
if err != nil || !bytes.Equal(again, key) {
log.Fatalf("%s: the salt is not the one of wordkey.Key", name)
}
short, err := pbkdf2.Key(sha256.New, strings.Join(words, " "), []byte(salt), 1000, 32)
if err != nil {
log.Fatal(err)
}
id, err := wordkey.Identity(words, chain, round, capsuleID)
if err != nil {
log.Fatal(err)
}
return keyCase{name, hexAll(words), hex.EncodeToString(chain), round, hex.EncodeToString(capsuleID), h(strings.Join(words, " ")), salt,
hex.EncodeToString(key), hex.EncodeToString(short), id.Recipient().String()}
}
func normalizeCases() []normalizeCase {
named := [][2]string{
{"TestNormalize of datekeys-go", " \u00c1baco\u00a0\u00c1RBOL\tni\u00f1o \u03a3\u0391\u03a3 \u0130 "},
{"spaces only", " "},
{"empty", ""},
{"U+A7CB, which lowers to U+0264 since Unicode 16.0", "\ua7cb"},
{"U+FEFF is not white space", "a\ufeffb"},
{"the six words of TestCheck", "Perro LUNA casa verde tr\u00e9n mar"},
{"the six words of the vector of \u00a738.1", "perro luna casa verde tren mar"},
{"\u00a738.1: \u00c1baco \u00c1RBOL", "\u00c1baco \u00c1RBOL"},
{"\u00a738.1: abaco arbol", "abaco arbol"},
{"\u00a738.1: \u00e1baco \u00e1rbol", "\u00e1baco \u00e1rbol"},
{"\u00a738.1: punctuation counts", "perro, perro"},
{"the Kelvin sign", "\u212a"},
{"the capital sharp s", "\u1e9e"},
{"a Cherokee capital lowers to the small letter", "\u13a0"},
{"a Cherokee small letter stays", "\uab70"},
{"DZ with caron, titlecase", "\u01c5"},
{"a ligature has no lower case", "\ufb00"},
{"Hangul decomposes", "\ud55c\uad6d\uc5b4"},
{"full-width capitals", "\uff26\uff35\uff2c\uff2c"},
{"U+0345 is in U+0300 to U+036F", "x\u0345y"},
{"marks outside U+0300 to U+036F stay", "\u0483x\u1ab0y\u20d0z\ufe20"},
{"marks out of canonical order", "a\u0323\u0301 a\u0301\u0323"},
{"a mark first", "\u0301a"},
{"a mark alone", "\u0301"},
{"a dotless i and a dotted capital I", "\u0131 \u0130"},
{"final sigma", "\u03c2 \u03a3 \u03c3"},
{"an astral capital", "\U00010400\U0001E900"},
{"bytes that are not valid UTF-8", "a\xffb \xc0\x80 \xed\xa0\x80"},
{"a truncated sequence at the end", "palabra\xe2\x82"},
{"NEL and NBSP", "uno\u0085dos\u00a0tres"},
{"U+180E is not white space", "uno\u180edos"},
{"U+200B is not white space", "uno\u200bdos"},
{"the ideographic space", "uno\u3000dos"},
{"line and paragraph separators", "uno\u2028dos\u2029tres"},
{"control characters that are not white space", "uno\x01dos\x7fres"},
{"the first and the last mark of U+0300 to U+036F", "a\u0300b\u036fc"},
{"U+02FF and U+0370, around U+0300 to U+036F", "a\u02ffb\u0370c"},
{"every mark of U+0300 to U+036F", "x\u0300\u0301\u0302\u0303\u0304\u0305\u0306\u0307\u0308\u0309\u030a\u030b\u030c\u030d\u030e\u030f\u0310\u0311\u0312\u0313\u0314\u0315\u0316\u0317\u0318\u0319\u031a\u031b\u031c\u031d\u031e\u031f\u0320\u0321\u0322\u0323\u0324\u0325\u0326\u0327\u0328\u0329\u032a\u032b\u032c\u032d\u032e\u032f\u0330\u0331\u0332\u0333\u0334\u0335\u0336\u0337\u0338\u0339\u033a\u033b\u033c\u033d\u033e\u033f\u0340\u0341\u0342\u0343\u0344\u0345\u0346\u0347\u0348\u0349\u034a\u034b\u034c\u034d\u034e\u034f\u0350\u0351\u0352\u0353\u0354\u0355\u0356\u0357\u0358\u0359\u035a\u035b\u035c\u035d\u035e\u035f\u0360\u0361\u0362\u0363\u0364\u0365\u0366\u0367\u0368\u0369\u036a\u036b\u036c\u036d\u036e\u036fy"},
}
var out []normalizeCase
for _, n := range named {
out = append(out, normalizeOf(n[0], n[1]))
}
for i, s := range spaces {
out = append(out, normalizeOf(fmt.Sprintf("white space %d of \u00a738.1", i+1), "a"+s+"b"))
}
for i, s := range notSpaces {
out = append(out, normalizeOf(fmt.Sprintf("not white space %d", i+1), "a"+s+"b"))
}
seen := map[string]bool{}
for len(out) < 400 {
t := text()
if !seen[t] {
seen[t] = true
out = append(out, normalizeOf("", t))
}
}
return out
}
func checkCases(texts []normalizeCase) []checkCase {
six := "perro luna casa verde tren mar"
named := []struct {
name string
words []string
}{
{"the six words of TestCheck", wordkey.Normalize("Perro LUNA casa verde tr\u00e9n mar")},
{"three words", wordkey.Normalize("uno dos tres")},
{"short words only", wordkey.Normalize("a b c d e f g h")},
{"the same word six times", wordkey.Normalize("perro perro perro perro perro perro")},
{"four words of three letters or more", wordkey.Normalize("de la casa al mar en tren verde")},
{"ZWSP", wordkey.Normalize(six + "\u200b")},
{"U+0001", wordkey.Normalize(six + "\x01")},
{"U+009F", wordkey.Normalize(six + "\u009f")},
{"U+0378", wordkey.Normalize(six + "\u0378")},
{"no words", nil},
{"five different words", wordkey.Normalize("perro luna casa verde tren")},
{"seven different words", wordkey.Normalize(six + " sol")},
{"six words, one of two letters", wordkey.Normalize("perro luna casa verde tren ma")},
{"three letters of three bytes or two", []string{"\u00f1a\u00f1", "a\u00f1\u00e1", "abc", "abd", "abe", "abf"}},
{"astral letters count once", []string{"\U00010400\U00010401\U00010402", "bbb", "ccc", "ddd", "eee", "fff"}},
{"an astral word of two runes does not count", []string{"\U00010400\U00010401", "bbb", "ccc", "ddd", "eee", "fff"}},
{"U+FFFD is assigned", []string{"\ufffd\ufffd\ufffd", "bbb", "ccc", "ddd", "eee", "fff"}},
{"bytes that are not valid UTF-8 count one each", []string{"\xff\xfe\xfd", "bbb", "ccc", "ddd", "eee", "fff"}},
{"a surrogate in UTF-8 bytes is three", []string{"\xed\xa0\x80", "bbb", "ccc", "ddd", "eee", "fff"}},
{"two bytes that are not valid UTF-8", []string{"\xff\xfe", "bbb", "ccc", "ddd", "eee", "fff"}},
{"words that differ in their bytes only", []string{"\xffab", "\xfeab", "ccc", "ddd", "eee", "fff"}},
{"private use is visible to the rules", []string{"\ue000\ue001\ue002", "bbb", "ccc", "ddd", "eee", "fff"}},
{"a noncharacter is unassigned", []string{"ab\ufffe", "bbb", "ccc", "ddd", "eee", "fff"}},
{"VS16 is invisible", []string{"\u2764\ufe0f", "bbb", "ccc", "ddd", "eee", "fff"}},
{"a tag is invisible", []string{"abc\U000E0041"}},
{"TAB inside a word given as it is", []string{"a\tb"}},
{"NEL inside a word given as it is", []string{"a\u0085b"}},
{"a control after a short word", []string{"ab", "c\x00d"}},
{"the first refusal decides", []string{"a\u0378", "b\x01"}},
{"refusals before the count", []string{"aaa", "bbb", "c\u00ad"}},
{"empty words", []string{"", "", "aaa", "bbb", "ccc", "ddd", "eee", "fff"}},
{"a word of three spaces given as it is", []string{" ", "bbb", "ccc", "ddd", "eee", "fff"}},
// The edges of the controls of unicode.IsControl, C0 and C1.
{"U+001F is a control", []string{"a\x1fb", "bbb", "ccc", "ddd", "eee", "fff"}},
{"U+007E is not", []string{"a~b", "bbb", "ccc", "ddd", "eee", "fff"}},
{"U+007F is a control", []string{"a\x7fb", "bbb", "ccc", "ddd", "eee", "fff"}},
{"U+0080 is a control", []string{"a\u0080b", "bbb", "ccc", "ddd", "eee", "fff"}},
{"U+00A0 inside a word given as it is is not", []string{"a\u00a0b", "bbb", "ccc", "ddd", "eee", "fff"}},
}
var out []checkCase
for _, n := range named {
out = append(out, checkOf(n.name, n.words))
}
for _, t := range texts {
if t.Name != "" {
continue
}
words := make([]string, len(t.Words))
for i, w := range t.Words {
b, _ := hex.DecodeString(w)
words[i] = string(b)
}
out = append(out, checkOf("", words))
}
// Lists of words drawn from the seed, with repetitions and short words.
for k := 0; k < 150; k++ {
var words []string
for n := rng.IntN(10); n > 0; n-- {
w := pick(vocabulary)
if chance(0.3) {
w = pick([]string{"perro", "luna", "casa", "verde", "tren", "mar", "sol", "pan"})
}
if chance(0.05) {
w += pick(noise)
}
words = append(words, w)
}
out = append(out, checkOf("", words))
}
return out
}
func keyCases() []keyCase {
quicknet, err := hex.DecodeString(profile.QuicknetChainHash)
if err != nil {
log.Fatal(err)
}
seq, _ := hex.DecodeString("000102030405060708090a0b0c0d0e0f")
ff := bytes.Repeat([]byte{0xff}, 16)
out := []keyCase{
keyOf("the vector of spec \u00a738.1", wordkey.Normalize("perro luna casa verde tren mar"), quicknet, 1000, seq),
keyOf("normalized words of other scripts, the next round", wordkey.Normalize(" \u00c1baco \u00c1RBOL ni\u00f1o \u03a3\u0391\u03a3 \u0130stanbul \ud55c\uad6d\uc5b4 \U0001F600 "), quicknet, 1001, seq),
keyOf("the largest round of 53 bits, another chain hash and capsule_id", []string{"perro", "luna", "casa", "verde", "tren", "mar"}, bytes.Repeat([]byte{0xab}, 32), 1<<53-1, ff),
keyOf("a word of a surrogate in UTF-8 bytes, round 0, no capsule_id", []string{"\xed\xa0\x80", "luna", "casa", "verde", "tren", "mar"}, quicknet, 0, nil),
}
if out[0].Key != "fceec4d8ca8de86c85a1f26ed49f82a2b38431bd0ce36db995ae7dfd49b96e41" {
log.Fatalf("the vector of spec \u00a738.1 is %s", out[0].Key)
}
return out
}
// ascii escapes every code point above U+007F of JSON as \uXXXX, in UTF-16
// for those above U+FFFF: the same JSON, in ASCII.
func ascii(b []byte) []byte {
var out bytes.Buffer
for _, r := range string(b) {
switch {
case r < 0x80:
out.WriteByte(byte(r))
case r < 0x10000:
fmt.Fprintf(&out, `\u%04x`, r)
default:
r -= 0x10000
fmt.Fprintf(&out, `\u%04x\u%04x`, 0xd800+(r>>10), 0xdc00+(r&0x3ff))
}
}
return out.Bytes()
}
func marshal(v any) []byte {
var b bytes.Buffer
enc := json.NewEncoder(&b)
enc.SetEscapeHTML(false)
if err := enc.Encode(v); err != nil {
log.Fatal(err)
}
return ascii(bytes.TrimSuffix(b.Bytes(), []byte("\n")))
}
// list writes the items of a JSON array one per line.
func list[T any](w *bytes.Buffer, key string, items []T, last bool) {
fmt.Fprintf(w, " %q: [\n", key)
for i, it := range items {
w.WriteString(" ")
w.Write(marshal(it))
if i < len(items)-1 {
w.WriteByte(',')
}
w.WriteByte('\n')
}
w.WriteString(" ]")
if !last {
w.WriteByte(',')
}
w.WriteByte('\n')
}
func main() {
outDir := flag.String("out", "", "the directory test/vectors of datekeys-dart")
flag.Parse()
if *outDir == "" {
log.Fatal("usage: go run wordkey_go_vectors.go -out DIR")
}
normalize := normalizeCases()
check := checkCases(normalize)
keys := keyCases()
var w bytes.Buffer
w.WriteString("{\n")
head := []struct {
key string
v any
}{
{"description", "The results of package wordkey of the Go reference (spec \u00a738.1) for lib/src/wordkey.dart; see tool/wordkey_go_vectors.go. " +
"Binary values are lower-case hex. normalize: a text and the words of wordkey.Normalize. check: words and the result of wordkey.Check, ok or the text of the error. " +
"keys: words, chain_hash, round and capsule_id; password and salt, P and S of \u00a738.1; key, wordkey.Key; key_1000, the same PBKDF2-HMAC-SHA256 with 1000 iterations; recipient, that of wordkey.Identity. The first key is the vector of \u00a738.1."},
{"generator", "tool/wordkey_go_vectors.go, " + runtime.Version()},
{"rounds", wordkey.Rounds},
{"min_words", wordkey.MinWords},
{"min_letters", wordkey.MinLetters},
}
for _, kv := range head {
fmt.Fprintf(&w, " %q: %s,\n", kv.key, marshal(kv.v))
}
list(&w, "normalize", normalize, false)
list(&w, "check", check, false)
list(&w, "keys", keys, true)
w.WriteString("}\n")
var parsed any
if err := json.Unmarshal(w.Bytes(), &parsed); err != nil {
log.Fatalf("the JSON does not parse: %v", err)
}
if bytes.Contains(w.Bytes(), []byte("'''")) {
log.Fatal("the JSON holds three quotes")
}
path := filepath.Join(*outDir, "wordkey_vectors.json")
if err := os.WriteFile(path, w.Bytes(), 0o644); err != nil {
log.Fatal(err)
}
fmt.Printf("wrote %s, %d bytes: %d texts, %d lists of words, %d keys\n", path, w.Len(), len(normalize), len(check), len(keys))
dart := "// Generated by tool/wordkey_go_vectors.go from wordkey_vectors.json, for\n" +
"// the tests that also run compiled to JavaScript, where no file can be\n" +
"// read. Do not edit.\n\n" +
"/// The text of test/vectors/wordkey_vectors.json.\n" +
"const wordkeyVectorsJson = r'''\n" + w.String() + "''';\n"
dpath := filepath.Join(*outDir, "wordkey_vectors.g.dart")
if err := os.WriteFile(dpath, []byte(dart), 0o644); err != nil {
log.Fatal(err)
}
fmt.Printf("wrote %s\n", dpath)
}

Powered by TurnKey Linux.