You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
dateKeys-dart/tool/wordlist_go_vectors.go

1067 lines
42 KiB

This file contains ambiguous Unicode characters!

This file contains ambiguous Unicode characters that may be confused with others in your current locale. If your use case is intentional and legitimate, you can safely ignore this warning. Use the Escape button to highlight these characters.

//go:build ignore
// Writes test/vectors/wordlist_vectors.json and the same JSON as a Dart
// constant, wordlist_vectors.g.dart: the results of the word lists of
// package wordkey of the Go reference (generate.go: List, CheckList,
// Generate and Bits, the random words of spec §38.1; and dice.go:
// DiceNumber, DiceWord, DiceWords and DiceList, the words of dice) for
// lib/src/wordlist.dart. Every expected value is computed here by the Go
// reference; none is written by hand.
//
// - lists: the built-in lists, with the SHA-256, the size and the number
// of words of each file of wordkey/lists, which List reads.
// - parse: the body of List, restated because List reads only its
// built-in lists, on texts of the base list of TestCheckList, "pal" and
// three letters, joined by a separator, with a prefix, a suffix and
// edits: with and without the final LF, CR LF lines, an empty last
// line, a byte order mark, an empty text, bytes that are not valid
// UTF-8. The restatement gives the words of List on every built-in list.
// - check_list: CheckList on the base list of 0 to 17 576 words, edited:
// the cases of TestCheckList; languages without an alphabet; white
// space, controls, invisible and unassigned code points; letters of
// other scripts that look like those of the alphabet, and marks; bytes
// that are not valid UTF-8; words that are one once normalized; and the
// order of the checks within a word and across words.
// - alphabet: CheckList of the base list of 2048 words whose first word
// is "pala", one code point and "zz": the result of every code point up
// to U+017F, and the SHA-256 of the lines "U+XXXX result\n" of every
// code point of planes 0, 1 and 14 but the surrogates.
// - generate: Generate on the Spanish list and on lists of 0 to
// 2^20 + 1 words, while it reads fixed streams instead of crypto/rand:
// the keystream of SeededRandomSource of lib/src/random.dart (ChaCha20
// under SHA-256(seed), zero nonce); SHA-256 counter blocks; and bytes
// that end, the seed of TestGenerate among them. Each case has the
// indices drawn (and the words, from the Spanish list), the bytes read
// and the next four of an endless stream; or the text of the error and
// the bytes read.
// - bits: Bits of sizes and counts, as the 64 bits of the float64 and its
// shortest decimal; and bits_digests, the SHA-256 of the 8 big-endian
// bytes of each Bits over ranges of sizes and counts, sizes outer.
// - dice_number: DiceNumber of positions in the list of 7776 words, where
// a digit carries, and out of it, some of them large.
// - dice_word: DiceWord of five dice, as bytes, on the English and the
// Spanish lists (the cases of TestDiceWord) and on a list of the 7776
// indices: dice, and what is not five dice, such as white space around
// them, other digits, a combining mark, five bytes that are not five
// characters and bytes that are not valid UTF-8; and lists of other
// sizes.
// - dice_words: DiceWords of a text, as bytes, on the same lists and on
// lists of indices modulo a number, which give a word twice, also
// followed by characters that %q escapes: the cases of TestDiceWords;
// each white space of unicode.IsSpace between the numbers, and code
// points and bytes that are not; the order of the checks; and lists of
// other sizes.
// - dice_list: DiceList of each list of Go, the file of the EFF for en,
// and of a list of the 7776 indices, as the SHA-256 and the length of
// its text; and lists of other sizes.
// - dice_space: DiceWords of 11111, a code point and six more numbers on
// the list of the 7776 indices, for every code point of planes 0, 1 and
// 14 but the surrogates: the SHA-256 of the lines "U+XXXX result\n",
// and the code points at which the numbers are split.
//
// Binary values are lower-case hexadecimal, and the JSON is ASCII, so that
// the Dart constant is too. The output is the same on every run. It imports
// the package wordkey and reads wordkey/lists, so it runs in an export of
// datekeys-go made with git archive, without changing the repository, at
// 92e7154, the branch v0.15 after the tag spec-v0.15:
//
// commit=$(git -C ../datekeys-go rev-parse 92e7154)
// out=$PWD/test/vectors
// tmp=$(mktemp -d)
// git -C ../datekeys-go archive "$commit" | tar -x -C "$tmp"
// cp tool/wordlist_go_vectors.go "$tmp"
// (cd "$tmp" && go run ./wordlist_go_vectors.go -source "$commit" -out "$out")
// rm -rf "$tmp"
package main
import (
"bytes"
"crypto/sha256"
"encoding/binary"
"encoding/hex"
"encoding/json"
"flag"
"fmt"
"io"
"log"
"math"
"os"
"path/filepath"
"runtime"
"strconv"
"strings"
"unicode"
"golang.org/x/crypto/chacha20"
"g.activething.com/go/DateKeys/wordkey"
)
func check(err error) {
if err != nil {
log.Fatal(err)
}
}
func h(s string) string { return hex.EncodeToString([]byte(s)) }
func hexAll(ss []string) []string {
var out []string
for _, s := range ss {
out = append(out, h(s))
}
return out
}
func sum(s string) string {
b := sha256.Sum256([]byte(s))
return hex.EncodeToString(b[:])
}
// result is "ok", or the text of err.
func result(err error) string {
if err != nil {
return err.Error()
}
return "ok"
}
// ---------------------------------------------------------------------------
// Lists
// baseWord is word i of the base list of TestCheckList, "pal" and three
// letters: palaaa, palaab, …, up to i = 17 575, palzzz.
func baseWord(i int) string {
return "pal" + string([]rune{'a' + rune(i/676), 'a' + rune(i/26%26), 'a' + rune(i%26)})
}
// edit is the word at the index At of a base list replaced by Word, in hex.
type edit struct {
At int `json:"at"`
Word string `json:"word"`
}
func set(at int, word string) edit { return edit{at, h(word)} }
// edited is the base list of size words with the edits.
func edited(size int, sets []edit) []string {
l := make([]string, size)
for i := range l {
l[i] = baseWord(i)
}
for _, e := range sets {
b, err := hex.DecodeString(e.Word)
check(err)
l[e.At] = string(b)
}
return l
}
// listOf is the body of wordkey.List for the text of a list of the caller,
// restated because List reads only its built-in lists.
func listOf(lang, text string) ([]string, error) {
words := strings.Split(strings.TrimSuffix(text, "\n"), "\n")
if err := wordkey.CheckList(lang, words); err != nil {
return nil, fmt.Errorf("wordkey: the list %q: %w", lang, err)
}
return words, nil
}
type listInfo struct {
Lang string `json:"lang"`
Sha256 string `json:"sha256"`
Bytes int `json:"bytes"`
Words int `json:"words"`
}
func listInfos() []listInfo {
var out []listInfo
for _, lang := range wordkey.Languages() {
text, err := os.ReadFile(filepath.Join("wordkey", "lists", lang+".txt"))
check(err)
words, err := wordkey.List(lang)
check(err)
again, err := listOf(lang, string(text))
check(err)
if strings.Join(again, "\n") != strings.Join(words, "\n") {
log.Fatalf("wordkey/lists/%s.txt is not the list of List, or listOf is not its body", lang)
}
out = append(out, listInfo{lang, sum(string(text)), len(text), len(words)})
}
return out
}
type parseCase struct {
Name string `json:"name"`
Lang string `json:"lang"`
Size int `json:"size"`
Set []edit `json:"set,omitempty"`
Prefix string `json:"prefix"`
Sep string `json:"sep"`
Suffix string `json:"suffix"`
Sha256 string `json:"sha256"`
Words int `json:"words"`
Result string `json:"result"`
}
// parseOf reads with listOf the text prefix, the edited base list joined
// by sep, and suffix.
func parseOf(name, lang string, size int, sets []edit, prefix, sep, suffix string) parseCase {
text := prefix + strings.Join(edited(size, sets), sep) + suffix
words, err := listOf(lang, text)
return parseCase{name, lang, size, sets, h(prefix), h(sep), h(suffix), sum(text), len(words), result(err)}
}
func parseCases() []parseCase {
return []parseCase{
parseOf("2048 words, one per line, with a final LF", "es", 2048, nil, "", "\n", "\n"),
parseOf("without the final LF", "es", 2048, nil, "", "\n", ""),
parseOf("7776 words", "es", 7776, nil, "", "\n", "\n"),
parseOf("an empty last line: two final LF", "es", 2048, nil, "", "\n", "\n\n"),
parseOf("an empty first line", "es", 2048, nil, "\n", "\n", "\n"),
parseOf("CR LF lines", "es", 2048, nil, "", "\r\n", "\r\n"),
parseOf("CR LF lines, the last without", "es", 2048, nil, "", "\r\n", ""),
parseOf("CR lines", "es", 2048, nil, "", "\r", "\r"),
parseOf("a byte order mark", "es", 2048, nil, "\ufeff", "\n", "\n"),
parseOf("an empty text", "es", 0, nil, "", "\n", ""),
parseOf("an LF only", "es", 0, nil, "", "\n", "\n"),
parseOf("2047 words", "es", 2047, nil, "", "\n", "\n"),
parseOf("the words separated by spaces", "es", 2048, nil, "", " ", "\n"),
parseOf("a language without an alphabet", "xx", 2048, nil, "", "\n", "\n"),
parseOf("a byte that is not valid UTF-8", "es", 2048, []edit{set(6, "pal\xffaa")}, "", "\n", "\n"),
parseOf("a sequence cut at the end of the text", "es", 2048, []edit{set(2047, "palzz\xc3")}, "", "\n", ""),
parseOf("the same word twice", "es", 2048, []edit{set(2000, "palaaa")}, "", "\n", "\n"),
parseOf("a capital letter", "es", 2048, []edit{set(100, "Paldww")}, "", "\n", "\n"),
parseOf("letters of the alphabet", "es", 2048, []edit{set(0, "\u00e1\u00e9\u00ed\u00f3\u00fa\u00fc\u00f1")}, "", "\n", "\n"),
}
}
type checkCase struct {
Name string `json:"name"`
Lang string `json:"lang"`
Size int `json:"size"`
Set []edit `json:"set,omitempty"`
Append []string `json:"append,omitempty"`
Sha256 string `json:"sha256"`
Result string `json:"result"`
}
// checkOf runs CheckList on the edited base list of size words followed by
// appends. Sha256 is that of the words joined by LF.
func checkOf(name, lang string, size int, sets []edit, appends ...string) checkCase {
l := append(edited(size, sets), appends...)
return checkCase{name, lang, size, sets, hexAll(appends), sum(strings.Join(l, "\n")), result(wordkey.CheckList(lang, l))}
}
func checkCases() []checkCase {
at := func(i int, w string) []edit { return []edit{set(i, w)} }
out := []checkCase{
// TestCheckList, in its order.
checkOf("the base list", "es", 2048, nil),
checkOf("a language without an alphabet", "xx", 2048, nil),
checkOf("2047 words", "es", 2047, nil),
checkOf("two words in a line", "es", 2048, at(5, "dos palabras")),
checkOf("two spaces", "es", 2048, at(5, " ")),
checkOf("two letters", "es", 2048, at(5, "mi")),
checkOf("ZWSP", "es", 2048, at(5, "casa\u200b")),
checkOf("a capital", "es", 2048, at(5, "Palaaf")),
checkOf("a Cyrillic letter that looks like a Latin c", "es", 2048, at(5, "\u0441asa")),
checkOf("a digit", "es", 2048, at(5, "pal1")),
checkOf("the CR of a CR LF line", "es", 2048, at(5, "palaaf\r")),
checkOf("the same word", "es", 2048, at(5, baseWord(4))),
checkOf("the same word once normalized", "es", 2048, at(5, "pala\u00e1e")),
checkOf("papa after pap\u00e1", "es", 2048, at(0, "pap\u00e1"), "papa"),
// The language and the size.
checkOf("no words", "es", 0, nil),
checkOf("no words in a language without an alphabet", "xx", 0, nil),
checkOf("the empty language", "", 2048, nil),
checkOf("ES, in capitals", "ES", 2048, nil),
checkOf("es and a space", "es ", 2048, nil),
checkOf("espa\u00f1ol", "espa\u00f1ol", 2048, nil),
checkOf("2049 words", "es", 2049, nil),
checkOf("7776 words", "es", 7776, nil),
checkOf("17576 words", "es", 17576, nil),
// One word of three letters or more, once normalized.
checkOf("an empty word", "es", 2048, at(5, "")),
checkOf("a word of three letters", "es", 2048, at(5, "pal")),
checkOf("\u00f1u, two letters", "es", 2048, at(5, "\u00f1u")),
checkOf("\u00f1u\u00f1, three letters", "es", 2048, at(5, "\u00f1u\u00f1")),
checkOf("a space before", "es", 2048, at(5, " palaaf")),
checkOf("a space after", "es", 2048, at(5, "palaaf ")),
checkOf("a tab inside", "es", 2048, at(5, "pal\taf")),
checkOf("NBSP inside", "es", 2048, at(5, "pal\u00a0af")),
checkOf("NEL inside", "es", 2048, at(5, "pal\u0085af")),
checkOf("the ideographic space inside", "es", 2048, at(5, "pal\u3000af")),
checkOf("LS inside", "es", 2048, at(5, "pal\u2028af")),
checkOf("an LF inside", "es", 2048, at(5, "pal\naf")),
checkOf("two runes and ZWSP: the count before the runes", "es", 2048, at(5, "p\u200b")),
checkOf("three letters with a mark that goes: two", "es", 2048, at(5, "pa\u0301")),
checkOf("a\u0301b, two letters once normalized", "es", 2048, at(5, "a\u0301b")),
// The runes of the normalized word.
checkOf("a control", "es", 2048, at(5, "pal\x01af")),
checkOf("DEL", "es", 2048, at(5, "pal\x7faf")),
checkOf("a C1 control", "es", 2048, at(5, "pal\u0080af")),
checkOf("a control and a capital: the runes before the alphabet", "es", 2048, at(5, "Pal\x01af")),
checkOf("a soft hyphen", "es", 2048, at(5, "pa\u00ad")),
checkOf("ZWJ", "es", 2048, at(5, "pal\u200daf")),
checkOf("a byte order mark", "es", 2048, at(5, "\ufeffpalaaf")),
checkOf("an unassigned code point", "es", 2048, at(5, "pal\u0378")),
checkOf("a noncharacter", "es", 2048, at(5, "pal\ufffe")),
checkOf("a tag", "es", 2048, at(5, "pal\U000e0041af")),
// The alphabet, on the word as the list writes it.
checkOf("the letters of the alphabet", "es", 2048, at(5, "abcdefghijklmnopqrstuvwxyz\u00e1\u00e9\u00ed\u00f3\u00fa\u00fc\u00f1")),
checkOf("an acute accent in NFD", "es", 2048, at(5, "pala\u0301e")),
checkOf("a diaeresis in NFD", "es", 2048, at(5, "palaa\u0308")),
checkOf("a tilde in NFD", "es", 2048, at(5, "pan\u0303o")),
checkOf("\u00c1, a capital with an accent", "es", 2048, at(5, "\u00c1baco")),
checkOf("\u00d1, a capital", "es", 2048, at(5, "\u00d1and\u00fa")),
checkOf("a grave accent", "es", 2048, at(5, "pal\u00e0a")),
checkOf("a circumflex", "es", 2048, at(5, "pal\u00e2a")),
checkOf("\u00e4", "es", 2048, at(5, "pal\u00e4a")),
checkOf("\u00e7", "es", 2048, at(5, "pal\u00e7a")),
checkOf("\u00f6", "es", 2048, at(5, "pal\u00f6a")),
checkOf("\u00fd", "es", 2048, at(5, "pal\u00fda")),
checkOf("\u00df", "es", 2048, at(5, "pal\u00dfa")),
checkOf("a Cyrillic a", "es", 2048, at(5, "pal\u0430a")),
checkOf("a Greek omicron", "es", 2048, at(5, "pal\u03bfa")),
checkOf("a full-width a", "es", 2048, at(5, "pal\uff41a")),
checkOf("a mathematical bold a", "es", 2048, at(5, "pal\U0001d41aa")),
checkOf("the Kelvin sign", "es", 2048, at(5, "pal\u212aa")),
checkOf("a dotless i", "es", 2048, at(5, "pal\u0131a")),
checkOf("a hyphen", "es", 2048, at(5, "pal-af")),
checkOf("an apostrophe", "es", 2048, at(5, "pal'af")),
checkOf("a backtick and a brace, around a to z", "es", 2048, at(5, "pal`{")),
checkOf("an emoji", "es", 2048, at(5, "pal\U0001f600")),
// Bytes that are not valid UTF-8: U+FFFD, which no alphabet holds.
checkOf("a byte that is not valid UTF-8", "es", 2048, at(5, "pal\xffaa")),
checkOf("a surrogate in UTF-8 bytes", "es", 2048, at(5, "\xed\xa0\x80aaa")),
checkOf("a sequence cut short", "es", 2048, at(5, "pala\xc3")),
checkOf("an overlong NUL", "es", 2048, at(5, "\xc0\x80aaa")),
checkOf("U+FFFD itself", "es", 2048, at(5, "pal\ufffdaa")),
// The same word once normalized, and the order across words.
checkOf("a\u00f1o and ano", "es", 2048, []edit{set(5, "a\u00f1o"), set(6, "ano")}),
checkOf("ping\u00fcino and pinguino", "es", 2048, []edit{set(5, "pinguino"), set(9, "ping\u00fcino")}),
checkOf("the same word, the first of the list last", "es", 2048, nil, "palaaa"),
checkOf("a capital of a word of the list: the alphabet before the same word", "es", 2048, at(5, "PALAAE")),
checkOf("an error at line 4 and another at line 6", "es", 2048, []edit{set(3, "Mal"), set(5, "pa")}),
checkOf("an error at the last line", "es", 2048, at(2047, "x")),
checkOf("an empty last line", "es", 2048, nil, ""),
}
return out
}
// ---------------------------------------------------------------------------
// The alphabet of every code point
type runeResult struct {
Rune int `json:"rune"`
Result string `json:"result"`
}
type alphabetHead struct {
Word string `json:"word"`
Planes []int `json:"planes"`
Count int `json:"count"`
Sha256 string `json:"sha256"`
Ok []string `json:"ok"`
}
func alphabet() (alphabetHead, []runeResult) {
planes := []int{0, 1, 14}
l := edited(2048, nil)
digest := sha256.New()
var results []runeResult
var ok []string
count := 0
for _, p := range planes {
for r := rune(p << 16); r <= rune(p<<16|0xffff); r++ {
if r >= 0xd800 && r <= 0xdfff {
continue
}
l[0] = "pala" + string(r) + "zz"
res := result(wordkey.CheckList("es", l))
fmt.Fprintf(digest, "U+%04X %s\n", r, res)
count++
if r <= 0x17f {
results = append(results, runeResult{int(r), res})
}
if res == "ok" {
ok = append(ok, string(r))
}
}
}
head := alphabetHead{"pala, the code point and zz, the first word of the base list of 2048", planes, count, hex.EncodeToString(digest.Sum(nil)), ok}
return head, results
}
// ---------------------------------------------------------------------------
// Generate
// seeded is the keystream of SeededRandomSource of lib/src/random.dart:
// ChaCha20 under SHA-256(seed), with a zero nonce, from block 0.
type seeded struct{ c *chacha20.Cipher }
func newSeeded(seed string) *seeded {
key := sha256.Sum256([]byte(seed))
c, err := chacha20.NewUnauthenticatedCipher(key[:], make([]byte, chacha20.NonceSize))
check(err)
return &seeded{c}
}
func (s *seeded) Read(p []byte) (int, error) {
clear(p)
s.c.XORKeyStream(p, p)
return len(p), nil
}
// counter is SHA-256(label ‖ i) for i = 0, 1, …, a 32-bit big-endian
// counter, one block after the other.
type counter struct {
label []byte
i uint32
block []byte
}
func (c *counter) Read(p []byte) (int, error) {
for n := 0; n < len(p); {
if len(c.block) == 0 {
b := sha256.Sum256(binary.BigEndian.AppendUint32(bytes.Clone(c.label), c.i))
c.block = b[:]
c.i++
}
k := copy(p[n:], c.block)
c.block = c.block[k:]
n += k
}
return len(p), nil
}
// counting counts the bytes read from r.
type counting struct {
r io.Reader
n int
}
func (c *counting) Read(p []byte) (int, error) {
n, err := c.r.Read(p)
c.n += n
return n, err
}
// stream is what a case of generate reads instead of crypto/rand: kind
// seeded, the keystream of Seed; counter, the blocks of the label Seed; or
// bytes, Hex repeated Repeat times, which end.
type stream struct {
Kind string `json:"kind"`
Seed string `json:"seed,omitempty"`
Hex string `json:"hex,omitempty"`
Repeat int `json:"repeat,omitempty"`
}
func (s stream) reader() io.Reader {
switch s.Kind {
case "seeded":
return newSeeded(s.Seed)
case "counter":
return &counter{label: []byte(s.Seed)}
case "bytes":
b, err := hex.DecodeString(s.Hex)
check(err)
return bytes.NewReader(bytes.Repeat(b, s.Repeat))
}
log.Fatalf("stream kind %q", s.Kind)
return nil
}
func seed(s string) stream { return stream{Kind: "seeded", Seed: s} }
func given(hexBytes string, repeat int) stream {
return stream{Kind: "bytes", Hex: hexBytes, Repeat: repeat}
}
type generateCase struct {
Name string `json:"name"`
List string `json:"list,omitempty"`
Size int `json:"size"`
N int `json:"n"`
Stream stream `json:"stream"`
Indices []int `json:"indices,omitempty"`
Words []string `json:"words,omitempty"`
Read int `json:"read"`
Next string `json:"next,omitempty"`
Error string `json:"error,omitempty"`
}
// numbers is a list of size words, "0", "1", …: Generate reads only its
// length and the words it draws, whose indices they are.
func numbers(size int) []string {
l := make([]string, size)
for i := range l {
l[i] = strconv.Itoa(i)
}
return l
}
func generateOf(name, listName string, list []string, n int, s stream) generateCase {
r := &counting{r: s.reader()}
words, err := wordkey.Generate(list, n, r)
c := generateCase{Name: name, List: listName, Size: len(list), N: n, Stream: s, Read: r.n}
if err != nil {
c.Error = err.Error()
return c
}
index := make(map[string]int, len(list))
for i, w := range list {
index[w] = i
}
for _, w := range words {
c.Indices = append(c.Indices, index[w])
}
if listName != "" && n <= 24 {
c.Words = words
}
if s.Kind != "bytes" {
next := make([]byte, 4)
_, err := io.ReadFull(r.r, next)
check(err)
c.Next = hex.EncodeToString(next)
}
return c
}
func generateCases() []generateCase {
es, err := wordkey.List("es")
check(err)
var out []generateCase
add := func(c generateCase) { out = append(out, c) }
// The Spanish list, from the keystream of a seed and from SHA-256
// counter blocks.
for _, n := range []int{6, wordkey.DefaultCount, 8, 12, 24} {
add(generateOf(fmt.Sprintf("%d words of the Spanish list", n), "es", es, n, seed(fmt.Sprintf("wordkey.Generate es %d", n))))
}
for _, n := range []int{6, wordkey.DefaultCount, 12} {
add(generateOf(fmt.Sprintf("%d words of the Spanish list, SHA-256 counter blocks", n), "es", es, n, stream{Kind: "counter", Seed: fmt.Sprintf("wordkey.Generate counter %d", n)}))
}
add(generateOf("3888 words of the Spanish list, the most", "es", es, len(es)/2, seed("wordkey.Generate es 3888")))
// Lists of every number of bytes of a draw of crypto/rand.Int, with and
// without draws again.
for _, size := range []int{12, 13, 16, 17, 100, 255, 256, 257, 2047, 2048, 2049, 4096, 7775, 7776, 7777, 8192, 65535, 65536, 65537, 100000, 1<<20 + 1} {
add(generateOf(fmt.Sprintf("6 words of %d", size), "", numbers(size), 6, seed(fmt.Sprintf("wordkey.Generate %d", size))))
}
add(generateOf("12 words of 65537, SHA-256 counter blocks", "", numbers(65537), 12, stream{Kind: "counter", Seed: "wordkey.Generate counter 65537"}))
// The most words of small lists: indices drawn twice, again and again.
for _, size := range []int{12, 13, 17, 100} {
add(generateOf(fmt.Sprintf("%d words of %d, the most", size/2, size), "", numbers(size), size/2, seed(fmt.Sprintf("wordkey.Generate %d most", size))))
}
// Bytes that end.
add(generateOf("the seed of TestGenerate: two indices only, then EOF", "es", es, 6, given("0701c821", 64)))
add(generateOf("twelve bytes, six indices", "es", es, 6, given("000000010002000300040005", 1)))
add(generateOf("8191 and 8032 drawn again, 0 drawn twice, 7775 the last index", "es", es, 6, given("1fff00001f601e5f00000001000200030004", 1)))
add(generateOf("the three bits above the 13 of 7775 are cleared", "es", es, 6, given("e000ffff21002200230024002500", 1)))
add(generateOf("the bytes end inside a draw", "es", es, 6, given("0000000100", 1)))
add(generateOf("no bytes", "es", es, 6, given("", 0)))
// The arguments, refused before anything is read.
for _, n := range []int{5, 0, -1, len(es)/2 + 1, len(es)} {
add(generateOf(fmt.Sprintf("%d words of the Spanish list", n), "es", es, n, seed("never read")))
}
for _, c := range []struct{ size, n int }{{11, 6}, {0, 6}, {0, 5}, {12, 7}, {13, 7}} {
add(generateOf(fmt.Sprintf("%d words of %d", c.n, c.size), "", numbers(c.size), c.n, seed("never read")))
}
return out
}
// ---------------------------------------------------------------------------
// Bits
type bitsCase struct {
Size int `json:"size"`
Count int `json:"count"`
Bits string `json:"bits"`
Hex string `json:"hex"`
}
func bitsOf(size, count int) bitsCase {
b := wordkey.Bits(size, count)
return bitsCase{size, count, strconv.FormatFloat(b, 'g', -1, 64), fmt.Sprintf("%016x", math.Float64bits(b))}
}
func bitsCases() []bitsCase {
var out []bitsCase
for _, size := range []int{1, 2, 3, 5, 12, 13, 100, 255, 256, 257, 2047, 2048, 2049, 4096, 7775, 7776, 7777, 8192, 10000, 65535, 65536, 65537, 1 << 20, 1<<31 - 1, 1 << 31, 1<<32 - 1, 1 << 32, 1<<32 + 1, 1<<52 + 1, 1<<53 - 1, 1 << 53} {
for _, count := range []int{0, 1, 2, 6, 7, 8, 12} {
out = append(out, bitsOf(size, count))
}
}
// Go adds log2 of 0, −Inf, and of a negative number, NaN, and adds
// nothing for a negative count.
for _, c := range [][2]int{{0, 1}, {-1, 1}, {7776, -1}, {0, 0}} {
out = append(out, bitsOf(c[0], c[1]))
}
return out
}
type bitsDigest struct {
Name string `json:"name"`
Sizes [2]int `json:"sizes"`
Counts [2]int `json:"counts"`
Sha256 string `json:"sha256"`
}
func digestOf(name string, sizes, counts [2]int) bitsDigest {
d := sha256.New()
var b [8]byte
for size := sizes[0]; size <= sizes[1]; size++ {
for count := counts[0]; count <= counts[1]; count++ {
binary.BigEndian.PutUint64(b[:], math.Float64bits(wordkey.Bits(size, count)))
d.Write(b[:])
}
}
return bitsDigest{name, sizes, counts, hex.EncodeToString(d.Sum(nil))}
}
func bitsDigests() []bitsDigest {
return []bitsDigest{
digestOf("one word of 1 to 65536: the log2 of Go", [2]int{1, 65536}, [2]int{1, 1}),
digestOf("6 to 8 words of 2048 to 10000", [2]int{2048, 10000}, [2]int{6, 8}),
digestOf("0 to 600 words of 7776", [2]int{7776, 7776}, [2]int{0, 600}),
digestOf("1 word of 2^32 - 4096 to 2^32 + 4096", [2]int{1<<32 - 4096, 1<<32 + 4096}, [2]int{1, 1}),
}
}
// ---------------------------------------------------------------------------
// Dice
// diceListOf is a list for the dice: the built-in list of the language
// list or, when list is empty, size words, each its index in decimal,
// modulo mod when mod is not 0, so that a word comes back every mod
// positions, followed by suffix.
func diceListOf(list string, size, mod int, suffix string) []string {
if list != "" {
l, err := wordkey.List(list)
check(err)
return l
}
l := numbers(size)
for i := range l {
if mod != 0 {
l[i] = strconv.Itoa(i % mod)
}
l[i] += suffix
}
return l
}
// errText is the text of err, or nothing.
func errText(err error) string {
if err != nil {
return err.Error()
}
return ""
}
type diceNumberCase struct {
I int `json:"i"`
Dice string `json:"dice,omitempty"`
Error string `json:"error,omitempty"`
}
func diceNumberCases() []diceNumberCase {
var out []diceNumberCase
// The positions of TestDiceNumber and others where a digit carries,
// then positions out of the list, some of them large.
for _, i := range []int{0, 1, 5, 6, 35, 36, 215, 216, 1295, 1296, 2047, 2048, 3495, 7774, 7775,
-1, 7776, 7777, -7776, 1 << 31, -1 << 31, 1 << 32, 1<<53 - 1, -(1<<53 - 1)} {
d, err := wordkey.DiceNumber(i)
out = append(out, diceNumberCase{i, d, errText(err)})
}
return out
}
type diceWordCase struct {
Name string `json:"name"`
List string `json:"list,omitempty"`
Size int `json:"size"`
Dice string `json:"dice"`
Word string `json:"word,omitempty"`
Error string `json:"error,omitempty"`
}
// diceWordOf runs DiceWord on the list of diceListOf and the bytes dice,
// which the case keeps in hex.
func diceWordOf(name, list string, size int, dice string) diceWordCase {
l := diceListOf(list, size, 0, "")
w, err := wordkey.DiceWord(l, dice)
return diceWordCase{name, list, len(l), h(dice), w, errText(err)}
}
func diceWordCases() []diceWordCase {
var out []diceWordCase
add := func(c diceWordCase) { out = append(out, c) }
// TestDiceWord, on the lists of Go.
for _, c := range []struct{ list, dice string }{
{"en", "11111"}, {"en", "11112"}, {"en", "35214"}, {"en", "66666"},
{"es", "11111"}, {"es", "35214"}, {"es", "66666"},
} {
add(diceWordOf(c.list+" "+c.dice, c.list, 0, c.dice))
}
// Five dice on the list of the indices: where a digit carries, and
// others.
for _, dice := range []string{"11111", "11112", "11116", "11121", "11166", "11211", "16666", "21111", "12345", "35214", "65432", "66665", "66666"} {
add(diceWordOf(dice, "", wordkey.DiceListSize, dice))
}
// Not five dice: the cases of TestDiceWord, then white space, other
// digits and characters, marks, five bytes that are not five
// characters, and bytes that are not valid UTF-8.
for _, c := range []struct{ name, dice string }{
{"no dice", ""},
{"four dice", "1111"},
{"six dice", "111111"},
{"a 0", "11110"},
{"a 7", "11117"},
{"a letter", "a1111"},
{"four dice and a space", "1111 "},
{"a full-width 1 and four dice", "\uff111111"},
{"a space and four dice", " 1111"},
{"a space and five dice", " 11111"},
{"five dice and a space", "11111 "},
{"five dice and LF", "11111\n"},
{"a tab and four dice", "\t1111"},
{"two and two dice around a space", "11 11"},
{"NUL", "1111\x00"},
{"DEL", "1111\x7f"},
{"a quotation mark", "1111\""},
{"a backslash", "1111\\"},
{"00000", "00000"},
{"77777", "77777"},
{"99999", "99999"},
{"a slash, the byte before 0", "1111/"},
{"a colon, the byte after 9", "1111:"},
{"five full-width digits", "\uff11\uff12\uff13\uff14\uff15"},
{"five Arabic-Indic digits", "\u0661\u0662\u0663\u0664\u0665"},
{"a superscript 1", "1111\u00b9"},
{"a circled 1 and two dice: five bytes", "\u246011"},
{"a mathematical bold 1 and a die: five bytes", "\U0001d7cf1"},
{"four dice and an e with an acute accent: five characters", "1111\u00e9"},
{"three dice and an e with an acute accent: five bytes", "111\u00e9"},
{"five dice and a combining acute accent", "11111\u0301"},
{"three dice and the emoji of a die: five UTF-16 code units", "111\U0001f3b2"},
{"NBSP and four dice", "\u00a01111"},
{"an ideographic space and five dice", "\u300011111"},
{"two dice and a surrogate in UTF-8 bytes: five bytes", "11\xed\xa0\x80"},
{"four dice and a byte that is not valid UTF-8", "1111\xff"},
{"five bytes that are not valid UTF-8", "\xff\xfe\xfd\xfc\xfb"},
{"four dice and a sequence cut short", "1111\xc3"},
} {
add(diceWordOf(c.name, "", wordkey.DiceListSize, c.dice))
}
// A list of another size, refused before the dice.
for _, c := range []struct {
size int
dice string
}{{2048, "11111"}, {7775, "11111"}, {7777, "66666"}, {0, ""}, {1, "x"}} {
add(diceWordOf(fmt.Sprintf("%q on a list of %d words", c.dice, c.size), "", c.size, c.dice))
}
return out
}
type diceWordsCase struct {
Name string `json:"name"`
List string `json:"list,omitempty"`
Size int `json:"size"`
Mod int `json:"mod,omitempty"`
Suffix string `json:"suffix,omitempty"`
Dice string `json:"dice"`
Words []string `json:"words,omitempty"`
Error string `json:"error,omitempty"`
}
// diceWordsOf runs DiceWords on the list of diceListOf and the bytes dice,
// which the case keeps in hex.
func diceWordsOf(name, list string, size, mod int, suffix, dice string) diceWordsCase {
l := diceListOf(list, size, mod, suffix)
words, err := wordkey.DiceWords(l, dice)
return diceWordsCase{name, list, len(l), mod, suffix, h(dice), words, errText(err)}
}
func diceWordsCases() []diceWordsCase {
var out []diceWordsCase
add := func(c diceWordsCase) { out = append(out, c) }
const size = wordkey.DiceListSize
// Seven numbers, and the six after the first.
seven := "11111 11112 11113 11114 11115 11116 11121"
six := seven[len("11111 "):]
// TestDiceWords, on the English list and on the list of the indices.
for _, list := range []string{"en", ""} {
for _, c := range []struct{ name, dice string }{
{"seven numbers among white space", " 11111 11112\t11113\n11114 11115 11116 11121 "},
{"five numbers", "11111 11112 11113 11114 11115"},
{"no numbers", ""},
{"four dice", "11111 11112 11113 11114 11115 1116"},
{"11111 twice", "11111 11112 11113 11114 11115 11111"},
{"a comma between two numbers", "11111,11112 11113 11114 11115 11116 11121"},
} {
name := c.name
if list != "" {
name = list + ", " + name
}
add(diceWordsOf(name, list, size, 0, "", c.dice))
}
}
add(diceWordsOf("a list of 7775 words", "", 7775, 0, "", "11111 11112 11113 11114 11115 11116"))
// The Spanish list: words with accents, and the last word given twice.
add(diceWordsOf("es, seven numbers", "es", 0, 0, "", "66666 11112 35214 12345 54321 23456 11111"))
add(diceWordsOf("es, 66666 twice", "es", 0, 0, "", "66666 11111 35214 12345 54321 23456 66666"))
// The fewest numbers, more, and none.
add(diceWordsOf("six numbers", "", size, 0, "", six))
add(diceWordsOf("twelve numbers", "", size, 0, "", "11111 22222 33333 44444 55555 66666 12345 23456 34561 45612 56123 61234"))
add(diceWordsOf("white space only", "", size, 0, "", " \t\n\u00a0\u3000 "))
// Each white space of unicode.IsSpace between the numbers; then runes
// and bytes that are not white space between the first two, or around
// a number, which is then not five dice.
for r := rune(0); r <= unicode.MaxRune; r++ {
if unicode.IsSpace(r) {
add(diceWordsOf(fmt.Sprintf("U+%04X between the numbers", r), "", size, 0, "", strings.ReplaceAll(seven, " ", string(r))))
}
}
for _, r := range []rune{0x1c, 0x1d, 0x1e, 0x1f, 0xad, 0x301, 0x180e, 0x200b, 0x200c, 0x200d, 0x2060, 0x2800, 0x3164, 0xfeff, 0xfffd} {
add(diceWordsOf(fmt.Sprintf("U+%04X between the first two numbers", r), "", size, 0, "", "11111"+string(r)+six))
}
for _, c := range []struct{ name, dice string }{
{"the byte A0 alone, which is not NBSP", "11111\xa0" + six},
{"the byte 85 alone, which is not NEL", "11111\x85" + six},
{"a byte that is not valid UTF-8 after a number", "11111\xff " + six},
{"a surrogate in UTF-8 bytes after a number", "11111\xed\xa0\x80 " + six},
{"five full-width digits", "\uff11\uff11\uff11\uff11\uff11 " + six},
{"a combining acute accent after a number", "11111\u0301 " + six},
{"a combining acute accent before a number", "\u030111111 " + six},
// The order of the checks.
{"five fields that are not dice: the count first", "a b c d e"},
{"six fields, the first not dice", "abc 11111 11112 11113 11114 11115"},
{"not dice before a word given twice", "11111 11112 abc 11111 11113 11114"},
{"a word given twice before not dice", "11111 11112 11113 11114 11115 11111 abc"},
} {
add(diceWordsOf(c.name, "", size, 0, "", c.dice))
}
// Lists that hold a word more than once: DiceWords refuses the word,
// whatever the dice that give it.
n1000, err := wordkey.DiceNumber(1000)
check(err)
for _, c := range []struct {
mod int
dice string
}{{1, six}, {7, seven}, {7, seven + " 11122"}, {1000, "11111 11112 11113 11114 11115 " + n1000}} {
add(diceWordsOf(fmt.Sprintf("indices modulo %d: %s", c.mod, c.dice), "", size, c.mod, "", c.dice))
}
// Words that %q escapes: a quotation mark, a backslash, a tab, NBSP and
// ZWSP after each index modulo 7, so that the eighth number gives the
// word of the first again.
for _, dice := range []string{seven, seven + " 11122"} {
add(diceWordsOf("indices modulo 7 and a suffix: "+dice, "", size, 7, "\"\\\t\u00a0\u200b", dice))
}
// A list of another size, refused before the numbers.
for _, c := range []struct {
size int
dice string
}{{0, ""}, {2048, seven}, {7777, "x"}} {
add(diceWordsOf(fmt.Sprintf("%q on a list of %d words", c.dice, c.size), "", c.size, 0, "", c.dice))
}
return out
}
type diceListCase struct {
List string `json:"list,omitempty"`
Size int `json:"size"`
Sha256 string `json:"sha256,omitempty"`
Bytes int `json:"bytes,omitempty"`
Error string `json:"error,omitempty"`
}
func diceListCases() []diceListCase {
var out []diceListCase
add := func(list string, size int) {
l := diceListOf(list, size, 0, "")
text, err := wordkey.DiceList(l)
c := diceListCase{List: list, Size: len(l), Error: errText(err)}
if err == nil {
c.Sha256, c.Bytes = sum(text), len(text)
}
out = append(out, c)
}
for _, lang := range wordkey.Languages() {
add(lang, 0)
}
for _, size := range []int{wordkey.DiceListSize, 0, 1, 2048, 7775, 7777} {
add("", size)
}
return out
}
type diceSpaceHead struct {
Text string `json:"text"`
Planes []int `json:"planes"`
Count int `json:"count"`
Sha256 string `json:"sha256"`
Split []int `json:"split"`
}
// diceSpace runs DiceWords of 11111, a code point and six more numbers on
// the list of the indices, for every code point of the planes but the
// surrogates.
func diceSpace() diceSpaceHead {
planes := []int{0, 1, 14}
l := numbers(wordkey.DiceListSize)
digest := sha256.New()
var split []int
count := 0
for _, p := range planes {
for r := rune(p << 16); r <= rune(p<<16|0xffff); r++ {
if r >= 0xd800 && r <= 0xdfff {
continue
}
_, err := wordkey.DiceWords(l, "11111"+string(r)+"11112 11113 11114 11115 11116 11121")
fmt.Fprintf(digest, "U+%04X %s\n", r, result(err))
count++
if err == nil {
split = append(split, int(r))
}
}
}
return diceSpaceHead{"11111, the code point and 11112 11113 11114 11115 11116 11121, on the list of the 7776 indices", planes, count, hex.EncodeToString(digest.Sum(nil)), split}
}
// ---------------------------------------------------------------------------
// JSON
// ascii escapes every code point above U+007F of JSON as \uXXXX, in UTF-16
// for those above U+FFFF: the same JSON, in ASCII.
func ascii(b []byte) []byte {
var out bytes.Buffer
for _, r := range string(b) {
switch {
case r < 0x80:
out.WriteByte(byte(r))
case r < 0x10000:
fmt.Fprintf(&out, `\u%04x`, r)
default:
r -= 0x10000
fmt.Fprintf(&out, `\u%04x\u%04x`, 0xd800+(r>>10), 0xdc00+(r&0x3ff))
}
}
return out.Bytes()
}
func marshal(v any) []byte {
var b bytes.Buffer
enc := json.NewEncoder(&b)
enc.SetEscapeHTML(false)
check(enc.Encode(v))
return ascii(bytes.TrimSuffix(b.Bytes(), []byte("\n")))
}
// list writes the items of a JSON array one per line.
func list[T any](w *bytes.Buffer, key string, items []T, last bool) {
fmt.Fprintf(w, " %q: [\n", key)
for i, it := range items {
w.WriteString(" ")
w.Write(marshal(it))
if i < len(items)-1 {
w.WriteByte(',')
}
w.WriteByte('\n')
}
w.WriteString(" ]")
if !last {
w.WriteByte(',')
}
w.WriteByte('\n')
}
func main() {
outDir := flag.String("out", "", "the directory test/vectors of datekeys-dart")
source := flag.String("source", "", "the commit of datekeys-go")
flag.Parse()
if *outDir == "" || *source == "" {
log.Fatal("usage: go run wordlist_go_vectors.go -source COMMIT -out DIR")
}
lists := listInfos()
parse := parseCases()
checks := checkCases()
alpha, runes := alphabet()
gen := generateCases()
bits := bitsCases()
digests := bitsDigests()
space := diceSpace()
diceNumbers := diceNumberCases()
diceWord := diceWordCases()
diceWords := diceWordsCases()
diceLists := diceListCases()
var w bytes.Buffer
w.WriteString("{\n")
head := []struct {
key string
v any
}{
{"description", "The word lists of package wordkey of the Go reference (generate.go and dice.go, spec \u00a738.1) for lib/src/wordlist.dart; see tool/wordlist_go_vectors.go. " +
"Binary values and words are lower-case hex. The base list is word i = pal and three letters, a + i/676, a + i/26%26 and a + i%26; a case of size n takes its first n words, replaces the word at each at of set, and adds those of append. " +
"lists: the SHA-256, bytes and words of each built-in list. parse: the body of List on the text prefix, the base list joined by sep, and suffix (sha256 is that of the text): the words read, and ok or the error. " +
"check_list: CheckList, ok or the error (sha256 is that of the words joined by LF). alphabet: CheckList of the base list of 2048 whose first word is pala, a code point and zz, for each code point of the planes but the surrogates: the SHA-256 of the lines U+XXXX result, the code points of the words it accepts, and alphabet_results, each result up to U+017F. " +
"generate: Generate on the list of the Spanish words (list es) or on a list of size words whose indices they are, reading a stream: seeded, the keystream of SeededRandomSource of the seed; counter, SHA-256 of the label and a 32-bit big-endian counter from 0, block after block; bytes, hex repeated repeat times, which end. The indices drawn, the bytes read and the next four of an endless stream, or the error. " +
"bits: Bits, its shortest decimal and the hex of its float64. bits_digests: the SHA-256 of the 8 big-endian bytes of each Bits of the sizes and counts, inclusive, sizes outer. " +
"dice_list_size: DiceListSize. dice_space: DiceWords of the text 11111, a code point and 11112 11113 11114 11115 11116 11121 on the list of the 7776 indices, for each code point of the planes but the surrogates: the SHA-256 of the lines U+XXXX result, and split, the code points at which the numbers are split. " +
"dice_number: DiceNumber of i, the dice or the error. dice_word and dice_words: DiceWord and DiceWords of the bytes dice, in hex, on the English or the Spanish list (list en or es) or on a list of size words, each its index in decimal, modulo mod if there is one, followed by suffix: the word or the words, as text, or the error. " +
"dice_list: DiceList of the list, the SHA-256 and the bytes of its text, or the error."},
{"generator", "tool/wordlist_go_vectors.go, " + runtime.Version()},
{"source", *source},
{"default_count", wordkey.DefaultCount},
{"min_list_size", wordkey.MinListSize},
{"min_words", wordkey.MinWords},
{"min_letters", wordkey.MinLetters},
{"alphabet", alpha},
{"dice_list_size", wordkey.DiceListSize},
{"dice_space", space},
}
for _, kv := range head {
fmt.Fprintf(&w, " %q: %s,\n", kv.key, marshal(kv.v))
}
list(&w, "lists", lists, false)
list(&w, "parse", parse, false)
list(&w, "check_list", checks, false)
list(&w, "alphabet_results", runes, false)
list(&w, "generate", gen, false)
list(&w, "bits", bits, false)
list(&w, "bits_digests", digests, false)
list(&w, "dice_number", diceNumbers, false)
list(&w, "dice_word", diceWord, false)
list(&w, "dice_words", diceWords, false)
list(&w, "dice_list", diceLists, true)
w.WriteString("}\n")
var parsed any
if err := json.Unmarshal(w.Bytes(), &parsed); err != nil {
log.Fatalf("the JSON does not parse: %v", err)
}
if bytes.Contains(w.Bytes(), []byte("'''")) {
log.Fatal("the JSON holds three quotes")
}
path := filepath.Join(*outDir, "wordlist_vectors.json")
check(os.WriteFile(path, w.Bytes(), 0o644))
fmt.Printf("wrote %s, %d bytes: %d lists, %d texts, %d lists checked, %d code points, %d draws, %d bits, %d digests; dice: %d code points, %d numbers, %d words, %d texts, %d lists\n",
path, w.Len(), len(lists), len(parse), len(checks), alpha.Count, len(gen), len(bits), len(digests),
space.Count, len(diceNumbers), len(diceWord), len(diceWords), len(diceLists))
dart := "// Generated by tool/wordlist_go_vectors.go from wordlist_vectors.json, for\n" +
"// the tests that also run compiled to JavaScript, where no file can be\n" +
"// read. Do not edit.\n\n" +
"/// The text of test/vectors/wordlist_vectors.json.\n" +
"const wordlistVectorsJson = r'''\n" + w.String() + "''';\n"
dpath := filepath.Join(*outDir, "wordlist_vectors.g.dart")
check(os.WriteFile(dpath, []byte(dart), 0o644))
fmt.Printf("wrote %s\n", dpath)
}

Powered by TurnKey Linux.