|
|
//go:build ignore
|
|
|
|
|
|
// Writes test/vectors/wordlist_vectors.json and the same JSON as a Dart
|
|
|
// constant, wordlist_vectors.g.dart: the results of the word lists of
|
|
|
// package wordkey of the Go reference (generate.go: List, CheckList,
|
|
|
// Generate and Bits, the random words of spec §38.1) for
|
|
|
// lib/src/wordlist.dart. Every expected value is computed here by the Go
|
|
|
// reference; none is written by hand.
|
|
|
//
|
|
|
// - lists: the built-in lists, with the SHA-256, the size and the number
|
|
|
// of words of each file of wordkey/lists, which List reads.
|
|
|
// - parse: the body of List, restated because List reads only its
|
|
|
// built-in lists, on texts of the base list of TestCheckList, "pal" and
|
|
|
// three letters, joined by a separator, with a prefix, a suffix and
|
|
|
// edits: with and without the final LF, CR LF lines, an empty last
|
|
|
// line, a byte order mark, an empty text, bytes that are not valid
|
|
|
// UTF-8. The restatement gives the words of List on every built-in list.
|
|
|
// - check_list: CheckList on the base list of 0 to 17 576 words, edited:
|
|
|
// the cases of TestCheckList; languages without an alphabet; white
|
|
|
// space, controls, invisible and unassigned code points; letters of
|
|
|
// other scripts that look like those of the alphabet, and marks; bytes
|
|
|
// that are not valid UTF-8; words that are one once normalized; and the
|
|
|
// order of the checks within a word and across words.
|
|
|
// - alphabet: CheckList of the base list of 2048 words whose first word
|
|
|
// is "pala", one code point and "zz": the result of every code point up
|
|
|
// to U+017F, and the SHA-256 of the lines "U+XXXX result\n" of every
|
|
|
// code point of planes 0, 1 and 14 but the surrogates.
|
|
|
// - generate: Generate on the Spanish list and on lists of 0 to
|
|
|
// 2^20 + 1 words, while it reads fixed streams instead of crypto/rand:
|
|
|
// the keystream of SeededRandomSource of lib/src/random.dart (ChaCha20
|
|
|
// under SHA-256(seed), zero nonce); SHA-256 counter blocks; and bytes
|
|
|
// that end, the seed of TestGenerate among them. Each case has the
|
|
|
// indices drawn (and the words, from the Spanish list), the bytes read
|
|
|
// and the next four of an endless stream; or the text of the error and
|
|
|
// the bytes read.
|
|
|
// - bits: Bits of sizes and counts, as the 64 bits of the float64 and its
|
|
|
// shortest decimal; and bits_digests, the SHA-256 of the 8 big-endian
|
|
|
// bytes of each Bits over ranges of sizes and counts, sizes outer.
|
|
|
//
|
|
|
// Binary values are lower-case hexadecimal, and the JSON is ASCII, so that
|
|
|
// the Dart constant is too. The output is the same on every run. It imports
|
|
|
// the package wordkey and reads wordkey/lists, so it runs in an export of
|
|
|
// datekeys-go made with git archive, without changing the repository, at
|
|
|
// 27a75ee, the branch v0.15 after the tag spec-v0.15:
|
|
|
//
|
|
|
// commit=$(git -C ../datekeys-go rev-parse 27a75ee)
|
|
|
// out=$PWD/test/vectors
|
|
|
// tmp=$(mktemp -d)
|
|
|
// git -C ../datekeys-go archive "$commit" | tar -x -C "$tmp"
|
|
|
// cp tool/wordlist_go_vectors.go "$tmp"
|
|
|
// (cd "$tmp" && go run ./wordlist_go_vectors.go -source "$commit" -out "$out")
|
|
|
// rm -rf "$tmp"
|
|
|
package main
|
|
|
|
|
|
import (
|
|
|
"bytes"
|
|
|
"crypto/sha256"
|
|
|
"encoding/binary"
|
|
|
"encoding/hex"
|
|
|
"encoding/json"
|
|
|
"flag"
|
|
|
"fmt"
|
|
|
"io"
|
|
|
"log"
|
|
|
"math"
|
|
|
"os"
|
|
|
"path/filepath"
|
|
|
"runtime"
|
|
|
"strconv"
|
|
|
"strings"
|
|
|
|
|
|
"golang.org/x/crypto/chacha20"
|
|
|
|
|
|
"g.activething.com/go/DateKeys/wordkey"
|
|
|
)
|
|
|
|
|
|
func check(err error) {
|
|
|
if err != nil {
|
|
|
log.Fatal(err)
|
|
|
}
|
|
|
}
|
|
|
|
|
|
func h(s string) string { return hex.EncodeToString([]byte(s)) }
|
|
|
|
|
|
func hexAll(ss []string) []string {
|
|
|
var out []string
|
|
|
for _, s := range ss {
|
|
|
out = append(out, h(s))
|
|
|
}
|
|
|
return out
|
|
|
}
|
|
|
|
|
|
func sum(s string) string {
|
|
|
b := sha256.Sum256([]byte(s))
|
|
|
return hex.EncodeToString(b[:])
|
|
|
}
|
|
|
|
|
|
// result is "ok", or the text of err.
|
|
|
func result(err error) string {
|
|
|
if err != nil {
|
|
|
return err.Error()
|
|
|
}
|
|
|
return "ok"
|
|
|
}
|
|
|
|
|
|
// ---------------------------------------------------------------------------
|
|
|
// Lists
|
|
|
|
|
|
// baseWord is word i of the base list of TestCheckList, "pal" and three
|
|
|
// letters: palaaa, palaab, …, up to i = 17 575, palzzz.
|
|
|
func baseWord(i int) string {
|
|
|
return "pal" + string([]rune{'a' + rune(i/676), 'a' + rune(i/26%26), 'a' + rune(i%26)})
|
|
|
}
|
|
|
|
|
|
// edit is the word at the index At of a base list replaced by Word, in hex.
|
|
|
type edit struct {
|
|
|
At int `json:"at"`
|
|
|
Word string `json:"word"`
|
|
|
}
|
|
|
|
|
|
func set(at int, word string) edit { return edit{at, h(word)} }
|
|
|
|
|
|
// edited is the base list of size words with the edits.
|
|
|
func edited(size int, sets []edit) []string {
|
|
|
l := make([]string, size)
|
|
|
for i := range l {
|
|
|
l[i] = baseWord(i)
|
|
|
}
|
|
|
for _, e := range sets {
|
|
|
b, err := hex.DecodeString(e.Word)
|
|
|
check(err)
|
|
|
l[e.At] = string(b)
|
|
|
}
|
|
|
return l
|
|
|
}
|
|
|
|
|
|
// listOf is the body of wordkey.List for the text of a list of the caller,
|
|
|
// restated because List reads only its built-in lists.
|
|
|
func listOf(lang, text string) ([]string, error) {
|
|
|
words := strings.Split(strings.TrimSuffix(text, "\n"), "\n")
|
|
|
if err := wordkey.CheckList(lang, words); err != nil {
|
|
|
return nil, fmt.Errorf("wordkey: the list %q: %w", lang, err)
|
|
|
}
|
|
|
return words, nil
|
|
|
}
|
|
|
|
|
|
type listInfo struct {
|
|
|
Lang string `json:"lang"`
|
|
|
Sha256 string `json:"sha256"`
|
|
|
Bytes int `json:"bytes"`
|
|
|
Words int `json:"words"`
|
|
|
}
|
|
|
|
|
|
func listInfos() []listInfo {
|
|
|
var out []listInfo
|
|
|
for _, lang := range wordkey.Languages() {
|
|
|
text, err := os.ReadFile(filepath.Join("wordkey", "lists", lang+".txt"))
|
|
|
check(err)
|
|
|
words, err := wordkey.List(lang)
|
|
|
check(err)
|
|
|
again, err := listOf(lang, string(text))
|
|
|
check(err)
|
|
|
if strings.Join(again, "\n") != strings.Join(words, "\n") {
|
|
|
log.Fatalf("wordkey/lists/%s.txt is not the list of List, or listOf is not its body", lang)
|
|
|
}
|
|
|
out = append(out, listInfo{lang, sum(string(text)), len(text), len(words)})
|
|
|
}
|
|
|
return out
|
|
|
}
|
|
|
|
|
|
type parseCase struct {
|
|
|
Name string `json:"name"`
|
|
|
Lang string `json:"lang"`
|
|
|
Size int `json:"size"`
|
|
|
Set []edit `json:"set,omitempty"`
|
|
|
Prefix string `json:"prefix"`
|
|
|
Sep string `json:"sep"`
|
|
|
Suffix string `json:"suffix"`
|
|
|
Sha256 string `json:"sha256"`
|
|
|
Words int `json:"words"`
|
|
|
Result string `json:"result"`
|
|
|
}
|
|
|
|
|
|
// parseOf reads with listOf the text prefix, the edited base list joined
|
|
|
// by sep, and suffix.
|
|
|
func parseOf(name, lang string, size int, sets []edit, prefix, sep, suffix string) parseCase {
|
|
|
text := prefix + strings.Join(edited(size, sets), sep) + suffix
|
|
|
words, err := listOf(lang, text)
|
|
|
return parseCase{name, lang, size, sets, h(prefix), h(sep), h(suffix), sum(text), len(words), result(err)}
|
|
|
}
|
|
|
|
|
|
func parseCases() []parseCase {
|
|
|
return []parseCase{
|
|
|
parseOf("2048 words, one per line, with a final LF", "es", 2048, nil, "", "\n", "\n"),
|
|
|
parseOf("without the final LF", "es", 2048, nil, "", "\n", ""),
|
|
|
parseOf("7776 words", "es", 7776, nil, "", "\n", "\n"),
|
|
|
parseOf("an empty last line: two final LF", "es", 2048, nil, "", "\n", "\n\n"),
|
|
|
parseOf("an empty first line", "es", 2048, nil, "\n", "\n", "\n"),
|
|
|
parseOf("CR LF lines", "es", 2048, nil, "", "\r\n", "\r\n"),
|
|
|
parseOf("CR LF lines, the last without", "es", 2048, nil, "", "\r\n", ""),
|
|
|
parseOf("CR lines", "es", 2048, nil, "", "\r", "\r"),
|
|
|
parseOf("a byte order mark", "es", 2048, nil, "\ufeff", "\n", "\n"),
|
|
|
parseOf("an empty text", "es", 0, nil, "", "\n", ""),
|
|
|
parseOf("an LF only", "es", 0, nil, "", "\n", "\n"),
|
|
|
parseOf("2047 words", "es", 2047, nil, "", "\n", "\n"),
|
|
|
parseOf("the words separated by spaces", "es", 2048, nil, "", " ", "\n"),
|
|
|
parseOf("a language without an alphabet", "xx", 2048, nil, "", "\n", "\n"),
|
|
|
parseOf("a byte that is not valid UTF-8", "es", 2048, []edit{set(6, "pal\xffaa")}, "", "\n", "\n"),
|
|
|
parseOf("a sequence cut at the end of the text", "es", 2048, []edit{set(2047, "palzz\xc3")}, "", "\n", ""),
|
|
|
parseOf("the same word twice", "es", 2048, []edit{set(2000, "palaaa")}, "", "\n", "\n"),
|
|
|
parseOf("a capital letter", "es", 2048, []edit{set(100, "Paldww")}, "", "\n", "\n"),
|
|
|
parseOf("letters of the alphabet", "es", 2048, []edit{set(0, "\u00e1\u00e9\u00ed\u00f3\u00fa\u00fc\u00f1")}, "", "\n", "\n"),
|
|
|
}
|
|
|
}
|
|
|
|
|
|
type checkCase struct {
|
|
|
Name string `json:"name"`
|
|
|
Lang string `json:"lang"`
|
|
|
Size int `json:"size"`
|
|
|
Set []edit `json:"set,omitempty"`
|
|
|
Append []string `json:"append,omitempty"`
|
|
|
Sha256 string `json:"sha256"`
|
|
|
Result string `json:"result"`
|
|
|
}
|
|
|
|
|
|
// checkOf runs CheckList on the edited base list of size words followed by
|
|
|
// appends. Sha256 is that of the words joined by LF.
|
|
|
func checkOf(name, lang string, size int, sets []edit, appends ...string) checkCase {
|
|
|
l := append(edited(size, sets), appends...)
|
|
|
return checkCase{name, lang, size, sets, hexAll(appends), sum(strings.Join(l, "\n")), result(wordkey.CheckList(lang, l))}
|
|
|
}
|
|
|
|
|
|
func checkCases() []checkCase {
|
|
|
at := func(i int, w string) []edit { return []edit{set(i, w)} }
|
|
|
out := []checkCase{
|
|
|
// TestCheckList, in its order.
|
|
|
checkOf("the base list", "es", 2048, nil),
|
|
|
checkOf("a language without an alphabet", "xx", 2048, nil),
|
|
|
checkOf("2047 words", "es", 2047, nil),
|
|
|
checkOf("two words in a line", "es", 2048, at(5, "dos palabras")),
|
|
|
checkOf("two spaces", "es", 2048, at(5, " ")),
|
|
|
checkOf("two letters", "es", 2048, at(5, "mi")),
|
|
|
checkOf("ZWSP", "es", 2048, at(5, "casa\u200b")),
|
|
|
checkOf("a capital", "es", 2048, at(5, "Palaaf")),
|
|
|
checkOf("a Cyrillic letter that looks like a Latin c", "es", 2048, at(5, "\u0441asa")),
|
|
|
checkOf("a digit", "es", 2048, at(5, "pal1")),
|
|
|
checkOf("the CR of a CR LF line", "es", 2048, at(5, "palaaf\r")),
|
|
|
checkOf("the same word", "es", 2048, at(5, baseWord(4))),
|
|
|
checkOf("the same word once normalized", "es", 2048, at(5, "pala\u00e1e")),
|
|
|
checkOf("papa after pap\u00e1", "es", 2048, at(0, "pap\u00e1"), "papa"),
|
|
|
|
|
|
// The language and the size.
|
|
|
checkOf("no words", "es", 0, nil),
|
|
|
checkOf("no words in a language without an alphabet", "xx", 0, nil),
|
|
|
checkOf("the empty language", "", 2048, nil),
|
|
|
checkOf("ES, in capitals", "ES", 2048, nil),
|
|
|
checkOf("es and a space", "es ", 2048, nil),
|
|
|
checkOf("espa\u00f1ol", "espa\u00f1ol", 2048, nil),
|
|
|
checkOf("2049 words", "es", 2049, nil),
|
|
|
checkOf("7776 words", "es", 7776, nil),
|
|
|
checkOf("17576 words", "es", 17576, nil),
|
|
|
|
|
|
// One word of three letters or more, once normalized.
|
|
|
checkOf("an empty word", "es", 2048, at(5, "")),
|
|
|
checkOf("a word of three letters", "es", 2048, at(5, "pal")),
|
|
|
checkOf("\u00f1u, two letters", "es", 2048, at(5, "\u00f1u")),
|
|
|
checkOf("\u00f1u\u00f1, three letters", "es", 2048, at(5, "\u00f1u\u00f1")),
|
|
|
checkOf("a space before", "es", 2048, at(5, " palaaf")),
|
|
|
checkOf("a space after", "es", 2048, at(5, "palaaf ")),
|
|
|
checkOf("a tab inside", "es", 2048, at(5, "pal\taf")),
|
|
|
checkOf("NBSP inside", "es", 2048, at(5, "pal\u00a0af")),
|
|
|
checkOf("NEL inside", "es", 2048, at(5, "pal\u0085af")),
|
|
|
checkOf("the ideographic space inside", "es", 2048, at(5, "pal\u3000af")),
|
|
|
checkOf("LS inside", "es", 2048, at(5, "pal\u2028af")),
|
|
|
checkOf("an LF inside", "es", 2048, at(5, "pal\naf")),
|
|
|
checkOf("two runes and ZWSP: the count before the runes", "es", 2048, at(5, "p\u200b")),
|
|
|
checkOf("three letters with a mark that goes: two", "es", 2048, at(5, "pa\u0301")),
|
|
|
checkOf("a\u0301b, two letters once normalized", "es", 2048, at(5, "a\u0301b")),
|
|
|
|
|
|
// The runes of the normalized word.
|
|
|
checkOf("a control", "es", 2048, at(5, "pal\x01af")),
|
|
|
checkOf("DEL", "es", 2048, at(5, "pal\x7faf")),
|
|
|
checkOf("a C1 control", "es", 2048, at(5, "pal\u0080af")),
|
|
|
checkOf("a control and a capital: the runes before the alphabet", "es", 2048, at(5, "Pal\x01af")),
|
|
|
checkOf("a soft hyphen", "es", 2048, at(5, "pa\u00ad")),
|
|
|
checkOf("ZWJ", "es", 2048, at(5, "pal\u200daf")),
|
|
|
checkOf("a byte order mark", "es", 2048, at(5, "\ufeffpalaaf")),
|
|
|
checkOf("an unassigned code point", "es", 2048, at(5, "pal\u0378")),
|
|
|
checkOf("a noncharacter", "es", 2048, at(5, "pal\ufffe")),
|
|
|
checkOf("a tag", "es", 2048, at(5, "pal\U000e0041af")),
|
|
|
|
|
|
// The alphabet, on the word as the list writes it.
|
|
|
checkOf("the letters of the alphabet", "es", 2048, at(5, "abcdefghijklmnopqrstuvwxyz\u00e1\u00e9\u00ed\u00f3\u00fa\u00fc\u00f1")),
|
|
|
checkOf("an acute accent in NFD", "es", 2048, at(5, "pala\u0301e")),
|
|
|
checkOf("a diaeresis in NFD", "es", 2048, at(5, "palaa\u0308")),
|
|
|
checkOf("a tilde in NFD", "es", 2048, at(5, "pan\u0303o")),
|
|
|
checkOf("\u00c1, a capital with an accent", "es", 2048, at(5, "\u00c1baco")),
|
|
|
checkOf("\u00d1, a capital", "es", 2048, at(5, "\u00d1and\u00fa")),
|
|
|
checkOf("a grave accent", "es", 2048, at(5, "pal\u00e0a")),
|
|
|
checkOf("a circumflex", "es", 2048, at(5, "pal\u00e2a")),
|
|
|
checkOf("\u00e4", "es", 2048, at(5, "pal\u00e4a")),
|
|
|
checkOf("\u00e7", "es", 2048, at(5, "pal\u00e7a")),
|
|
|
checkOf("\u00f6", "es", 2048, at(5, "pal\u00f6a")),
|
|
|
checkOf("\u00fd", "es", 2048, at(5, "pal\u00fda")),
|
|
|
checkOf("\u00df", "es", 2048, at(5, "pal\u00dfa")),
|
|
|
checkOf("a Cyrillic a", "es", 2048, at(5, "pal\u0430a")),
|
|
|
checkOf("a Greek omicron", "es", 2048, at(5, "pal\u03bfa")),
|
|
|
checkOf("a full-width a", "es", 2048, at(5, "pal\uff41a")),
|
|
|
checkOf("a mathematical bold a", "es", 2048, at(5, "pal\U0001d41aa")),
|
|
|
checkOf("the Kelvin sign", "es", 2048, at(5, "pal\u212aa")),
|
|
|
checkOf("a dotless i", "es", 2048, at(5, "pal\u0131a")),
|
|
|
checkOf("a hyphen", "es", 2048, at(5, "pal-af")),
|
|
|
checkOf("an apostrophe", "es", 2048, at(5, "pal'af")),
|
|
|
checkOf("a backtick and a brace, around a to z", "es", 2048, at(5, "pal`{")),
|
|
|
checkOf("an emoji", "es", 2048, at(5, "pal\U0001f600")),
|
|
|
|
|
|
// Bytes that are not valid UTF-8: U+FFFD, which no alphabet holds.
|
|
|
checkOf("a byte that is not valid UTF-8", "es", 2048, at(5, "pal\xffaa")),
|
|
|
checkOf("a surrogate in UTF-8 bytes", "es", 2048, at(5, "\xed\xa0\x80aaa")),
|
|
|
checkOf("a sequence cut short", "es", 2048, at(5, "pala\xc3")),
|
|
|
checkOf("an overlong NUL", "es", 2048, at(5, "\xc0\x80aaa")),
|
|
|
checkOf("U+FFFD itself", "es", 2048, at(5, "pal\ufffdaa")),
|
|
|
|
|
|
// The same word once normalized, and the order across words.
|
|
|
checkOf("a\u00f1o and ano", "es", 2048, []edit{set(5, "a\u00f1o"), set(6, "ano")}),
|
|
|
checkOf("ping\u00fcino and pinguino", "es", 2048, []edit{set(5, "pinguino"), set(9, "ping\u00fcino")}),
|
|
|
checkOf("the same word, the first of the list last", "es", 2048, nil, "palaaa"),
|
|
|
checkOf("a capital of a word of the list: the alphabet before the same word", "es", 2048, at(5, "PALAAE")),
|
|
|
checkOf("an error at line 4 and another at line 6", "es", 2048, []edit{set(3, "Mal"), set(5, "pa")}),
|
|
|
checkOf("an error at the last line", "es", 2048, at(2047, "x")),
|
|
|
checkOf("an empty last line", "es", 2048, nil, ""),
|
|
|
}
|
|
|
return out
|
|
|
}
|
|
|
|
|
|
// ---------------------------------------------------------------------------
|
|
|
// The alphabet of every code point
|
|
|
|
|
|
type runeResult struct {
|
|
|
Rune int `json:"rune"`
|
|
|
Result string `json:"result"`
|
|
|
}
|
|
|
|
|
|
type alphabetHead struct {
|
|
|
Word string `json:"word"`
|
|
|
Planes []int `json:"planes"`
|
|
|
Count int `json:"count"`
|
|
|
Sha256 string `json:"sha256"`
|
|
|
Ok []string `json:"ok"`
|
|
|
}
|
|
|
|
|
|
func alphabet() (alphabetHead, []runeResult) {
|
|
|
planes := []int{0, 1, 14}
|
|
|
l := edited(2048, nil)
|
|
|
digest := sha256.New()
|
|
|
var results []runeResult
|
|
|
var ok []string
|
|
|
count := 0
|
|
|
for _, p := range planes {
|
|
|
for r := rune(p << 16); r <= rune(p<<16|0xffff); r++ {
|
|
|
if r >= 0xd800 && r <= 0xdfff {
|
|
|
continue
|
|
|
}
|
|
|
l[0] = "pala" + string(r) + "zz"
|
|
|
res := result(wordkey.CheckList("es", l))
|
|
|
fmt.Fprintf(digest, "U+%04X %s\n", r, res)
|
|
|
count++
|
|
|
if r <= 0x17f {
|
|
|
results = append(results, runeResult{int(r), res})
|
|
|
}
|
|
|
if res == "ok" {
|
|
|
ok = append(ok, string(r))
|
|
|
}
|
|
|
}
|
|
|
}
|
|
|
head := alphabetHead{"pala, the code point and zz, the first word of the base list of 2048", planes, count, hex.EncodeToString(digest.Sum(nil)), ok}
|
|
|
return head, results
|
|
|
}
|
|
|
|
|
|
// ---------------------------------------------------------------------------
|
|
|
// Generate
|
|
|
|
|
|
// seeded is the keystream of SeededRandomSource of lib/src/random.dart:
|
|
|
// ChaCha20 under SHA-256(seed), with a zero nonce, from block 0.
|
|
|
type seeded struct{ c *chacha20.Cipher }
|
|
|
|
|
|
func newSeeded(seed string) *seeded {
|
|
|
key := sha256.Sum256([]byte(seed))
|
|
|
c, err := chacha20.NewUnauthenticatedCipher(key[:], make([]byte, chacha20.NonceSize))
|
|
|
check(err)
|
|
|
return &seeded{c}
|
|
|
}
|
|
|
|
|
|
func (s *seeded) Read(p []byte) (int, error) {
|
|
|
clear(p)
|
|
|
s.c.XORKeyStream(p, p)
|
|
|
return len(p), nil
|
|
|
}
|
|
|
|
|
|
// counter is SHA-256(label ‖ i) for i = 0, 1, …, a 32-bit big-endian
|
|
|
// counter, one block after the other.
|
|
|
type counter struct {
|
|
|
label []byte
|
|
|
i uint32
|
|
|
block []byte
|
|
|
}
|
|
|
|
|
|
func (c *counter) Read(p []byte) (int, error) {
|
|
|
for n := 0; n < len(p); {
|
|
|
if len(c.block) == 0 {
|
|
|
b := sha256.Sum256(binary.BigEndian.AppendUint32(bytes.Clone(c.label), c.i))
|
|
|
c.block = b[:]
|
|
|
c.i++
|
|
|
}
|
|
|
k := copy(p[n:], c.block)
|
|
|
c.block = c.block[k:]
|
|
|
n += k
|
|
|
}
|
|
|
return len(p), nil
|
|
|
}
|
|
|
|
|
|
// counting counts the bytes read from r.
|
|
|
type counting struct {
|
|
|
r io.Reader
|
|
|
n int
|
|
|
}
|
|
|
|
|
|
func (c *counting) Read(p []byte) (int, error) {
|
|
|
n, err := c.r.Read(p)
|
|
|
c.n += n
|
|
|
return n, err
|
|
|
}
|
|
|
|
|
|
// stream is what a case of generate reads instead of crypto/rand: kind
|
|
|
// seeded, the keystream of Seed; counter, the blocks of the label Seed; or
|
|
|
// bytes, Hex repeated Repeat times, which end.
|
|
|
type stream struct {
|
|
|
Kind string `json:"kind"`
|
|
|
Seed string `json:"seed,omitempty"`
|
|
|
Hex string `json:"hex,omitempty"`
|
|
|
Repeat int `json:"repeat,omitempty"`
|
|
|
}
|
|
|
|
|
|
func (s stream) reader() io.Reader {
|
|
|
switch s.Kind {
|
|
|
case "seeded":
|
|
|
return newSeeded(s.Seed)
|
|
|
case "counter":
|
|
|
return &counter{label: []byte(s.Seed)}
|
|
|
case "bytes":
|
|
|
b, err := hex.DecodeString(s.Hex)
|
|
|
check(err)
|
|
|
return bytes.NewReader(bytes.Repeat(b, s.Repeat))
|
|
|
}
|
|
|
log.Fatalf("stream kind %q", s.Kind)
|
|
|
return nil
|
|
|
}
|
|
|
|
|
|
func seed(s string) stream { return stream{Kind: "seeded", Seed: s} }
|
|
|
|
|
|
func given(hexBytes string, repeat int) stream {
|
|
|
return stream{Kind: "bytes", Hex: hexBytes, Repeat: repeat}
|
|
|
}
|
|
|
|
|
|
type generateCase struct {
|
|
|
Name string `json:"name"`
|
|
|
List string `json:"list,omitempty"`
|
|
|
Size int `json:"size"`
|
|
|
N int `json:"n"`
|
|
|
Stream stream `json:"stream"`
|
|
|
Indices []int `json:"indices,omitempty"`
|
|
|
Words []string `json:"words,omitempty"`
|
|
|
Read int `json:"read"`
|
|
|
Next string `json:"next,omitempty"`
|
|
|
Error string `json:"error,omitempty"`
|
|
|
}
|
|
|
|
|
|
// numbers is a list of size words, "0", "1", …: Generate reads only its
|
|
|
// length and the words it draws, whose indices they are.
|
|
|
func numbers(size int) []string {
|
|
|
l := make([]string, size)
|
|
|
for i := range l {
|
|
|
l[i] = strconv.Itoa(i)
|
|
|
}
|
|
|
return l
|
|
|
}
|
|
|
|
|
|
func generateOf(name, listName string, list []string, n int, s stream) generateCase {
|
|
|
r := &counting{r: s.reader()}
|
|
|
words, err := wordkey.Generate(list, n, r)
|
|
|
c := generateCase{Name: name, List: listName, Size: len(list), N: n, Stream: s, Read: r.n}
|
|
|
if err != nil {
|
|
|
c.Error = err.Error()
|
|
|
return c
|
|
|
}
|
|
|
index := make(map[string]int, len(list))
|
|
|
for i, w := range list {
|
|
|
index[w] = i
|
|
|
}
|
|
|
for _, w := range words {
|
|
|
c.Indices = append(c.Indices, index[w])
|
|
|
}
|
|
|
if listName != "" && n <= 24 {
|
|
|
c.Words = words
|
|
|
}
|
|
|
if s.Kind != "bytes" {
|
|
|
next := make([]byte, 4)
|
|
|
_, err := io.ReadFull(r.r, next)
|
|
|
check(err)
|
|
|
c.Next = hex.EncodeToString(next)
|
|
|
}
|
|
|
return c
|
|
|
}
|
|
|
|
|
|
func generateCases() []generateCase {
|
|
|
es, err := wordkey.List("es")
|
|
|
check(err)
|
|
|
var out []generateCase
|
|
|
add := func(c generateCase) { out = append(out, c) }
|
|
|
// The Spanish list, from the keystream of a seed and from SHA-256
|
|
|
// counter blocks.
|
|
|
for _, n := range []int{6, wordkey.DefaultCount, 8, 12, 24} {
|
|
|
add(generateOf(fmt.Sprintf("%d words of the Spanish list", n), "es", es, n, seed(fmt.Sprintf("wordkey.Generate es %d", n))))
|
|
|
}
|
|
|
for _, n := range []int{6, wordkey.DefaultCount, 12} {
|
|
|
add(generateOf(fmt.Sprintf("%d words of the Spanish list, SHA-256 counter blocks", n), "es", es, n, stream{Kind: "counter", Seed: fmt.Sprintf("wordkey.Generate counter %d", n)}))
|
|
|
}
|
|
|
add(generateOf("3888 words of the Spanish list, the most", "es", es, len(es)/2, seed("wordkey.Generate es 3888")))
|
|
|
// Lists of every number of bytes of a draw of crypto/rand.Int, with and
|
|
|
// without draws again.
|
|
|
for _, size := range []int{12, 13, 16, 17, 100, 255, 256, 257, 2047, 2048, 2049, 4096, 7775, 7776, 7777, 8192, 65535, 65536, 65537, 100000, 1<<20 + 1} {
|
|
|
add(generateOf(fmt.Sprintf("6 words of %d", size), "", numbers(size), 6, seed(fmt.Sprintf("wordkey.Generate %d", size))))
|
|
|
}
|
|
|
add(generateOf("12 words of 65537, SHA-256 counter blocks", "", numbers(65537), 12, stream{Kind: "counter", Seed: "wordkey.Generate counter 65537"}))
|
|
|
// The most words of small lists: indices drawn twice, again and again.
|
|
|
for _, size := range []int{12, 13, 17, 100} {
|
|
|
add(generateOf(fmt.Sprintf("%d words of %d, the most", size/2, size), "", numbers(size), size/2, seed(fmt.Sprintf("wordkey.Generate %d most", size))))
|
|
|
}
|
|
|
// Bytes that end.
|
|
|
add(generateOf("the seed of TestGenerate: two indices only, then EOF", "es", es, 6, given("0701c821", 64)))
|
|
|
add(generateOf("twelve bytes, six indices", "es", es, 6, given("000000010002000300040005", 1)))
|
|
|
add(generateOf("8191 and 8032 drawn again, 0 drawn twice, 7775 the last index", "es", es, 6, given("1fff00001f601e5f00000001000200030004", 1)))
|
|
|
add(generateOf("the three bits above the 13 of 7775 are cleared", "es", es, 6, given("e000ffff21002200230024002500", 1)))
|
|
|
add(generateOf("the bytes end inside a draw", "es", es, 6, given("0000000100", 1)))
|
|
|
add(generateOf("no bytes", "es", es, 6, given("", 0)))
|
|
|
// The arguments, refused before anything is read.
|
|
|
for _, n := range []int{5, 0, -1, len(es)/2 + 1, len(es)} {
|
|
|
add(generateOf(fmt.Sprintf("%d words of the Spanish list", n), "es", es, n, seed("never read")))
|
|
|
}
|
|
|
for _, c := range []struct{ size, n int }{{11, 6}, {0, 6}, {0, 5}, {12, 7}, {13, 7}} {
|
|
|
add(generateOf(fmt.Sprintf("%d words of %d", c.n, c.size), "", numbers(c.size), c.n, seed("never read")))
|
|
|
}
|
|
|
return out
|
|
|
}
|
|
|
|
|
|
// ---------------------------------------------------------------------------
|
|
|
// Bits
|
|
|
|
|
|
type bitsCase struct {
|
|
|
Size int `json:"size"`
|
|
|
Count int `json:"count"`
|
|
|
Bits string `json:"bits"`
|
|
|
Hex string `json:"hex"`
|
|
|
}
|
|
|
|
|
|
func bitsOf(size, count int) bitsCase {
|
|
|
b := wordkey.Bits(size, count)
|
|
|
return bitsCase{size, count, strconv.FormatFloat(b, 'g', -1, 64), fmt.Sprintf("%016x", math.Float64bits(b))}
|
|
|
}
|
|
|
|
|
|
func bitsCases() []bitsCase {
|
|
|
var out []bitsCase
|
|
|
for _, size := range []int{1, 2, 3, 5, 12, 13, 100, 255, 256, 257, 2047, 2048, 2049, 4096, 7775, 7776, 7777, 8192, 10000, 65535, 65536, 65537, 1 << 20, 1<<31 - 1, 1 << 31, 1<<32 - 1, 1 << 32, 1<<32 + 1, 1<<52 + 1, 1<<53 - 1, 1 << 53} {
|
|
|
for _, count := range []int{0, 1, 2, 6, 7, 8, 12} {
|
|
|
out = append(out, bitsOf(size, count))
|
|
|
}
|
|
|
}
|
|
|
// Go adds log2 of 0, −Inf, and of a negative number, NaN, and adds
|
|
|
// nothing for a negative count.
|
|
|
for _, c := range [][2]int{{0, 1}, {-1, 1}, {7776, -1}, {0, 0}} {
|
|
|
out = append(out, bitsOf(c[0], c[1]))
|
|
|
}
|
|
|
return out
|
|
|
}
|
|
|
|
|
|
type bitsDigest struct {
|
|
|
Name string `json:"name"`
|
|
|
Sizes [2]int `json:"sizes"`
|
|
|
Counts [2]int `json:"counts"`
|
|
|
Sha256 string `json:"sha256"`
|
|
|
}
|
|
|
|
|
|
func digestOf(name string, sizes, counts [2]int) bitsDigest {
|
|
|
d := sha256.New()
|
|
|
var b [8]byte
|
|
|
for size := sizes[0]; size <= sizes[1]; size++ {
|
|
|
for count := counts[0]; count <= counts[1]; count++ {
|
|
|
binary.BigEndian.PutUint64(b[:], math.Float64bits(wordkey.Bits(size, count)))
|
|
|
d.Write(b[:])
|
|
|
}
|
|
|
}
|
|
|
return bitsDigest{name, sizes, counts, hex.EncodeToString(d.Sum(nil))}
|
|
|
}
|
|
|
|
|
|
func bitsDigests() []bitsDigest {
|
|
|
return []bitsDigest{
|
|
|
digestOf("one word of 1 to 65536: the log2 of Go", [2]int{1, 65536}, [2]int{1, 1}),
|
|
|
digestOf("6 to 8 words of 2048 to 10000", [2]int{2048, 10000}, [2]int{6, 8}),
|
|
|
digestOf("0 to 600 words of 7776", [2]int{7776, 7776}, [2]int{0, 600}),
|
|
|
digestOf("1 word of 2^32 - 4096 to 2^32 + 4096", [2]int{1<<32 - 4096, 1<<32 + 4096}, [2]int{1, 1}),
|
|
|
}
|
|
|
}
|
|
|
|
|
|
// ---------------------------------------------------------------------------
|
|
|
// JSON
|
|
|
|
|
|
// ascii escapes every code point above U+007F of JSON as \uXXXX, in UTF-16
|
|
|
// for those above U+FFFF: the same JSON, in ASCII.
|
|
|
func ascii(b []byte) []byte {
|
|
|
var out bytes.Buffer
|
|
|
for _, r := range string(b) {
|
|
|
switch {
|
|
|
case r < 0x80:
|
|
|
out.WriteByte(byte(r))
|
|
|
case r < 0x10000:
|
|
|
fmt.Fprintf(&out, `\u%04x`, r)
|
|
|
default:
|
|
|
r -= 0x10000
|
|
|
fmt.Fprintf(&out, `\u%04x\u%04x`, 0xd800+(r>>10), 0xdc00+(r&0x3ff))
|
|
|
}
|
|
|
}
|
|
|
return out.Bytes()
|
|
|
}
|
|
|
|
|
|
func marshal(v any) []byte {
|
|
|
var b bytes.Buffer
|
|
|
enc := json.NewEncoder(&b)
|
|
|
enc.SetEscapeHTML(false)
|
|
|
check(enc.Encode(v))
|
|
|
return ascii(bytes.TrimSuffix(b.Bytes(), []byte("\n")))
|
|
|
}
|
|
|
|
|
|
// list writes the items of a JSON array one per line.
|
|
|
func list[T any](w *bytes.Buffer, key string, items []T, last bool) {
|
|
|
fmt.Fprintf(w, " %q: [\n", key)
|
|
|
for i, it := range items {
|
|
|
w.WriteString(" ")
|
|
|
w.Write(marshal(it))
|
|
|
if i < len(items)-1 {
|
|
|
w.WriteByte(',')
|
|
|
}
|
|
|
w.WriteByte('\n')
|
|
|
}
|
|
|
w.WriteString(" ]")
|
|
|
if !last {
|
|
|
w.WriteByte(',')
|
|
|
}
|
|
|
w.WriteByte('\n')
|
|
|
}
|
|
|
|
|
|
func main() {
|
|
|
outDir := flag.String("out", "", "the directory test/vectors of datekeys-dart")
|
|
|
source := flag.String("source", "", "the commit of datekeys-go")
|
|
|
flag.Parse()
|
|
|
if *outDir == "" || *source == "" {
|
|
|
log.Fatal("usage: go run wordlist_go_vectors.go -source COMMIT -out DIR")
|
|
|
}
|
|
|
lists := listInfos()
|
|
|
parse := parseCases()
|
|
|
checks := checkCases()
|
|
|
alpha, runes := alphabet()
|
|
|
gen := generateCases()
|
|
|
bits := bitsCases()
|
|
|
digests := bitsDigests()
|
|
|
|
|
|
var w bytes.Buffer
|
|
|
w.WriteString("{\n")
|
|
|
head := []struct {
|
|
|
key string
|
|
|
v any
|
|
|
}{
|
|
|
{"description", "The word lists of package wordkey of the Go reference (generate.go, spec \u00a738.1) for lib/src/wordlist.dart; see tool/wordlist_go_vectors.go. " +
|
|
|
"Binary values and words are lower-case hex. The base list is word i = pal and three letters, a + i/676, a + i/26%26 and a + i%26; a case of size n takes its first n words, replaces the word at each at of set, and adds those of append. " +
|
|
|
"lists: the SHA-256, bytes and words of each built-in list. parse: the body of List on the text prefix, the base list joined by sep, and suffix (sha256 is that of the text): the words read, and ok or the error. " +
|
|
|
"check_list: CheckList, ok or the error (sha256 is that of the words joined by LF). alphabet: CheckList of the base list of 2048 whose first word is pala, a code point and zz, for each code point of the planes but the surrogates: the SHA-256 of the lines U+XXXX result, the code points of the words it accepts, and alphabet_results, each result up to U+017F. " +
|
|
|
"generate: Generate on the list of the Spanish words (list es) or on a list of size words whose indices they are, reading a stream: seeded, the keystream of SeededRandomSource of the seed; counter, SHA-256 of the label and a 32-bit big-endian counter from 0, block after block; bytes, hex repeated repeat times, which end. The indices drawn, the bytes read and the next four of an endless stream, or the error. " +
|
|
|
"bits: Bits, its shortest decimal and the hex of its float64. bits_digests: the SHA-256 of the 8 big-endian bytes of each Bits of the sizes and counts, inclusive, sizes outer."},
|
|
|
{"generator", "tool/wordlist_go_vectors.go, " + runtime.Version()},
|
|
|
{"source", *source},
|
|
|
{"default_count", wordkey.DefaultCount},
|
|
|
{"min_list_size", wordkey.MinListSize},
|
|
|
{"min_words", wordkey.MinWords},
|
|
|
{"min_letters", wordkey.MinLetters},
|
|
|
{"alphabet", alpha},
|
|
|
}
|
|
|
for _, kv := range head {
|
|
|
fmt.Fprintf(&w, " %q: %s,\n", kv.key, marshal(kv.v))
|
|
|
}
|
|
|
list(&w, "lists", lists, false)
|
|
|
list(&w, "parse", parse, false)
|
|
|
list(&w, "check_list", checks, false)
|
|
|
list(&w, "alphabet_results", runes, false)
|
|
|
list(&w, "generate", gen, false)
|
|
|
list(&w, "bits", bits, false)
|
|
|
list(&w, "bits_digests", digests, true)
|
|
|
w.WriteString("}\n")
|
|
|
|
|
|
var parsed any
|
|
|
if err := json.Unmarshal(w.Bytes(), &parsed); err != nil {
|
|
|
log.Fatalf("the JSON does not parse: %v", err)
|
|
|
}
|
|
|
if bytes.Contains(w.Bytes(), []byte("'''")) {
|
|
|
log.Fatal("the JSON holds three quotes")
|
|
|
}
|
|
|
path := filepath.Join(*outDir, "wordlist_vectors.json")
|
|
|
check(os.WriteFile(path, w.Bytes(), 0o644))
|
|
|
fmt.Printf("wrote %s, %d bytes: %d lists, %d texts, %d lists checked, %d code points, %d draws, %d bits, %d digests\n",
|
|
|
path, w.Len(), len(lists), len(parse), len(checks), alpha.Count, len(gen), len(bits), len(digests))
|
|
|
dart := "// Generated by tool/wordlist_go_vectors.go from wordlist_vectors.json, for\n" +
|
|
|
"// the tests that also run compiled to JavaScript, where no file can be\n" +
|
|
|
"// read. Do not edit.\n\n" +
|
|
|
"/// The text of test/vectors/wordlist_vectors.json.\n" +
|
|
|
"const wordlistVectorsJson = r'''\n" + w.String() + "''';\n"
|
|
|
dpath := filepath.Join(*outDir, "wordlist_vectors.g.dart")
|
|
|
check(os.WriteFile(dpath, []byte(dart), 0o644))
|
|
|
fmt.Printf("wrote %s\n", dpath)
|
|
|
}
|