You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
dateKeys-dart/tool/wordlist_go_vectors.go

728 lines
28 KiB

This file contains ambiguous Unicode characters!

This file contains ambiguous Unicode characters that may be confused with others in your current locale. If your use case is intentional and legitimate, you can safely ignore this warning. Use the Escape button to highlight these characters.

//go:build ignore
// Writes test/vectors/wordlist_vectors.json and the same JSON as a Dart
// constant, wordlist_vectors.g.dart: the results of the word lists of
// package wordkey of the Go reference (generate.go: List, CheckList,
// Generate and Bits, the random words of spec §38.1) for
// lib/src/wordlist.dart. Every expected value is computed here by the Go
// reference; none is written by hand.
//
// - lists: the built-in lists, with the SHA-256, the size and the number
// of words of each file of wordkey/lists, which List reads.
// - parse: the body of List, restated because List reads only its
// built-in lists, on texts of the base list of TestCheckList, "pal" and
// three letters, joined by a separator, with a prefix, a suffix and
// edits: with and without the final LF, CR LF lines, an empty last
// line, a byte order mark, an empty text, bytes that are not valid
// UTF-8. The restatement gives the words of List on every built-in list.
// - check_list: CheckList on the base list of 0 to 17 576 words, edited:
// the cases of TestCheckList; languages without an alphabet; white
// space, controls, invisible and unassigned code points; letters of
// other scripts that look like those of the alphabet, and marks; bytes
// that are not valid UTF-8; words that are one once normalized; and the
// order of the checks within a word and across words.
// - alphabet: CheckList of the base list of 2048 words whose first word
// is "pala", one code point and "zz": the result of every code point up
// to U+017F, and the SHA-256 of the lines "U+XXXX result\n" of every
// code point of planes 0, 1 and 14 but the surrogates.
// - generate: Generate on the Spanish list and on lists of 0 to
// 2^20 + 1 words, while it reads fixed streams instead of crypto/rand:
// the keystream of SeededRandomSource of lib/src/random.dart (ChaCha20
// under SHA-256(seed), zero nonce); SHA-256 counter blocks; and bytes
// that end, the seed of TestGenerate among them. Each case has the
// indices drawn (and the words, from the Spanish list), the bytes read
// and the next four of an endless stream; or the text of the error and
// the bytes read.
// - bits: Bits of sizes and counts, as the 64 bits of the float64 and its
// shortest decimal; and bits_digests, the SHA-256 of the 8 big-endian
// bytes of each Bits over ranges of sizes and counts, sizes outer.
//
// Binary values are lower-case hexadecimal, and the JSON is ASCII, so that
// the Dart constant is too. The output is the same on every run. It imports
// the package wordkey and reads wordkey/lists, so it runs in an export of
// datekeys-go made with git archive, without changing the repository, at
// e671032, the branch v0.15 after the tag spec-v0.15:
//
// commit=$(git -C ../datekeys-go rev-parse e671032)
// out=$PWD/test/vectors
// tmp=$(mktemp -d)
// git -C ../datekeys-go archive "$commit" | tar -x -C "$tmp"
// cp tool/wordlist_go_vectors.go "$tmp"
// (cd "$tmp" && go run ./wordlist_go_vectors.go -source "$commit" -out "$out")
// rm -rf "$tmp"
package main
import (
"bytes"
"crypto/sha256"
"encoding/binary"
"encoding/hex"
"encoding/json"
"flag"
"fmt"
"io"
"log"
"math"
"os"
"path/filepath"
"runtime"
"strconv"
"strings"
"golang.org/x/crypto/chacha20"
"g.activething.com/go/DateKeys/wordkey"
)
func check(err error) {
if err != nil {
log.Fatal(err)
}
}
func h(s string) string { return hex.EncodeToString([]byte(s)) }
func hexAll(ss []string) []string {
var out []string
for _, s := range ss {
out = append(out, h(s))
}
return out
}
func sum(s string) string {
b := sha256.Sum256([]byte(s))
return hex.EncodeToString(b[:])
}
// result is "ok", or the text of err.
func result(err error) string {
if err != nil {
return err.Error()
}
return "ok"
}
// ---------------------------------------------------------------------------
// Lists
// baseWord is word i of the base list of TestCheckList, "pal" and three
// letters: palaaa, palaab, …, up to i = 17 575, palzzz.
func baseWord(i int) string {
return "pal" + string([]rune{'a' + rune(i/676), 'a' + rune(i/26%26), 'a' + rune(i%26)})
}
// edit is the word at the index At of a base list replaced by Word, in hex.
type edit struct {
At int `json:"at"`
Word string `json:"word"`
}
func set(at int, word string) edit { return edit{at, h(word)} }
// edited is the base list of size words with the edits.
func edited(size int, sets []edit) []string {
l := make([]string, size)
for i := range l {
l[i] = baseWord(i)
}
for _, e := range sets {
b, err := hex.DecodeString(e.Word)
check(err)
l[e.At] = string(b)
}
return l
}
// listOf is the body of wordkey.List for the text of a list of the caller,
// restated because List reads only its built-in lists.
func listOf(lang, text string) ([]string, error) {
words := strings.Split(strings.TrimSuffix(text, "\n"), "\n")
if err := wordkey.CheckList(lang, words); err != nil {
return nil, fmt.Errorf("wordkey: the list %q: %w", lang, err)
}
return words, nil
}
type listInfo struct {
Lang string `json:"lang"`
Sha256 string `json:"sha256"`
Bytes int `json:"bytes"`
Words int `json:"words"`
}
func listInfos() []listInfo {
var out []listInfo
for _, lang := range wordkey.Languages() {
text, err := os.ReadFile(filepath.Join("wordkey", "lists", lang+".txt"))
check(err)
words, err := wordkey.List(lang)
check(err)
again, err := listOf(lang, string(text))
check(err)
if strings.Join(again, "\n") != strings.Join(words, "\n") {
log.Fatalf("wordkey/lists/%s.txt is not the list of List, or listOf is not its body", lang)
}
out = append(out, listInfo{lang, sum(string(text)), len(text), len(words)})
}
return out
}
type parseCase struct {
Name string `json:"name"`
Lang string `json:"lang"`
Size int `json:"size"`
Set []edit `json:"set,omitempty"`
Prefix string `json:"prefix"`
Sep string `json:"sep"`
Suffix string `json:"suffix"`
Sha256 string `json:"sha256"`
Words int `json:"words"`
Result string `json:"result"`
}
// parseOf reads with listOf the text prefix, the edited base list joined
// by sep, and suffix.
func parseOf(name, lang string, size int, sets []edit, prefix, sep, suffix string) parseCase {
text := prefix + strings.Join(edited(size, sets), sep) + suffix
words, err := listOf(lang, text)
return parseCase{name, lang, size, sets, h(prefix), h(sep), h(suffix), sum(text), len(words), result(err)}
}
func parseCases() []parseCase {
return []parseCase{
parseOf("2048 words, one per line, with a final LF", "es", 2048, nil, "", "\n", "\n"),
parseOf("without the final LF", "es", 2048, nil, "", "\n", ""),
parseOf("7776 words", "es", 7776, nil, "", "\n", "\n"),
parseOf("an empty last line: two final LF", "es", 2048, nil, "", "\n", "\n\n"),
parseOf("an empty first line", "es", 2048, nil, "\n", "\n", "\n"),
parseOf("CR LF lines", "es", 2048, nil, "", "\r\n", "\r\n"),
parseOf("CR LF lines, the last without", "es", 2048, nil, "", "\r\n", ""),
parseOf("CR lines", "es", 2048, nil, "", "\r", "\r"),
parseOf("a byte order mark", "es", 2048, nil, "\ufeff", "\n", "\n"),
parseOf("an empty text", "es", 0, nil, "", "\n", ""),
parseOf("an LF only", "es", 0, nil, "", "\n", "\n"),
parseOf("2047 words", "es", 2047, nil, "", "\n", "\n"),
parseOf("the words separated by spaces", "es", 2048, nil, "", " ", "\n"),
parseOf("a language without an alphabet", "xx", 2048, nil, "", "\n", "\n"),
parseOf("a byte that is not valid UTF-8", "es", 2048, []edit{set(6, "pal\xffaa")}, "", "\n", "\n"),
parseOf("a sequence cut at the end of the text", "es", 2048, []edit{set(2047, "palzz\xc3")}, "", "\n", ""),
parseOf("the same word twice", "es", 2048, []edit{set(2000, "palaaa")}, "", "\n", "\n"),
parseOf("a capital letter", "es", 2048, []edit{set(100, "Paldww")}, "", "\n", "\n"),
parseOf("letters of the alphabet", "es", 2048, []edit{set(0, "\u00e1\u00e9\u00ed\u00f3\u00fa\u00fc\u00f1")}, "", "\n", "\n"),
}
}
type checkCase struct {
Name string `json:"name"`
Lang string `json:"lang"`
Size int `json:"size"`
Set []edit `json:"set,omitempty"`
Append []string `json:"append,omitempty"`
Sha256 string `json:"sha256"`
Result string `json:"result"`
}
// checkOf runs CheckList on the edited base list of size words followed by
// appends. Sha256 is that of the words joined by LF.
func checkOf(name, lang string, size int, sets []edit, appends ...string) checkCase {
l := append(edited(size, sets), appends...)
return checkCase{name, lang, size, sets, hexAll(appends), sum(strings.Join(l, "\n")), result(wordkey.CheckList(lang, l))}
}
func checkCases() []checkCase {
at := func(i int, w string) []edit { return []edit{set(i, w)} }
out := []checkCase{
// TestCheckList, in its order.
checkOf("the base list", "es", 2048, nil),
checkOf("a language without an alphabet", "xx", 2048, nil),
checkOf("2047 words", "es", 2047, nil),
checkOf("two words in a line", "es", 2048, at(5, "dos palabras")),
checkOf("two spaces", "es", 2048, at(5, " ")),
checkOf("two letters", "es", 2048, at(5, "mi")),
checkOf("ZWSP", "es", 2048, at(5, "casa\u200b")),
checkOf("a capital", "es", 2048, at(5, "Palaaf")),
checkOf("a Cyrillic letter that looks like a Latin c", "es", 2048, at(5, "\u0441asa")),
checkOf("a digit", "es", 2048, at(5, "pal1")),
checkOf("the CR of a CR LF line", "es", 2048, at(5, "palaaf\r")),
checkOf("the same word", "es", 2048, at(5, baseWord(4))),
checkOf("the same word once normalized", "es", 2048, at(5, "pala\u00e1e")),
checkOf("papa after pap\u00e1", "es", 2048, at(0, "pap\u00e1"), "papa"),
// The language and the size.
checkOf("no words", "es", 0, nil),
checkOf("no words in a language without an alphabet", "xx", 0, nil),
checkOf("the empty language", "", 2048, nil),
checkOf("ES, in capitals", "ES", 2048, nil),
checkOf("es and a space", "es ", 2048, nil),
checkOf("espa\u00f1ol", "espa\u00f1ol", 2048, nil),
checkOf("2049 words", "es", 2049, nil),
checkOf("7776 words", "es", 7776, nil),
checkOf("17576 words", "es", 17576, nil),
// One word of three letters or more, once normalized.
checkOf("an empty word", "es", 2048, at(5, "")),
checkOf("a word of three letters", "es", 2048, at(5, "pal")),
checkOf("\u00f1u, two letters", "es", 2048, at(5, "\u00f1u")),
checkOf("\u00f1u\u00f1, three letters", "es", 2048, at(5, "\u00f1u\u00f1")),
checkOf("a space before", "es", 2048, at(5, " palaaf")),
checkOf("a space after", "es", 2048, at(5, "palaaf ")),
checkOf("a tab inside", "es", 2048, at(5, "pal\taf")),
checkOf("NBSP inside", "es", 2048, at(5, "pal\u00a0af")),
checkOf("NEL inside", "es", 2048, at(5, "pal\u0085af")),
checkOf("the ideographic space inside", "es", 2048, at(5, "pal\u3000af")),
checkOf("LS inside", "es", 2048, at(5, "pal\u2028af")),
checkOf("an LF inside", "es", 2048, at(5, "pal\naf")),
checkOf("two runes and ZWSP: the count before the runes", "es", 2048, at(5, "p\u200b")),
checkOf("three letters with a mark that goes: two", "es", 2048, at(5, "pa\u0301")),
checkOf("a\u0301b, two letters once normalized", "es", 2048, at(5, "a\u0301b")),
// The runes of the normalized word.
checkOf("a control", "es", 2048, at(5, "pal\x01af")),
checkOf("DEL", "es", 2048, at(5, "pal\x7faf")),
checkOf("a C1 control", "es", 2048, at(5, "pal\u0080af")),
checkOf("a control and a capital: the runes before the alphabet", "es", 2048, at(5, "Pal\x01af")),
checkOf("a soft hyphen", "es", 2048, at(5, "pa\u00ad")),
checkOf("ZWJ", "es", 2048, at(5, "pal\u200daf")),
checkOf("a byte order mark", "es", 2048, at(5, "\ufeffpalaaf")),
checkOf("an unassigned code point", "es", 2048, at(5, "pal\u0378")),
checkOf("a noncharacter", "es", 2048, at(5, "pal\ufffe")),
checkOf("a tag", "es", 2048, at(5, "pal\U000e0041af")),
// The alphabet, on the word as the list writes it.
checkOf("the letters of the alphabet", "es", 2048, at(5, "abcdefghijklmnopqrstuvwxyz\u00e1\u00e9\u00ed\u00f3\u00fa\u00fc\u00f1")),
checkOf("an acute accent in NFD", "es", 2048, at(5, "pala\u0301e")),
checkOf("a diaeresis in NFD", "es", 2048, at(5, "palaa\u0308")),
checkOf("a tilde in NFD", "es", 2048, at(5, "pan\u0303o")),
checkOf("\u00c1, a capital with an accent", "es", 2048, at(5, "\u00c1baco")),
checkOf("\u00d1, a capital", "es", 2048, at(5, "\u00d1and\u00fa")),
checkOf("a grave accent", "es", 2048, at(5, "pal\u00e0a")),
checkOf("a circumflex", "es", 2048, at(5, "pal\u00e2a")),
checkOf("\u00e4", "es", 2048, at(5, "pal\u00e4a")),
checkOf("\u00e7", "es", 2048, at(5, "pal\u00e7a")),
checkOf("\u00f6", "es", 2048, at(5, "pal\u00f6a")),
checkOf("\u00fd", "es", 2048, at(5, "pal\u00fda")),
checkOf("\u00df", "es", 2048, at(5, "pal\u00dfa")),
checkOf("a Cyrillic a", "es", 2048, at(5, "pal\u0430a")),
checkOf("a Greek omicron", "es", 2048, at(5, "pal\u03bfa")),
checkOf("a full-width a", "es", 2048, at(5, "pal\uff41a")),
checkOf("a mathematical bold a", "es", 2048, at(5, "pal\U0001d41aa")),
checkOf("the Kelvin sign", "es", 2048, at(5, "pal\u212aa")),
checkOf("a dotless i", "es", 2048, at(5, "pal\u0131a")),
checkOf("a hyphen", "es", 2048, at(5, "pal-af")),
checkOf("an apostrophe", "es", 2048, at(5, "pal'af")),
checkOf("a backtick and a brace, around a to z", "es", 2048, at(5, "pal`{")),
checkOf("an emoji", "es", 2048, at(5, "pal\U0001f600")),
// Bytes that are not valid UTF-8: U+FFFD, which no alphabet holds.
checkOf("a byte that is not valid UTF-8", "es", 2048, at(5, "pal\xffaa")),
checkOf("a surrogate in UTF-8 bytes", "es", 2048, at(5, "\xed\xa0\x80aaa")),
checkOf("a sequence cut short", "es", 2048, at(5, "pala\xc3")),
checkOf("an overlong NUL", "es", 2048, at(5, "\xc0\x80aaa")),
checkOf("U+FFFD itself", "es", 2048, at(5, "pal\ufffdaa")),
// The same word once normalized, and the order across words.
checkOf("a\u00f1o and ano", "es", 2048, []edit{set(5, "a\u00f1o"), set(6, "ano")}),
checkOf("ping\u00fcino and pinguino", "es", 2048, []edit{set(5, "pinguino"), set(9, "ping\u00fcino")}),
checkOf("the same word, the first of the list last", "es", 2048, nil, "palaaa"),
checkOf("a capital of a word of the list: the alphabet before the same word", "es", 2048, at(5, "PALAAE")),
checkOf("an error at line 4 and another at line 6", "es", 2048, []edit{set(3, "Mal"), set(5, "pa")}),
checkOf("an error at the last line", "es", 2048, at(2047, "x")),
checkOf("an empty last line", "es", 2048, nil, ""),
}
return out
}
// ---------------------------------------------------------------------------
// The alphabet of every code point
type runeResult struct {
Rune int `json:"rune"`
Result string `json:"result"`
}
type alphabetHead struct {
Word string `json:"word"`
Planes []int `json:"planes"`
Count int `json:"count"`
Sha256 string `json:"sha256"`
Ok []string `json:"ok"`
}
func alphabet() (alphabetHead, []runeResult) {
planes := []int{0, 1, 14}
l := edited(2048, nil)
digest := sha256.New()
var results []runeResult
var ok []string
count := 0
for _, p := range planes {
for r := rune(p << 16); r <= rune(p<<16|0xffff); r++ {
if r >= 0xd800 && r <= 0xdfff {
continue
}
l[0] = "pala" + string(r) + "zz"
res := result(wordkey.CheckList("es", l))
fmt.Fprintf(digest, "U+%04X %s\n", r, res)
count++
if r <= 0x17f {
results = append(results, runeResult{int(r), res})
}
if res == "ok" {
ok = append(ok, string(r))
}
}
}
head := alphabetHead{"pala, the code point and zz, the first word of the base list of 2048", planes, count, hex.EncodeToString(digest.Sum(nil)), ok}
return head, results
}
// ---------------------------------------------------------------------------
// Generate
// seeded is the keystream of SeededRandomSource of lib/src/random.dart:
// ChaCha20 under SHA-256(seed), with a zero nonce, from block 0.
type seeded struct{ c *chacha20.Cipher }
func newSeeded(seed string) *seeded {
key := sha256.Sum256([]byte(seed))
c, err := chacha20.NewUnauthenticatedCipher(key[:], make([]byte, chacha20.NonceSize))
check(err)
return &seeded{c}
}
func (s *seeded) Read(p []byte) (int, error) {
clear(p)
s.c.XORKeyStream(p, p)
return len(p), nil
}
// counter is SHA-256(label ‖ i) for i = 0, 1, …, a 32-bit big-endian
// counter, one block after the other.
type counter struct {
label []byte
i uint32
block []byte
}
func (c *counter) Read(p []byte) (int, error) {
for n := 0; n < len(p); {
if len(c.block) == 0 {
b := sha256.Sum256(binary.BigEndian.AppendUint32(bytes.Clone(c.label), c.i))
c.block = b[:]
c.i++
}
k := copy(p[n:], c.block)
c.block = c.block[k:]
n += k
}
return len(p), nil
}
// counting counts the bytes read from r.
type counting struct {
r io.Reader
n int
}
func (c *counting) Read(p []byte) (int, error) {
n, err := c.r.Read(p)
c.n += n
return n, err
}
// stream is what a case of generate reads instead of crypto/rand: kind
// seeded, the keystream of Seed; counter, the blocks of the label Seed; or
// bytes, Hex repeated Repeat times, which end.
type stream struct {
Kind string `json:"kind"`
Seed string `json:"seed,omitempty"`
Hex string `json:"hex,omitempty"`
Repeat int `json:"repeat,omitempty"`
}
func (s stream) reader() io.Reader {
switch s.Kind {
case "seeded":
return newSeeded(s.Seed)
case "counter":
return &counter{label: []byte(s.Seed)}
case "bytes":
b, err := hex.DecodeString(s.Hex)
check(err)
return bytes.NewReader(bytes.Repeat(b, s.Repeat))
}
log.Fatalf("stream kind %q", s.Kind)
return nil
}
func seed(s string) stream { return stream{Kind: "seeded", Seed: s} }
func given(hexBytes string, repeat int) stream {
return stream{Kind: "bytes", Hex: hexBytes, Repeat: repeat}
}
type generateCase struct {
Name string `json:"name"`
List string `json:"list,omitempty"`
Size int `json:"size"`
N int `json:"n"`
Stream stream `json:"stream"`
Indices []int `json:"indices,omitempty"`
Words []string `json:"words,omitempty"`
Read int `json:"read"`
Next string `json:"next,omitempty"`
Error string `json:"error,omitempty"`
}
// numbers is a list of size words, "0", "1", …: Generate reads only its
// length and the words it draws, whose indices they are.
func numbers(size int) []string {
l := make([]string, size)
for i := range l {
l[i] = strconv.Itoa(i)
}
return l
}
func generateOf(name, listName string, list []string, n int, s stream) generateCase {
r := &counting{r: s.reader()}
words, err := wordkey.Generate(list, n, r)
c := generateCase{Name: name, List: listName, Size: len(list), N: n, Stream: s, Read: r.n}
if err != nil {
c.Error = err.Error()
return c
}
index := make(map[string]int, len(list))
for i, w := range list {
index[w] = i
}
for _, w := range words {
c.Indices = append(c.Indices, index[w])
}
if listName != "" && n <= 24 {
c.Words = words
}
if s.Kind != "bytes" {
next := make([]byte, 4)
_, err := io.ReadFull(r.r, next)
check(err)
c.Next = hex.EncodeToString(next)
}
return c
}
func generateCases() []generateCase {
es, err := wordkey.List("es")
check(err)
var out []generateCase
add := func(c generateCase) { out = append(out, c) }
// The Spanish list, from the keystream of a seed and from SHA-256
// counter blocks.
for _, n := range []int{6, wordkey.DefaultCount, 8, 12, 24} {
add(generateOf(fmt.Sprintf("%d words of the Spanish list", n), "es", es, n, seed(fmt.Sprintf("wordkey.Generate es %d", n))))
}
for _, n := range []int{6, wordkey.DefaultCount, 12} {
add(generateOf(fmt.Sprintf("%d words of the Spanish list, SHA-256 counter blocks", n), "es", es, n, stream{Kind: "counter", Seed: fmt.Sprintf("wordkey.Generate counter %d", n)}))
}
add(generateOf("3888 words of the Spanish list, the most", "es", es, len(es)/2, seed("wordkey.Generate es 3888")))
// Lists of every number of bytes of a draw of crypto/rand.Int, with and
// without draws again.
for _, size := range []int{12, 13, 16, 17, 100, 255, 256, 257, 2047, 2048, 2049, 4096, 7775, 7776, 7777, 8192, 65535, 65536, 65537, 100000, 1<<20 + 1} {
add(generateOf(fmt.Sprintf("6 words of %d", size), "", numbers(size), 6, seed(fmt.Sprintf("wordkey.Generate %d", size))))
}
add(generateOf("12 words of 65537, SHA-256 counter blocks", "", numbers(65537), 12, stream{Kind: "counter", Seed: "wordkey.Generate counter 65537"}))
// The most words of small lists: indices drawn twice, again and again.
for _, size := range []int{12, 13, 17, 100} {
add(generateOf(fmt.Sprintf("%d words of %d, the most", size/2, size), "", numbers(size), size/2, seed(fmt.Sprintf("wordkey.Generate %d most", size))))
}
// Bytes that end.
add(generateOf("the seed of TestGenerate: two indices only, then EOF", "es", es, 6, given("0701c821", 64)))
add(generateOf("twelve bytes, six indices", "es", es, 6, given("000000010002000300040005", 1)))
add(generateOf("8191 and 8032 drawn again, 0 drawn twice, 7775 the last index", "es", es, 6, given("1fff00001f601e5f00000001000200030004", 1)))
add(generateOf("the three bits above the 13 of 7775 are cleared", "es", es, 6, given("e000ffff21002200230024002500", 1)))
add(generateOf("the bytes end inside a draw", "es", es, 6, given("0000000100", 1)))
add(generateOf("no bytes", "es", es, 6, given("", 0)))
// The arguments, refused before anything is read.
for _, n := range []int{5, 0, -1, len(es)/2 + 1, len(es)} {
add(generateOf(fmt.Sprintf("%d words of the Spanish list", n), "es", es, n, seed("never read")))
}
for _, c := range []struct{ size, n int }{{11, 6}, {0, 6}, {0, 5}, {12, 7}, {13, 7}} {
add(generateOf(fmt.Sprintf("%d words of %d", c.n, c.size), "", numbers(c.size), c.n, seed("never read")))
}
return out
}
// ---------------------------------------------------------------------------
// Bits
type bitsCase struct {
Size int `json:"size"`
Count int `json:"count"`
Bits string `json:"bits"`
Hex string `json:"hex"`
}
func bitsOf(size, count int) bitsCase {
b := wordkey.Bits(size, count)
return bitsCase{size, count, strconv.FormatFloat(b, 'g', -1, 64), fmt.Sprintf("%016x", math.Float64bits(b))}
}
func bitsCases() []bitsCase {
var out []bitsCase
for _, size := range []int{1, 2, 3, 5, 12, 13, 100, 255, 256, 257, 2047, 2048, 2049, 4096, 7775, 7776, 7777, 8192, 10000, 65535, 65536, 65537, 1 << 20, 1<<31 - 1, 1 << 31, 1<<32 - 1, 1 << 32, 1<<32 + 1, 1<<52 + 1, 1<<53 - 1, 1 << 53} {
for _, count := range []int{0, 1, 2, 6, 7, 8, 12} {
out = append(out, bitsOf(size, count))
}
}
// Go adds log2 of 0, −Inf, and of a negative number, NaN, and adds
// nothing for a negative count.
for _, c := range [][2]int{{0, 1}, {-1, 1}, {7776, -1}, {0, 0}} {
out = append(out, bitsOf(c[0], c[1]))
}
return out
}
type bitsDigest struct {
Name string `json:"name"`
Sizes [2]int `json:"sizes"`
Counts [2]int `json:"counts"`
Sha256 string `json:"sha256"`
}
func digestOf(name string, sizes, counts [2]int) bitsDigest {
d := sha256.New()
var b [8]byte
for size := sizes[0]; size <= sizes[1]; size++ {
for count := counts[0]; count <= counts[1]; count++ {
binary.BigEndian.PutUint64(b[:], math.Float64bits(wordkey.Bits(size, count)))
d.Write(b[:])
}
}
return bitsDigest{name, sizes, counts, hex.EncodeToString(d.Sum(nil))}
}
func bitsDigests() []bitsDigest {
return []bitsDigest{
digestOf("one word of 1 to 65536: the log2 of Go", [2]int{1, 65536}, [2]int{1, 1}),
digestOf("6 to 8 words of 2048 to 10000", [2]int{2048, 10000}, [2]int{6, 8}),
digestOf("0 to 600 words of 7776", [2]int{7776, 7776}, [2]int{0, 600}),
digestOf("1 word of 2^32 - 4096 to 2^32 + 4096", [2]int{1<<32 - 4096, 1<<32 + 4096}, [2]int{1, 1}),
}
}
// ---------------------------------------------------------------------------
// JSON
// ascii escapes every code point above U+007F of JSON as \uXXXX, in UTF-16
// for those above U+FFFF: the same JSON, in ASCII.
func ascii(b []byte) []byte {
var out bytes.Buffer
for _, r := range string(b) {
switch {
case r < 0x80:
out.WriteByte(byte(r))
case r < 0x10000:
fmt.Fprintf(&out, `\u%04x`, r)
default:
r -= 0x10000
fmt.Fprintf(&out, `\u%04x\u%04x`, 0xd800+(r>>10), 0xdc00+(r&0x3ff))
}
}
return out.Bytes()
}
func marshal(v any) []byte {
var b bytes.Buffer
enc := json.NewEncoder(&b)
enc.SetEscapeHTML(false)
check(enc.Encode(v))
return ascii(bytes.TrimSuffix(b.Bytes(), []byte("\n")))
}
// list writes the items of a JSON array one per line.
func list[T any](w *bytes.Buffer, key string, items []T, last bool) {
fmt.Fprintf(w, " %q: [\n", key)
for i, it := range items {
w.WriteString(" ")
w.Write(marshal(it))
if i < len(items)-1 {
w.WriteByte(',')
}
w.WriteByte('\n')
}
w.WriteString(" ]")
if !last {
w.WriteByte(',')
}
w.WriteByte('\n')
}
func main() {
outDir := flag.String("out", "", "the directory test/vectors of datekeys-dart")
source := flag.String("source", "", "the commit of datekeys-go")
flag.Parse()
if *outDir == "" || *source == "" {
log.Fatal("usage: go run wordlist_go_vectors.go -source COMMIT -out DIR")
}
lists := listInfos()
parse := parseCases()
checks := checkCases()
alpha, runes := alphabet()
gen := generateCases()
bits := bitsCases()
digests := bitsDigests()
var w bytes.Buffer
w.WriteString("{\n")
head := []struct {
key string
v any
}{
{"description", "The word lists of package wordkey of the Go reference (generate.go, spec \u00a738.1) for lib/src/wordlist.dart; see tool/wordlist_go_vectors.go. " +
"Binary values and words are lower-case hex. The base list is word i = pal and three letters, a + i/676, a + i/26%26 and a + i%26; a case of size n takes its first n words, replaces the word at each at of set, and adds those of append. " +
"lists: the SHA-256, bytes and words of each built-in list. parse: the body of List on the text prefix, the base list joined by sep, and suffix (sha256 is that of the text): the words read, and ok or the error. " +
"check_list: CheckList, ok or the error (sha256 is that of the words joined by LF). alphabet: CheckList of the base list of 2048 whose first word is pala, a code point and zz, for each code point of the planes but the surrogates: the SHA-256 of the lines U+XXXX result, the code points of the words it accepts, and alphabet_results, each result up to U+017F. " +
"generate: Generate on the list of the Spanish words (list es) or on a list of size words whose indices they are, reading a stream: seeded, the keystream of SeededRandomSource of the seed; counter, SHA-256 of the label and a 32-bit big-endian counter from 0, block after block; bytes, hex repeated repeat times, which end. The indices drawn, the bytes read and the next four of an endless stream, or the error. " +
"bits: Bits, its shortest decimal and the hex of its float64. bits_digests: the SHA-256 of the 8 big-endian bytes of each Bits of the sizes and counts, inclusive, sizes outer."},
{"generator", "tool/wordlist_go_vectors.go, " + runtime.Version()},
{"source", *source},
{"default_count", wordkey.DefaultCount},
{"min_list_size", wordkey.MinListSize},
{"min_words", wordkey.MinWords},
{"min_letters", wordkey.MinLetters},
{"alphabet", alpha},
}
for _, kv := range head {
fmt.Fprintf(&w, " %q: %s,\n", kv.key, marshal(kv.v))
}
list(&w, "lists", lists, false)
list(&w, "parse", parse, false)
list(&w, "check_list", checks, false)
list(&w, "alphabet_results", runes, false)
list(&w, "generate", gen, false)
list(&w, "bits", bits, false)
list(&w, "bits_digests", digests, true)
w.WriteString("}\n")
var parsed any
if err := json.Unmarshal(w.Bytes(), &parsed); err != nil {
log.Fatalf("the JSON does not parse: %v", err)
}
if bytes.Contains(w.Bytes(), []byte("'''")) {
log.Fatal("the JSON holds three quotes")
}
path := filepath.Join(*outDir, "wordlist_vectors.json")
check(os.WriteFile(path, w.Bytes(), 0o644))
fmt.Printf("wrote %s, %d bytes: %d lists, %d texts, %d lists checked, %d code points, %d draws, %d bits, %d digests\n",
path, w.Len(), len(lists), len(parse), len(checks), alpha.Count, len(gen), len(bits), len(digests))
dart := "// Generated by tool/wordlist_go_vectors.go from wordlist_vectors.json, for\n" +
"// the tests that also run compiled to JavaScript, where no file can be\n" +
"// read. Do not edit.\n\n" +
"/// The text of test/vectors/wordlist_vectors.json.\n" +
"const wordlistVectorsJson = r'''\n" + w.String() + "''';\n"
dpath := filepath.Join(*outDir, "wordlist_vectors.g.dart")
check(os.WriteFile(dpath, []byte(dart), 0o644))
fmt.Printf("wrote %s\n", dpath)
}

Powered by TurnKey Linux.