package main import ( "encoding/hex" "io" "os" "path/filepath" "slices" "strings" "testing" ) // unicodeDataPath is the copy of UnicodeData.txt of Unicode 18.0.0 that // internal/pathrule/gen reads; it is not committed. The tests of the full // normalization skip without it. const unicodeDataPath = "../../.cache/unicode/18.0.0/UnicodeData.txt" func loadUnicodeData(t *testing.T) *unicodeData { t.Helper() b, err := os.ReadFile(unicodeDataPath) if err != nil { t.Skipf("no UnicodeData.txt of Unicode 18.0.0 in %s: %v", unicodeDataPath, err) } u, err := parseUnicodeData(b) if err != nil { t.Fatal(err) } return u } // The vectors of the annex, 79.7: the words without tables and their key, // for the chain hash of Quicknet, round 1000 and capsule_id 00 to 0f. func TestAnnexWordVectors(t *testing.T) { id := unhex(t, "000102030405060708090a0b0c0d0e0f") for _, c := range []struct{ text, words, key string }{ {"perro luna casa verde tren mar", "perro luna casa verde tren mar", "fceec4d8ca8de86c85a1f26ed49f82a2b38431bd0ce36db995ae7dfd49b96e41"}, {"\u00d1and\u00fa PING\u00dcINO\tcami\u00f3n \u00e1rbol \u00c9ter ola", "nandu pinguino camion arbol eter ola", "273295d29370126a3be50b743132718d3cd9137fb3bb4cb20aa23163d2e19bb7"}, {"N\u0303andu\u0301 PINGU\u0308INO\tcamio\u0301n a\u0301rbol E\u0301ter ola", "nandu pinguino camion arbol eter ola", "273295d29370126a3be50b743132718d3cd9137fb3bb4cb20aa23163d2e19bb7"}, } { words, ok := normalizeSimple(c.text) if !ok || strings.Join(words, " ") != c.words { t.Fatalf("%q: %q, %v", c.text, words, ok) } key, err := wordKey(words, 1000, id) if err != nil || hex.EncodeToString(key) != c.key { t.Fatalf("%q: key %x, %v", c.text, key, err) } } } // The recipe without tables refuses what it does not cover, and then the // words need UnicodeData.txt. func TestNormalizeSimpleRefuses(t *testing.T) { for _, text := range []string{"\u00e7a va", "stra\u00dfe", "\u00e0 la", "\u03c3\u03b1\u03c3", "a\u00a0b", "\u212b", "a\u200bb"} { if _, ok := normalizeSimple(text); ok { t.Errorf("%q: the recipe without tables does not cover it", text) } if _, err := normalizeWords(text, nil); err == nil || !strings.Contains(err.Error(), "-unicodedata") { t.Errorf("%q: %v", text, err) } } // The whole of printable ASCII stays, and only A to Z change. var ascii []rune for r := rune(0x21); r <= 0x7e; r++ { ascii = append(ascii, r) } words, ok := normalizeSimple(string(ascii)) if !ok || len(words) != 1 || words[0] != strings.ToLower(string(ascii)) { t.Errorf("printable ASCII: %q, %v", words, ok) } } // The full normalization gives the words of every case of wordkey.json, and // the recipe without tables, where it applies, the same. func TestNormalizeVectors(t *testing.T) { u := loadUnicodeData(t) var f struct { Normalize []struct { Name string `json:"name"` Text string `json:"text"` Words []string `json:"words"` } `json:"normalize"` } readJSON(t, filepath.Join(vectorsDir, "wordkey.json"), &f) if len(f.Normalize) < 40 { t.Fatalf("%d cases of normalize", len(f.Normalize)) } simple := 0 for _, c := range f.Normalize { got := u.normalize(c.Text) if got == nil { got = []string{} } if !slices.Equal(got, c.Words) { t.Errorf("%s: %q, want %q", c.Name, got, c.Words) } if words, ok := normalizeSimple(c.Text); ok { simple++ if words == nil { words = []string{} } if !slices.Equal(words, c.Words) { t.Errorf("%s, without tables: %q, want %q", c.Name, words, c.Words) } } } if simple < 8 { t.Errorf("only %d cases without tables", simple) } } // Only UnicodeData.txt of Unicode 18.0.0 is accepted: another version can // give other words. func TestUnicodeDataVersion(t *testing.T) { b, err := os.ReadFile(unicodeDataPath) if err != nil { t.Skip(err) } b[len(b)-2] ^= 1 if _, err := parseUnicodeData(b); err == nil || !strings.Contains(err.Error(), "not that of Unicode 18.0.0") { t.Fatalf("a changed UnicodeData.txt: %v", err) } } // format3_time_and_key_words opens with the text of its words, which needs // no tables, and with UnicodeData.txt too. func TestRecoverWithWords(t *testing.T) { rec, dkc, relObj, _ := loadFixture(t, "format3_time_and_key_words") if rec.WordsText == "" { t.Fatal("the record has no words_text") } text := rec.WordsText res, err := recoverCapsule(dkc, relObj, credential{words: &text}, io.Discard) if err != nil || res.format != 3 || len(res.body.files) != 1 { t.Fatalf("%v", err) } other := "nandu pinguino camion arbol eter mar" if _, err := recoverCapsule(dkc, relObj, credential{words: &other}, io.Discard); err == nil || !strings.Contains(err.Error(), "access layer") { t.Fatalf("other words: %v", err) } if b, err := os.ReadFile(unicodeDataPath); err == nil { u, err := parseUnicodeData(b) if err != nil { t.Fatal(err) } if _, err := recoverCapsule(dkc, relObj, credential{words: &text, ucd: u}, io.Discard); err != nil { t.Fatal(err) } } } type ioDiscard struct{} func (ioDiscard) Write(p []byte) (int, error) { return len(p), nil }