package main import ( "bufio" "bytes" "crypto/pbkdf2" "crypto/sha256" "encoding/hex" "errors" "fmt" "strconv" "strings" ) // --------------------------------------------------------------------------- // The key of words (annex 79.7, spec §38.1): words give the X25519 identity // of a credential of a time_and_key capsule. // // P = the words, normalized, in UTF-8, separated by a space // S = "DateKeys llave de palabras v2|" || chain_hash || "|" || round || "|" || capsule_id // id = PBKDF2-HMAC-SHA256(P, S, 600000 iterations, 32 bytes) // // with the chain hash and capsule_id in lower-case hexadecimal and the round // in decimal. const wordKeyIterations = 600000 // wordKey derives the identity of the words of the capsule. func wordKey(words []string, round uint64, capsuleID []byte) ([]byte, error) { p := strings.Join(words, " ") s := "DateKeys llave de palabras v2|" + quicknetChainHash + "|" + strconv.FormatUint(round, 10) + "|" + hex.EncodeToString(capsuleID) return pbkdf2.Key(sha256.New, p, []byte(s), wordKeyIterations, 32) } // --------------------------------------------------------------------------- // The normalization without tables (79.7): for a text of printable ASCII, // the ASCII spaces, á, é, í, ó, ú, ü, ñ and their capitals, and the marks // U+0300 to U+036F, which is what any word of the DateKeys lists gives. It // is exactly the normalization of §38.1 for that text. // latinBase maps the letters of the recipe to their base letter, in lower // case. var latinBase = map[rune]rune{ 0x00e1: 'a', 0x00c1: 'a', // á Á 0x00e9: 'e', 0x00c9: 'e', // é É 0x00ed: 'i', 0x00cd: 'i', // í Í 0x00f3: 'o', 0x00d3: 'o', // ó Ó 0x00fa: 'u', 0x00fc: 'u', 0x00da: 'u', 0x00dc: 'u', // ú ü Ú Ü 0x00f1: 'n', 0x00d1: 'n', // ñ Ñ } func asciiSpace(r rune) bool { return r == ' ' || r >= 0x09 && r <= 0x0d } // normalizeSimple applies the recipe without tables, and returns false when // the text holds a character outside it. func normalizeSimple(text string) ([]string, bool) { var sb strings.Builder for _, r := range text { switch { case r >= 0x300 && r <= 0x36f: // 1. the marks go case latinBase[r] != 0: // 2. the letters with marks, to their base sb.WriteRune(latinBase[r]) case r >= 'A' && r <= 'Z': // 3. A to Z, to a to z sb.WriteRune(r + 'a' - 'A') case r >= 0x21 && r <= 0x7e || asciiSpace(r): sb.WriteRune(r) default: return nil, false } } return strings.FieldsFunc(sb.String(), asciiSpace), true // 4. } // --------------------------------------------------------------------------- // The full normalization (79.7), for any other text: NFD of UAX #15, without // U+0300 to U+036F, the simple lower case of each code point, and the split // at the spaces of §38.1. It reads UnicodeData.txt of Unicode 18.0.0, which // the annex names by its SHA-256. const unicodeDataSHA256 = "0736451de439ae7baf1425136617da495e09ee5afbe6e394374db7009ea08950" // unicodeData holds what the normalization takes from UnicodeData.txt: the // canonical decomposition (field 5, without the tagged ones), the canonical // combining class (field 3) and the simple lower case (field 13). type unicodeData struct { decomposition map[rune][]rune class map[rune]int lower map[rune]rune } // parseUnicodeData reads UnicodeData.txt, which must be that of Unicode // 18.0.0: another version can give other words. func parseUnicodeData(b []byte) (*unicodeData, error) { sum := sha256.Sum256(b) if hex.EncodeToString(sum[:]) != unicodeDataSHA256 { return nil, fmt.Errorf("UnicodeData.txt has the SHA-256 %x, not that of Unicode 18.0.0, %s", sum, unicodeDataSHA256) } u := &unicodeData{decomposition: map[rune][]rune{}, class: map[rune]int{}, lower: map[rune]rune{}} sc := bufio.NewScanner(bytes.NewReader(b)) for sc.Scan() { f := strings.Split(sc.Text(), ";") if len(f) != 15 { return nil, fmt.Errorf("UnicodeData.txt: a line of %d fields", len(f)) } cp, err := codePoint(f[0]) if err != nil { return nil, err } // A range, <…, First> to <…, Last>, has no decomposition, class or // lower case: its two lines say nothing more. if c, _ := strconv.Atoi(f[3]); c != 0 { u.class[cp] = c } if f[5] != "" && !strings.HasPrefix(f[5], "<") { for _, h := range strings.Fields(f[5]) { d, err := codePoint(h) if err != nil { return nil, err } u.decomposition[cp] = append(u.decomposition[cp], d) } } if f[13] != "" { if u.lower[cp], err = codePoint(f[13]); err != nil { return nil, err } } } return u, sc.Err() } func codePoint(h string) (rune, error) { n, err := strconv.ParseUint(h, 16, 32) if err != nil || n > 0x10ffff { return 0, fmt.Errorf("UnicodeData.txt: %q is not a code point", h) } return rune(n), nil } // The Hangul syllables decompose by computation (Unicode, §3.12). const ( hangulS, hangulL, hangulV, hangulT = 0xac00, 0x1100, 0x1161, 0x11a7 hangulVCount, hangulTCount = 21, 28 hangulCount = 19 * hangulVCount * hangulTCount ) // decompose appends the full canonical decomposition of r. func (u *unicodeData) decompose(out []rune, r rune) []rune { if s := r - hangulS; s >= 0 && s < hangulCount { out = append(out, hangulL+s/(hangulVCount*hangulTCount), hangulV+s%(hangulVCount*hangulTCount)/hangulTCount) if t := s % hangulTCount; t != 0 { out = append(out, hangulT+t) } return out } d, ok := u.decomposition[r] if !ok { return append(out, r) } for _, x := range d { out = u.decompose(out, x) } return out } // nfd is the NFD of text: the full canonical decomposition, then the // canonical ordering of each run of marks by their combining class. func (u *unicodeData) nfd(text string) []rune { var out []rune for _, r := range text { out = u.decompose(out, r) } for i := 1; i < len(out); i++ { for j := i; j > 0; j-- { a, b := u.class[out[j-1]], u.class[out[j]] if b == 0 || a <= b { break } out[j-1], out[j] = out[j], out[j-1] } } return out } // wordSpace is a space of §38.1, step 4: those of Go's unicode.IsSpace. func wordSpace(r rune) bool { switch { case asciiSpace(r), r == 0x85, r == 0xa0, r == 0x1680, r >= 0x2000 && r <= 0x200a, r == 0x2028, r == 0x2029, r == 0x202f, r == 0x205f, r == 0x3000: return true } return false } // normalize applies the full normalization. func (u *unicodeData) normalize(text string) []string { var sb strings.Builder for _, r := range u.nfd(text) { if r >= 0x300 && r <= 0x36f { continue } if l, ok := u.lower[r]; ok { r = l } sb.WriteRune(r) } return strings.FieldsFunc(sb.String(), wordSpace) } // normalizeWords normalizes text without tables when it can, and with u // otherwise. func normalizeWords(text string, u *unicodeData) ([]string, error) { if words, ok := normalizeSimple(text); ok { return words, nil } if u == nil { return nil, errors.New("the words hold a character outside the normalization without tables of the annex (79.7): give UnicodeData.txt of Unicode 18.0.0 with -unicodedata") } return u.normalize(text), nil }