You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
DateKeys/internal/pathrule/unicode.go

258 lines
6.7 KiB

This file contains ambiguous Unicode characters!

This file contains ambiguous Unicode characters that may be confused with others in your current locale. If your use case is intentional and legitimate, you can safely ignore this warning. Use the Escape button to highlight these characters.

// Package pathrule checks the paths and the texts of the head of a format 3
// capsule (spec §29.5, §29.6) with fixed Unicode 18.0.0 and WindowsBestFit
// tables (spec §29.5.1), never with the Unicode functions of the platform,
// whose version changes with each runtime: Go 1.26.8 carries Unicode 15.0.0,
// and Node 24.9, 16.0.
//
// The tables are generated by internal/pathrule/gen; see tables.go.
package pathrule
//go:generate go run ./gen -data ../../.cache -go tables.go
import (
"crypto/sha256"
"encoding/hex"
"fmt"
"slices"
"strings"
)
// mapping is a sorted table from a code point to a sequence of code points:
// the value of keys[i] is data[start[i]:start[i+1]].
type mapping struct {
keys []rune
start []uint16
data []rune
}
func (m *mapping) lookup(r rune) ([]rune, bool) {
i, ok := slices.BinarySearch(m.keys, r)
if !ok {
return nil, false
}
return m.data[m.start[i]:m.start[i+1]], true
}
// bestFitTable is one code page of WindowsBestFit, reduced to the non-ASCII
// code points it maps to an ASCII byte.
type bestFitTable struct {
name string
from []rune
to []byte
}
func (t *bestFitTable) lookup(r rune) (byte, bool) {
i, ok := slices.BinarySearch(t.from, r)
if !ok {
return 0, false
}
return t.to[i], true
}
// inRanges reports whether r lies in one of the inclusive ranges lo, hi, ...
func inRanges(ranges []rune, r rune) bool {
// The first range whose hi is >= r.
lo, hi := 0, len(ranges)/2
for lo < hi {
m := int(uint(lo+hi) >> 1)
if ranges[2*m+1] < r {
lo = m + 1
} else {
hi = m
}
}
return lo < len(ranges)/2 && ranges[2*lo] <= r
}
// Assigned reports whether r is an assigned code point of Unicode 18.0.0: a
// code point that is not of general category Cn.
func Assigned(r rune) bool { return inRanges(assignedRanges, r) }
// DefaultIgnorable reports whether r has the property
// Default_Ignorable_Code_Point in Unicode 18.0.0.
func DefaultIgnorable(r rune) bool { return inRanges(ignorableRanges, r) }
// ccc is the canonical combining class of r.
func ccc(r rune) uint8 {
i, ok := slices.BinarySearchFunc(cccTable, r, func(e uint32, r rune) int {
switch c := rune(e >> 8); {
case c < r:
return -1
case c > r:
return 1
}
return 0
})
if !ok {
return 0
}
return uint8(cccTable[i])
}
// Hangul syllable decomposition (Unicode §3.12).
const (
hangulS = 0xAC00
hangulL = 0x1100
hangulV = 0x1161
hangulT = 0x11A7
hangulN = 21 * 28
hangulC = 19 * hangulN
)
// NFD returns the canonical decomposition of s, normalization form D of UAX
// #15 with the data of Unicode 18.0.0: every code point replaced by its full
// canonical decomposition, Hangul syllables decomposed algorithmically, and
// each run of combining marks put in canonical order, a stable sort by
// canonical combining class.
func NFD(s string) string {
var out []rune
for _, r := range s {
switch {
case r >= hangulS && r < hangulS+hangulC:
i := r - hangulS
out = append(out, hangulL+i/hangulN, hangulV+(i%hangulN)/28)
if t := i % 28; t != 0 {
out = append(out, hangulT+t)
}
default:
if d, ok := decompositions.lookup(r); ok {
out = append(out, d...)
} else {
out = append(out, r)
}
}
}
// Canonical ordering: insertion sort within each run of non-starters,
// stable, by combining class.
for i := 1; i < len(out); i++ {
c := ccc(out[i])
if c == 0 {
continue
}
for j := i; j > 0; j-- {
p := ccc(out[j-1])
if p <= c {
break
}
out[j-1], out[j] = out[j], out[j-1]
}
}
return string(out)
}
// Fold returns the case folding of s: the mappings of CaseFolding.txt with
// status C or F, and U+0131 (ı) to U+0069 (i), as NTFS does (spec §29.5, R7).
func Fold(s string) string {
var b strings.Builder
for _, r := range s {
if r == 0x0131 {
b.WriteRune(0x0069)
} else if f, ok := foldings.lookup(r); ok {
for _, x := range f {
b.WriteRune(x)
}
} else {
b.WriteRune(r)
}
}
return b.String()
}
// Lower returns the simple lowercase mapping of r in Unicode 18.0.0, field
// 13 of UnicodeData.txt, or r itself. The key of words uses it (spec §38.1).
func Lower(r rune) rune {
if l, ok := lowercases.lookup(r); ok {
return l[0]
}
return r
}
// The whitelist of R4: the Default_Ignorable_Code_Point that a path may hold
// (spec §29.5).
const (
zwnj = 0x200C
zwj = 0x200D
vs15 = 0xFE0E
vs16 = 0xFE0F
)
func whitelisted(r rune) bool { return r == zwnj || r == zwj || r == vs15 || r == vs16 }
// stripWhitelist removes the whitelisted code points of R4 from s.
func stripWhitelist(s string) string {
return strings.Map(func(r rune) rune {
if whitelisted(r) {
return -1
}
return r
}, s)
}
// Key is the key of a segment in R7: NFD(fold(NFD(s′))), where s′ is the
// segment without the whitelisted code points of R4. They are removed
// before normalizing because they have combining class 0: between two
// combining marks, they would change their canonical order.
func Key(segment string) string {
return NFD(Fold(NFD(stripWhitelist(segment))))
}
// allowsSelector reports whether base forms an emoji variation sequence with
// the selector sel, U+FE0E or U+FE0F (R4b).
func allowsSelector(base, sel rune) bool {
bases := vs16Bases
if sel == vs15 {
bases = vs15Bases
}
_, ok := slices.BinarySearch(bases, base)
return ok
}
// Canonical is the canonical text of the tables, whose SHA-256 is
// TablesDigest; the generator writes the same text from the data files. Each
// value is in upper-case hexadecimal of at least four digits, except the
// combining class and the code page, in decimal, and the best-fit byte, in
// two hexadecimal digits.
func Canonical() string {
var b strings.Builder
fmt.Fprintf(&b, "unicode %s\n", UnicodeVersion)
for i := 0; i < len(assignedRanges); i += 2 {
fmt.Fprintf(&b, "assigned %04X %04X\n", assignedRanges[i], assignedRanges[i+1])
}
for i := 0; i < len(ignorableRanges); i += 2 {
fmt.Fprintf(&b, "ignorable %04X %04X\n", ignorableRanges[i], ignorableRanges[i+1])
}
for _, e := range cccTable {
fmt.Fprintf(&b, "ccc %04X %d\n", e>>8, e&0xFF)
}
for _, m := range []struct {
name string
t *mapping
}{{"decomp", &decompositions}, {"fold", &foldings}, {"lower", &lowercases}} {
for i, k := range m.t.keys {
fmt.Fprintf(&b, "%s %04X", m.name, k)
for _, r := range m.t.data[m.t.start[i]:m.t.start[i+1]] {
fmt.Fprintf(&b, " %04X", r)
}
b.WriteByte('\n')
}
}
for _, r := range vs15Bases {
fmt.Fprintf(&b, "vs15 %04X\n", r)
}
for _, r := range vs16Bases {
fmt.Fprintf(&b, "vs16 %04X\n", r)
}
for _, t := range bestFitTables {
for i, r := range t.from {
fmt.Fprintf(&b, "bestfit %s %04X %02X\n", t.name, r, t.to[i])
}
}
return b.String()
}
// digest is the SHA-256 of Canonical, in hexadecimal.
func digest() string {
sum := sha256.Sum256([]byte(Canonical()))
return hex.EncodeToString(sum[:])
}

Powered by TurnKey Linux.