package pathrule import ( "fmt" "regexp" "strings" "unicode/utf8" ) // Limits of the paths and texts of a format 3 head, fixed with the format // (spec §29.4, §29.5). const ( MaxPathLen = 1024 // R1, bytes MaxSegments = 32 // R2 MaxSegmentLen = 255 // R3, bytes MaxSegmentUTF16 = 255 // R3, UTF-16 code units of the NFD of the segment MaxImplicitDirs = 65535 // R9 MaxCommentLen = 16384 // bytes MaxAuthorLen = 256 // bytes reservedDirPrefix = ".datekeys-" ) // Error is the violation of one rule of spec §29.5 or §29.6. Its text is // the same in every implementation, so that the message of a rejected head // does not depend on the reader. type Error struct { Rule string // "R2" to "R10", "text" Detail string // Paths are the positions, from 1, of the two paths of an R7 // violation, the later one first, so that a writer can name them; zero // for the other rules. Paths [2]int } func (e *Error) Error() string { return e.Rule + ": " + e.Detail } func fail(rule, format string, args ...any) error { return &Error{Rule: rule, Detail: fmt.Sprintf(format, args...)} } // codePoint formats r as in the spec: U+ and at least four upper-case // hexadecimal digits. func codePoint(r rune) string { return fmt.Sprintf("U+%04X", r) } // forbiddenASCII are the ASCII characters of R4 that are not controls. const forbiddenASCII = `"*:<>?\|` // CheckPath checks one path of the head with the rules of the fourth layer // that concern it alone, in this order: R2, and then for each segment R3, // R4, R4b, R5, R6, R6b and R6c; then R10. R1 and R8 belong to the third // layer, and R7 and R9 to the whole tree (CheckTree). The first violation // is returned. func CheckPath(path string) error { segs := strings.Split(path, "/") if len(segs) > MaxSegments { return fail("R2", "%d segments, more than %d", len(segs), MaxSegments) } for i, s := range segs { if s == "" { return fail("R2", "segment %d is empty", i+1) } } for i, s := range segs { if err := checkSegment(s); err != nil { e := err.(*Error) e.Detail = fmt.Sprintf("segment %d: %s", i+1, e.Detail) return e } } if strings.HasPrefix(Key(segs[0]), reservedDirPrefix) { return fail("R10", "the first segment starts with %q", reservedDirPrefix) } return nil } // checkSegment applies R3, R4, R4b, R5, R6, R6b and R6c to a segment. func checkSegment(s string) error { if err := checkR3(s); err != nil { return err } for _, r := range s { if err := checkR4(r); err != nil { return err } } if err := checkPlacement("R4b", s); err != nil { return err } if err := checkR5(s); err != nil { return err } if err := checkR6(s); err != nil { return err } if err := checkR6b(s); err != nil { return err } return checkR6c(s) } func checkR3(s string) error { if len(s) > MaxSegmentLen { return fail("R3", "%d bytes, more than %d", len(s), MaxSegmentLen) } switch s { case ".": return fail("R3", "the segment is a dot") case "..": return fail("R3", "the segment is two dots") } switch stripWhitelist(s) { case "": return fail("R3", "the segment is empty without ZWNJ, ZWJ, VS15 and VS16") case ".": return fail("R3", "the segment is a dot without ZWNJ, ZWJ, VS15 and VS16") case "..": return fail("R3", "the segment is two dots without ZWNJ, ZWJ, VS15 and VS16") } if n := utf16Len(NFD(s)); n > MaxSegmentUTF16 { return fail("R3", "its NFD is %d UTF-16 code units, more than %d", n, MaxSegmentUTF16) } return nil } func utf16Len(s string) int { n := 0 for _, r := range s { if r >= 0x10000 { n += 2 } else { n++ } } return n } // checkR4 rejects the code points of R4. func checkR4(r rune) error { switch { case r <= 0x1F || (r >= 0x7F && r <= 0x9F): return fail("R4", "control %s", codePoint(r)) case strings.ContainsRune(forbiddenASCII, r): return fail("R4", "character %s", codePoint(r)) case r == 0x2028 || r == 0x2029: return fail("R4", "separator %s", codePoint(r)) case DefaultIgnorable(r) && !whitelisted(r): return fail("R4", "invisible %s", codePoint(r)) case r >= 0xF000 && r <= 0xF0FF: return fail("R4", "private use %s", codePoint(r)) case !Assigned(r): return fail("R4", "unassigned %s", codePoint(r)) } return nil } // checkPlacement applies R4b to s, a segment or a line of a text: VS15 and // VS16 only right after a character that forms an emoji variation sequence // with them, and ZWJ and ZWNJ never first, last or right after another ZWJ // or ZWNJ. func checkPlacement(rule, s string) error { prev, i := rune(-1), 0 for _, r := range s { switch r { case vs15, vs16: if prev < 0 || !allowsSelector(prev, r) { return fail(rule, "%s is not part of an emoji variation sequence", codePoint(r)) } case zwj, zwnj: switch { case prev < 0: return fail(rule, "%s at the start", codePoint(r)) case prev == zwj || prev == zwnj: return fail(rule, "%s right after %s", codePoint(r), codePoint(prev)) case i+utf8.RuneLen(r) == len(s): return fail(rule, "%s at the end", codePoint(r)) } } prev = r i += utf8.RuneLen(r) } return nil } func checkR5(s string) error { switch { case strings.HasPrefix(s, " "): return fail("R5", "the segment starts with U+0020") case strings.HasSuffix(s, " "): return fail("R5", "the segment ends with U+0020") case strings.HasSuffix(s, "."): return fail("R5", "the segment ends with '.'") } return nil } // reserved are the device names of R6, in upper case. var reserved = map[string]bool{ "CON": true, "PRN": true, "AUX": true, "NUL": true, "CONIN$": true, "CONOUT$": true, "COM¹": true, "COM²": true, "COM³": true, "LPT¹": true, "LPT²": true, "LPT³": true, } func init() { for d := '0'; d <= '9'; d++ { reserved["COM"+string(d)] = true reserved["LPT"+string(d)] = true } } func checkR6(s string) error { base, _, _ := strings.Cut(s, ".") base = strings.TrimRight(base, " ") if name := asciiUpper(base); reserved[name] { return fail("R6", "%s is a reserved device name", name) } return nil } // asciiUpper upper-cases ASCII letters only. func asciiUpper(s string) string { return strings.Map(func(r rune) rune { if r >= 'a' && r <= 'z' { return r - 'a' + 'A' } return r }, s) } // shortNameBase is the base of an 8.3 alias in R6b: it ends in '~' and 1 to // 6 ASCII digits. It has no lookahead, which the regexp package lacks. var shortNameBase = regexp.MustCompile(`^[^.]*~[0-9]{1,6}$`) func checkR6b(s string) error { if strings.Count(s, ".") > 1 { return nil } base, ext, _ := strings.Cut(s, ".") if n := utf8.RuneCountInString(base); n < 1 || n > 8 { return nil } if utf8.RuneCountInString(ext) > 3 || !shortNameBase.MatchString(base) { return nil } return fail("R6b", "the segment has the form of an 8.3 alias") } // checkR6c projects the segment with each best-fit table: every non-ASCII // code point the table maps to an ASCII byte is replaced by that byte. A // projection must not hold '/', '\', ':' or U+0000, and must pass R3, R5, R6 // and R6b. func checkR6c(s string) error { for i := range bestFitTables { t := &bestFitTables[i] p, changed := project(s, t) if !changed { continue } if j := strings.IndexAny(p, "/\\:\x00"); j >= 0 { return fail("R6c", "code page %s maps the segment to one with %s", t.name, codePoint(rune(p[j]))) } for _, check := range []func(string) error{checkR3, checkR5, checkR6, checkR6b} { if err := check(p); err != nil { e := err.(*Error) return fail("R6c", "code page %s maps the segment to one that breaks %s: %s", t.name, e.Rule, e.Detail) } } } return nil } func project(s string, t *bestFitTable) (string, bool) { var b strings.Builder changed := false for _, r := range s { if r >= 0x80 { if c, ok := t.lookup(r); ok { b.WriteByte(c) changed = true continue } } b.WriteRune(r) } return b.String(), changed } // CheckTree applies R7 and then R9 to the paths of a head, which have passed // CheckPath: no two siblings with the same key, no path that is both a file // and a folder, comparing keys, and at most MaxImplicitDirs folders. Paths // are numbered from 1 in the messages. func CheckTree(paths []string) error { type node struct { name string dir bool path int // the first path that reached it } // children maps the key of a folder, "" for the root, to its children by // key; the key of a folder joins the keys of its segments with '/'. children := map[string]map[string]node{} for n, path := range paths { segs := strings.Split(path, "/") parent := "" for i, s := range segs { k := Key(s) isDir := i < len(segs)-1 kids := children[parent] if kids == nil { kids = map[string]node{} children[parent] = kids } switch old, ok := kids[k]; { case !ok: kids[k] = node{name: s, dir: isDir, path: n + 1} case old.name != s: return &Error{Rule: "R7", Detail: fmt.Sprintf("path %d collides with path %d in segment %d", n+1, old.path, i+1), Paths: [2]int{n + 1, old.path}} case old.dir != isDir || !isDir: return &Error{Rule: "R7", Detail: fmt.Sprintf("path %d makes a file of path %d a folder, or the reverse, in segment %d", n+1, old.path, i+1), Paths: [2]int{n + 1, old.path}} } if parent == "" { parent = k } else { parent += "/" + k } } } dirs := map[string]bool{} for _, path := range paths { for i := 0; ; { j := strings.IndexByte(path[i:], '/') if j < 0 { break } i += j dirs[path[:i]] = true i++ } } if len(dirs) > MaxImplicitDirs { return fail("R9", "%d folders, more than %d", len(dirs), MaxImplicitDirs) } return nil } // CheckComment checks the comment of a head (spec §29.6). func CheckComment(s string) error { return checkText(s, true) } // CheckAuthor checks the declared author of a head (spec §29.6). func CheckAuthor(s string) error { if err := checkText(s, false); err != nil { return err } if strings.HasPrefix(s, " ") || strings.HasSuffix(s, " ") { return fail("text", "the declared author starts or ends with U+0020") } return nil } // checkText applies the rules of spec §29.6, code point by code point, and // then R4b to each line. func checkText(s string, comment bool) error { for _, r := range s { switch { case r == '\t' || r == '\n': if !comment { return fail("text", "control %s in the declared author", codePoint(r)) } case r <= 0x1F || (r >= 0x7F && r <= 0x9F): return fail("text", "control %s", codePoint(r)) case (r >= 0x202A && r <= 0x202E) || (r >= 0x2066 && r <= 0x2069) || r == 0x061C || r == 0x200E || r == 0x200F: return fail("text", "bidirectional control %s", codePoint(r)) case r == 0x2028 || r == 0x2029: return fail("text", "separator %s", codePoint(r)) case r == 0xFEFF: return fail("text", "byte order mark %s", codePoint(r)) case isNoncharacter(r): return fail("text", "noncharacter %s", codePoint(r)) case DefaultIgnorable(r) && !whitelisted(r): return fail("text", "invisible %s", codePoint(r)) } } for i, line := range strings.Split(s, "\n") { if err := checkPlacement("text", line); err != nil { e := err.(*Error) e.Detail = fmt.Sprintf("line %d: %s", i+1, e.Detail) return e } } return nil } // isNoncharacter reports the 66 noncharacters: U+FDD0 to U+FDEF, and the last // two code points of each plane. func isNoncharacter(r rune) bool { return (r >= 0xFDD0 && r <= 0xFDEF) || r&0xFFFE == 0xFFFE }