You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
DateKeys/internal/pathrule/rules.go

397 lines
11 KiB

1 week ago
package pathrule
import (
"fmt"
"regexp"
"strings"
"unicode/utf8"
)
// Limits of the paths and texts of a format 3 head, fixed with the format
// (spec §29.4, §29.5).
const (
MaxPathLen = 1024 // R1, bytes
MaxSegments = 32 // R2
MaxSegmentLen = 255 // R3, bytes
MaxSegmentUTF16 = 255 // R3, UTF-16 code units of the NFD of the segment
MaxImplicitDirs = 65535 // R9
MaxCommentLen = 16384 // bytes
MaxAuthorLen = 256 // bytes
reservedDirPrefix = ".datekeys-"
)
// Error is the violation of one rule of spec §29.5 or §29.6. Its text is
// the same in every implementation, so that the message of a rejected head
// does not depend on the reader.
type Error struct {
Rule string // "R2" to "R10", "text"
Detail string
Format 3, step 4: the writer EncryptFiles writes format 3 (spec 29.2 to 29.6, 61, 62, 62.1): the files of a list of Sources, each read twice, with the comment and the declared author. - Before anything is written: the paths and the texts are checked with the rules of the reader, in the words of a writer, naming the rule and the character, and the two paths of an R7 collision (rule 15); the comment has its CR LF and lone CR turned into LF (29.6); L is measured with a head whose salt and SHA-256 are zero, as long as the final one, and the first reading hashes each file, which must have exactly its Size. - The files go in the byte order of their paths (R8), whatever the order of the Sources; the mtime is kept only from 1970 to 9999, never clipped (rule 16); at least one file or a comment (rule 14). - The head, with a fresh salt, the control and the security area are decoded with the rules of the reader before sealing (rule 17), and the frame is checked against L. The area is 512 bytes with the empty security, whatever the options (rule 13). - The second reading writes each file into PAYLOAD_AGE and fails if its size or SHA-256 changed (rule 18). - Encrypt and EncryptFiles share the sealing; Encrypt writes format 2 only with the new TestVectors option (rule 1), and takes no head. The test data generators set it, and so does the CLI until step 5 moves it to EncryptFiles. - Result.Head is the head written. DecodeHead keeps the check of the critical extensions apart, so that the self-check decodes the head as the one of the control does. - The examples and the live test write with EncryptFiles. - The reader tests had a literal U+202E, now escaped. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
1 week ago
// Paths are the positions, from 1, of the two paths of an R7
// violation, the later one first, so that a writer can name them; zero
// for the other rules.
Paths [2]int
1 week ago
}
func (e *Error) Error() string { return e.Rule + ": " + e.Detail }
func fail(rule, format string, args ...any) error {
return &Error{Rule: rule, Detail: fmt.Sprintf(format, args...)}
}
// codePoint formats r as in the spec: U+ and at least four upper-case
// hexadecimal digits.
func codePoint(r rune) string { return fmt.Sprintf("U+%04X", r) }
// forbiddenASCII are the ASCII characters of R4 that are not controls.
const forbiddenASCII = `"*:<>?\|`
// CheckPath checks one path of the head with the rules of the fourth layer
// that concern it alone, in this order: R2, and then for each segment R3,
// R4, R4b, R5, R6, R6b and R6c; then R10. R1 and R8 belong to the third
// layer, and R7 and R9 to the whole tree (CheckTree). The first violation
// is returned.
func CheckPath(path string) error {
segs := strings.Split(path, "/")
if len(segs) > MaxSegments {
return fail("R2", "%d segments, more than %d", len(segs), MaxSegments)
}
for i, s := range segs {
if s == "" {
return fail("R2", "segment %d is empty", i+1)
}
}
for i, s := range segs {
if err := checkSegment(s); err != nil {
e := err.(*Error)
e.Detail = fmt.Sprintf("segment %d: %s", i+1, e.Detail)
return e
}
}
if strings.HasPrefix(Key(segs[0]), reservedDirPrefix) {
return fail("R10", "the first segment starts with %q", reservedDirPrefix)
}
return nil
}
// checkSegment applies R3, R4, R4b, R5, R6, R6b and R6c to a segment.
func checkSegment(s string) error {
if err := checkR3(s); err != nil {
return err
}
for _, r := range s {
if err := checkR4(r); err != nil {
return err
}
}
if err := checkPlacement("R4b", s); err != nil {
return err
}
if err := checkR5(s); err != nil {
return err
}
if err := checkR6(s); err != nil {
return err
}
if err := checkR6b(s); err != nil {
return err
}
return checkR6c(s)
}
func checkR3(s string) error {
if len(s) > MaxSegmentLen {
return fail("R3", "%d bytes, more than %d", len(s), MaxSegmentLen)
}
switch s {
case ".":
return fail("R3", "the segment is a dot")
case "..":
return fail("R3", "the segment is two dots")
}
switch stripWhitelist(s) {
case "":
return fail("R3", "the segment is empty without ZWNJ, ZWJ, VS15 and VS16")
case ".":
return fail("R3", "the segment is a dot without ZWNJ, ZWJ, VS15 and VS16")
case "..":
return fail("R3", "the segment is two dots without ZWNJ, ZWJ, VS15 and VS16")
}
if n := utf16Len(NFD(s)); n > MaxSegmentUTF16 {
return fail("R3", "its NFD is %d UTF-16 code units, more than %d", n, MaxSegmentUTF16)
}
return nil
}
func utf16Len(s string) int {
n := 0
for _, r := range s {
if r >= 0x10000 {
n += 2
} else {
n++
}
}
return n
}
// checkR4 rejects the code points of R4.
func checkR4(r rune) error {
switch {
case r <= 0x1F || (r >= 0x7F && r <= 0x9F):
return fail("R4", "control %s", codePoint(r))
case strings.ContainsRune(forbiddenASCII, r):
return fail("R4", "character %s", codePoint(r))
case r == 0x2028 || r == 0x2029:
return fail("R4", "separator %s", codePoint(r))
case DefaultIgnorable(r) && !whitelisted(r):
return fail("R4", "invisible %s", codePoint(r))
case r >= 0xF000 && r <= 0xF0FF:
return fail("R4", "private use %s", codePoint(r))
case !Assigned(r):
return fail("R4", "unassigned %s", codePoint(r))
}
return nil
}
// checkPlacement applies R4b to s, a segment or a line of a text: VS15 and
// VS16 only right after a character that forms an emoji variation sequence
// with them, and ZWJ and ZWNJ never first, last or right after another ZWJ
// or ZWNJ.
func checkPlacement(rule, s string) error {
prev, i := rune(-1), 0
for _, r := range s {
switch r {
case vs15, vs16:
if prev < 0 || !allowsSelector(prev, r) {
return fail(rule, "%s is not part of an emoji variation sequence", codePoint(r))
}
case zwj, zwnj:
switch {
case prev < 0:
return fail(rule, "%s at the start", codePoint(r))
case prev == zwj || prev == zwnj:
return fail(rule, "%s right after %s", codePoint(r), codePoint(prev))
case i+utf8.RuneLen(r) == len(s):
return fail(rule, "%s at the end", codePoint(r))
}
}
prev = r
i += utf8.RuneLen(r)
}
return nil
}
func checkR5(s string) error {
switch {
case strings.HasPrefix(s, " "):
return fail("R5", "the segment starts with U+0020")
case strings.HasSuffix(s, " "):
return fail("R5", "the segment ends with U+0020")
case strings.HasSuffix(s, "."):
return fail("R5", "the segment ends with '.'")
}
return nil
}
// reserved are the device names of R6, in upper case.
var reserved = map[string]bool{
"CON": true, "PRN": true, "AUX": true, "NUL": true, "CONIN$": true, "CONOUT$": true,
"COM¹": true, "COM²": true, "COM³": true, "LPT¹": true, "LPT²": true, "LPT³": true,
}
func init() {
for d := '0'; d <= '9'; d++ {
reserved["COM"+string(d)] = true
reserved["LPT"+string(d)] = true
}
}
func checkR6(s string) error {
base, _, _ := strings.Cut(s, ".")
base = strings.TrimRight(base, " ")
if name := asciiUpper(base); reserved[name] {
return fail("R6", "%s is a reserved device name", name)
}
return nil
}
// asciiUpper upper-cases ASCII letters only.
func asciiUpper(s string) string {
return strings.Map(func(r rune) rune {
if r >= 'a' && r <= 'z' {
return r - 'a' + 'A'
}
return r
}, s)
}
// shortNameBase is the base of an 8.3 alias in R6b: it ends in '~' and 1 to
// 6 ASCII digits. It has no lookahead, which the regexp package lacks.
var shortNameBase = regexp.MustCompile(`^[^.]*~[0-9]{1,6}$`)
func checkR6b(s string) error {
if strings.Count(s, ".") > 1 {
return nil
}
base, ext, _ := strings.Cut(s, ".")
if n := utf8.RuneCountInString(base); n < 1 || n > 8 {
return nil
}
if utf8.RuneCountInString(ext) > 3 || !shortNameBase.MatchString(base) {
return nil
}
return fail("R6b", "the segment has the form of an 8.3 alias")
}
// checkR6c projects the segment with each best-fit table: every non-ASCII
// code point the table maps to an ASCII byte is replaced by that byte. A
// projection must not hold '/', '\', ':' or U+0000, and must pass R3, R5, R6
// and R6b.
func checkR6c(s string) error {
for i := range bestFitTables {
t := &bestFitTables[i]
p, changed := project(s, t)
if !changed {
continue
}
if j := strings.IndexAny(p, "/\\:\x00"); j >= 0 {
return fail("R6c", "code page %s maps the segment to one with %s", t.name, codePoint(rune(p[j])))
}
for _, check := range []func(string) error{checkR3, checkR5, checkR6, checkR6b} {
if err := check(p); err != nil {
e := err.(*Error)
return fail("R6c", "code page %s maps the segment to one that breaks %s: %s", t.name, e.Rule, e.Detail)
}
}
}
return nil
}
func project(s string, t *bestFitTable) (string, bool) {
var b strings.Builder
changed := false
for _, r := range s {
if r >= 0x80 {
if c, ok := t.lookup(r); ok {
b.WriteByte(c)
changed = true
continue
}
}
b.WriteRune(r)
}
return b.String(), changed
}
// CheckTree applies R7 and then R9 to the paths of a head, which have passed
// CheckPath: no two siblings with the same key, no path that is both a file
// and a folder, comparing keys, and at most MaxImplicitDirs folders. Paths
// are numbered from 1 in the messages.
func CheckTree(paths []string) error {
type node struct {
name string
dir bool
path int // the first path that reached it
}
// children maps the key of a folder, "" for the root, to its children by
// key; the key of a folder joins the keys of its segments with '/'.
children := map[string]map[string]node{}
for n, path := range paths {
segs := strings.Split(path, "/")
parent := ""
for i, s := range segs {
k := Key(s)
isDir := i < len(segs)-1
kids := children[parent]
if kids == nil {
kids = map[string]node{}
children[parent] = kids
}
switch old, ok := kids[k]; {
case !ok:
kids[k] = node{name: s, dir: isDir, path: n + 1}
case old.name != s:
Format 3, step 4: the writer EncryptFiles writes format 3 (spec 29.2 to 29.6, 61, 62, 62.1): the files of a list of Sources, each read twice, with the comment and the declared author. - Before anything is written: the paths and the texts are checked with the rules of the reader, in the words of a writer, naming the rule and the character, and the two paths of an R7 collision (rule 15); the comment has its CR LF and lone CR turned into LF (29.6); L is measured with a head whose salt and SHA-256 are zero, as long as the final one, and the first reading hashes each file, which must have exactly its Size. - The files go in the byte order of their paths (R8), whatever the order of the Sources; the mtime is kept only from 1970 to 9999, never clipped (rule 16); at least one file or a comment (rule 14). - The head, with a fresh salt, the control and the security area are decoded with the rules of the reader before sealing (rule 17), and the frame is checked against L. The area is 512 bytes with the empty security, whatever the options (rule 13). - The second reading writes each file into PAYLOAD_AGE and fails if its size or SHA-256 changed (rule 18). - Encrypt and EncryptFiles share the sealing; Encrypt writes format 2 only with the new TestVectors option (rule 1), and takes no head. The test data generators set it, and so does the CLI until step 5 moves it to EncryptFiles. - Result.Head is the head written. DecodeHead keeps the check of the critical extensions apart, so that the self-check decodes the head as the one of the control does. - The examples and the live test write with EncryptFiles. - The reader tests had a literal U+202E, now escaped. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
1 week ago
return &Error{Rule: "R7", Detail: fmt.Sprintf("path %d collides with path %d in segment %d", n+1, old.path, i+1), Paths: [2]int{n + 1, old.path}}
1 week ago
case old.dir != isDir || !isDir:
Format 3, step 4: the writer EncryptFiles writes format 3 (spec 29.2 to 29.6, 61, 62, 62.1): the files of a list of Sources, each read twice, with the comment and the declared author. - Before anything is written: the paths and the texts are checked with the rules of the reader, in the words of a writer, naming the rule and the character, and the two paths of an R7 collision (rule 15); the comment has its CR LF and lone CR turned into LF (29.6); L is measured with a head whose salt and SHA-256 are zero, as long as the final one, and the first reading hashes each file, which must have exactly its Size. - The files go in the byte order of their paths (R8), whatever the order of the Sources; the mtime is kept only from 1970 to 9999, never clipped (rule 16); at least one file or a comment (rule 14). - The head, with a fresh salt, the control and the security area are decoded with the rules of the reader before sealing (rule 17), and the frame is checked against L. The area is 512 bytes with the empty security, whatever the options (rule 13). - The second reading writes each file into PAYLOAD_AGE and fails if its size or SHA-256 changed (rule 18). - Encrypt and EncryptFiles share the sealing; Encrypt writes format 2 only with the new TestVectors option (rule 1), and takes no head. The test data generators set it, and so does the CLI until step 5 moves it to EncryptFiles. - Result.Head is the head written. DecodeHead keeps the check of the critical extensions apart, so that the self-check decodes the head as the one of the control does. - The examples and the live test write with EncryptFiles. - The reader tests had a literal U+202E, now escaped. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
1 week ago
return &Error{Rule: "R7", Detail: fmt.Sprintf("path %d makes a file of path %d a folder, or the reverse, in segment %d", n+1, old.path, i+1), Paths: [2]int{n + 1, old.path}}
1 week ago
}
if parent == "" {
parent = k
} else {
parent += "/" + k
}
}
}
dirs := map[string]bool{}
for _, path := range paths {
for i := 0; ; {
j := strings.IndexByte(path[i:], '/')
if j < 0 {
break
}
i += j
dirs[path[:i]] = true
i++
}
}
if len(dirs) > MaxImplicitDirs {
return fail("R9", "%d folders, more than %d", len(dirs), MaxImplicitDirs)
}
return nil
}
// CheckComment checks the comment of a head (spec §29.6).
func CheckComment(s string) error {
return checkText(s, true)
}
// CheckAuthor checks the declared author of a head (spec §29.6).
func CheckAuthor(s string) error {
if err := checkText(s, false); err != nil {
return err
}
if strings.HasPrefix(s, " ") || strings.HasSuffix(s, " ") {
return fail("text", "the declared author starts or ends with U+0020")
}
return nil
}
// checkText applies the rules of spec §29.6, code point by code point, and
// then R4b to each line.
func checkText(s string, comment bool) error {
for _, r := range s {
switch {
case r == '\t' || r == '\n':
if !comment {
return fail("text", "control %s in the declared author", codePoint(r))
}
case r <= 0x1F || (r >= 0x7F && r <= 0x9F):
return fail("text", "control %s", codePoint(r))
case (r >= 0x202A && r <= 0x202E) || (r >= 0x2066 && r <= 0x2069) || r == 0x061C || r == 0x200E || r == 0x200F:
return fail("text", "bidirectional control %s", codePoint(r))
case r == 0x2028 || r == 0x2029:
return fail("text", "separator %s", codePoint(r))
case r == 0xFEFF:
return fail("text", "byte order mark %s", codePoint(r))
case isNoncharacter(r):
return fail("text", "noncharacter %s", codePoint(r))
case DefaultIgnorable(r) && !whitelisted(r):
return fail("text", "invisible %s", codePoint(r))
}
}
for i, line := range strings.Split(s, "\n") {
if err := checkPlacement("text", line); err != nil {
e := err.(*Error)
e.Detail = fmt.Sprintf("line %d: %s", i+1, e.Detail)
return e
}
}
return nil
}
// isNoncharacter reports the 66 noncharacters: U+FDD0 to U+FDEF, and the last
// two code points of each plane.
func isNoncharacter(r rune) bool {
return (r >= 0xFDD0 && r <= 0xFDEF) || r&0xFFFE == 0xFFFE
}

Powered by TurnKey Linux.