diff --git a/internal/pathsafe/equivalence.go b/internal/pathsafe/equivalence.go new file mode 100644 index 0000000..3d9a30f --- /dev/null +++ b/internal/pathsafe/equivalence.go @@ -0,0 +1,51 @@ +package pathsafe + +import ( + "strings" + + "golang.org/x/text/unicode/norm" +) + +// SegmentsEquivalent reports whether two single path segments are portably +// equivalent for repository portability (PLAN.md Section 6.3). Each segment is +// normalized to Unicode NFC and then compared with strings.EqualFold. This +// deliberately treats case-only and NFC/NFD distinctions as collisions even on +// a host filesystem that could store both, preventing common APFS aliases from +// diverging across machines. +// +// EqualFold is a predicate rather than a canonicalizing transform, so callers +// must never derive a lowercase "key" from this comparison; use the pairwise +// helpers below instead. +func SegmentsEquivalent(a, b string) bool { + return strings.EqualFold(norm.NFC.String(a), norm.NFC.String(b)) +} + +// PathsEquivalent reports whether two complete segment lists collide: they must +// have the same length and every corresponding segment must be portably +// equivalent. +func PathsEquivalent(aSegments, bSegments []string) bool { + if len(aSegments) != len(bSegments) { + return false + } + for index, segment := range aSegments { + if !SegmentsEquivalent(segment, bSegments[index]) { + return false + } + } + return true +} + +// IsParentEquivalent reports whether parentSegments is a portable strict prefix +// of childSegments: the parent must be shorter, and every leading child segment +// must be portably equivalent to the corresponding parent segment. +func IsParentEquivalent(parentSegments, childSegments []string) bool { + if len(parentSegments) >= len(childSegments) { + return false + } + for index, segment := range parentSegments { + if !SegmentsEquivalent(segment, childSegments[index]) { + return false + } + } + return true +} diff --git a/internal/pathsafe/equivalence_test.go b/internal/pathsafe/equivalence_test.go new file mode 100644 index 0000000..99574b4 --- /dev/null +++ b/internal/pathsafe/equivalence_test.go @@ -0,0 +1,86 @@ +package pathsafe + +import "testing" + +func TestEquivalence(t *testing.T) { + scenarios := []struct { + name string + run func(*testing.T) + }{ + {"identical segments are equivalent", testIdenticalSegments}, + {"case differences are equivalent", testCaseDifferences}, + {"nfc and nfd forms are equivalent", testNormalForms}, + {"distinct segments are not equivalent", testDistinctSegments}, + {"equal length paths collide", testPathsCollideEqual}, + {"different length paths do not collide", testPathsDifferentLength}, + {"differing segment prevents collision", testPathsDifferingSegment}, + {"parent is strict prefix", testParentStrictPrefix}, + {"equal paths are not parent", testEqualNotParent}, + {"longer parent is rejected", testLongerParentRejected}, + } + for _, scenario := range scenarios { + t.Run(scenario.name, scenario.run) + } +} + +func testIdenticalSegments(t *testing.T) { + if !SegmentsEquivalent("config", "config") { + t.Fatal("identical segments must be equivalent") + } +} + +func testCaseDifferences(t *testing.T) { + if !SegmentsEquivalent("Config", "CONFIG") { + t.Fatal("case-only differences must be treated as equivalent") + } +} + +func testNormalForms(t *testing.T) { + composed := "\u00e9" + decomposed := "e\u0301" + if !SegmentsEquivalent(composed, decomposed) { + t.Fatal("NFC and NFD forms of the same code point must be equivalent") + } +} + +func testDistinctSegments(t *testing.T) { + if SegmentsEquivalent("config", "settings") { + t.Fatal("distinct segments must not be equivalent") + } +} + +func testPathsCollideEqual(t *testing.T) { + if !PathsEquivalent([]string{"config", "git"}, []string{"Config", "GIT"}) { + t.Fatal("equal-length portably-equivalent paths must collide") + } +} + +func testPathsDifferentLength(t *testing.T) { + if PathsEquivalent([]string{"config"}, []string{"config", "git"}) { + t.Fatal("different-length paths must not collide") + } +} + +func testPathsDifferingSegment(t *testing.T) { + if PathsEquivalent([]string{"config", "git"}, []string{"config", "shell"}) { + t.Fatal("paths differing in any segment must not collide") + } +} + +func testParentStrictPrefix(t *testing.T) { + if !IsParentEquivalent([]string{"config"}, []string{"config", "git", "ignore"}) { + t.Fatal("shorter equivalent prefix must be a parent") + } +} + +func testEqualNotParent(t *testing.T) { + if IsParentEquivalent([]string{"config", "git"}, []string{"config", "git"}) { + t.Fatal("equal paths must not be a parent relationship") + } +} + +func testLongerParentRejected(t *testing.T) { + if IsParentEquivalent([]string{"config", "git"}, []string{"config"}) { + t.Fatal("a longer parent must not be treated as a prefix") + } +} diff --git a/internal/pathsafe/path.go b/internal/pathsafe/path.go new file mode 100644 index 0000000..f0735c6 --- /dev/null +++ b/internal/pathsafe/path.go @@ -0,0 +1,133 @@ +// Package pathsafe owns the lexical, canonical, and portable-equivalence +// checks that keep every Cattery destination inside its allowed root. The +// rules come from PLAN.md Section 6: lexical validation (6.1), filesystem +// containment (6.2), and deployment collisions (6.3). +// +// Validation never silently rewrites unsafe input. A rejected path is reported +// verbatim alongside the reason it was refused, so a caller can surface the +// original user input rather than a cleaned substitute. +package pathsafe + +import ( + "path/filepath" + "strconv" + "strings" + "unicode/utf8" +) + +// PathError reports the original rejected input unchanged and the reason it was +// refused. The optional Cause preserves an underlying filesystem error so that +// errors.Is and errors.As keep working across the package boundary. +type PathError struct { + Input string + Reason string + Cause error +} + +// Error renders the reason, the verbatim input, and any wrapped cause. +func (e *PathError) Error() string { + if e == nil { + return "" + } + if e.Cause != nil { + return e.Reason + " " + strconv.Quote(e.Input) + ": " + e.Cause.Error() + } + return e.Reason + " " + strconv.Quote(e.Input) +} + +// Unwrap exposes the wrapped cause to errors.Is and errors.As. +func (e *PathError) Unwrap() error { + if e == nil { + return nil + } + return e.Cause +} + +// Segments validates a HOME-relative destination path and returns its segment +// list. The input must be a non-empty UTF-8 string without NUL bytes, without a +// leading separator or volume prefix, and without empty, ".", or ".." +// segments. Unsafe input is reported verbatim; it is never cleaned into a +// different accepted path. +func Segments(path string) ([]string, error) { + if reason := lexicalReason(path); reason != "" { + return nil, &PathError{Input: path, Reason: reason} + } + segments := strings.Split(path, "/") + for _, segment := range segments { + if reason := segmentReason(segment); reason != "" { + return nil, &PathError{Input: path, Reason: reason} + } + } + return segments, nil +} + +// GroupName validates a single repository group name. A group name is one +// non-empty filesystem segment: it may contain spaces and Unicode, is +// case-sensitive, and may not contain a slash or platform separator, a NUL +// byte, the reserved "." or ".." values, or a leading "." or "_" character. +func GroupName(name string) error { + if reason := groupNameReason(name); reason != "" { + return &PathError{Input: name, Reason: reason} + } + return nil +} + +// lexicalReason returns the first lexical rejection reason for path, or the +// empty string when the path passes the non-segment lexical checks. +func lexicalReason(path string) string { + switch { + case path == "": + return "empty path" + case !utf8.ValidString(path): + return "invalid utf-8" + case strings.ContainsRune(path, '\x00'): + return "nul byte" + case filepath.IsAbs(path): + return "absolute path" + case filepath.VolumeName(path) != "": + return "volume prefix" + } + return "" +} + +// segmentReason returns the rejection reason for a single split segment, or the +// empty string when the segment is acceptable. +func segmentReason(segment string) string { + switch segment { + case "": + return "empty segment" + case ".": + return "dot segment" + case "..": + return "dot-dot segment" + } + return "" +} + +// groupNameReason returns the first rejection reason for a group name, or the +// empty string when the name is acceptable. +func groupNameReason(name string) string { + switch { + case name == "": + return "empty group name" + case !utf8.ValidString(name): + return "invalid utf-8" + case strings.ContainsRune(name, '\x00'): + return "nul byte" + case strings.ContainsAny(name, "/\\"): + return "separator in group name" + case isReservedName(name): + return "reserved group name" + case hasReservedPrefix(name): + return "reserved group prefix" + } + return "" +} + +func isReservedName(name string) bool { + return name == "." || name == ".." +} + +func hasReservedPrefix(name string) bool { + return strings.HasPrefix(name, ".") || strings.HasPrefix(name, "_") +} diff --git a/internal/pathsafe/path_test.go b/internal/pathsafe/path_test.go new file mode 100644 index 0000000..c30ef97 --- /dev/null +++ b/internal/pathsafe/path_test.go @@ -0,0 +1,196 @@ +package pathsafe + +import ( + "strings" + "testing" +) + +func TestLexicalPath(t *testing.T) { + scenarios := []struct { + name string + run func(*testing.T) + }{ + {"accepts simple relative path", testAcceptsSimpleRelative}, + {"accepts nested relative path", testAcceptsNestedRelative}, + {"accepts unicode and spaces", testAcceptsUnicodeAndSpaces}, + {"rejects empty path", testRejectsEmptyPath}, + {"rejects absolute path", testRejectsAbsolutePath}, + {"rejects nul byte", testRejectsNulByte}, + {"rejects invalid utf-8", testRejectsInvalidUTF8}, + {"rejects empty segment", testRejectsEmptySegment}, + {"rejects dot segment", testRejectsDotSegment}, + {"rejects dot-dot segment", testRejectsDotDotSegment}, + {"rejects trailing separator", testRejectsTrailingSeparator}, + {"preserves original input in error", testPreservesOriginalInput}, + } + for _, scenario := range scenarios { + t.Run(scenario.name, scenario.run) + } +} + +func testAcceptsSimpleRelative(t *testing.T) { + segments, err := Segments(".bashrc") + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if len(segments) != 1 || segments[0] != ".bashrc" { + t.Fatalf("segments = %v", segments) + } +} + +func testAcceptsNestedRelative(t *testing.T) { + segments, err := Segments("config/git/ignore") + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + want := []string{"config", "git", "ignore"} + if len(segments) != len(want) { + t.Fatalf("segments = %v, want %v", segments, want) + } + for index, value := range want { + if segments[index] != value { + t.Fatalf("segments = %v, want %v", segments, want) + } + } +} + +func testAcceptsUnicodeAndSpaces(t *testing.T) { + if _, err := Segments("Bakgrund/bild namn"); err != nil { + t.Fatalf("unicode and spaces must be accepted: %v", err) + } +} + +func testRejectsEmptyPath(t *testing.T) { + assertRejected(t, "", "empty path") +} + +func testRejectsAbsolutePath(t *testing.T) { + assertRejected(t, "/home/user/file", "absolute path") +} + +func testRejectsNulByte(t *testing.T) { + assertRejected(t, "a\x00b", "nul byte") +} + +func testRejectsInvalidUTF8(t *testing.T) { + assertRejected(t, "bad\xff\xfe", "invalid utf-8") +} + +func testRejectsEmptySegment(t *testing.T) { + assertRejected(t, "a//b", "empty segment") +} + +func testRejectsDotSegment(t *testing.T) { + assertRejected(t, "./config", "dot segment") +} + +func testRejectsDotDotSegment(t *testing.T) { + assertRejected(t, "a/../b", "dot-dot segment") +} + +func testRejectsTrailingSeparator(t *testing.T) { + assertRejected(t, "config/", "empty segment") +} + +func testPreservesOriginalInput(t *testing.T) { + original := "a/../b" + _, err := Segments(original) + if err == nil { + t.Fatal("expected rejection") + } + if !strings.Contains(err.Error(), original) { + t.Fatalf("error %q omits original input %q", err.Error(), original) + } +} + +func assertRejected(t *testing.T, input, wantReason string) { + t.Helper() + segments, err := Segments(input) + if err == nil { + t.Fatalf("Segments(%q) = %v, want error containing %q", input, segments, wantReason) + } + pathError, ok := err.(*PathError) + if !ok { + t.Fatalf("error is %T, want *PathError", err) + } + if pathError.Input != input { + t.Fatalf("Input = %q, want %q", pathError.Input, input) + } + if pathError.Reason != wantReason { + t.Fatalf("Reason = %q, want %q", pathError.Reason, wantReason) + } +} + +func TestGroupName(t *testing.T) { + scenarios := []struct { + name string + run func(*testing.T) + }{ + {"accepts plain name", testGroupNameAcceptsPlain}, + {"accepts unicode and spaces", testGroupNameAcceptsUnicode}, + {"rejects empty", testGroupNameRejectsEmpty}, + {"rejects separator", testGroupNameRejectsSeparator}, + {"rejects dot value", testGroupNameRejectsDotValue}, + {"rejects dot-dot value", testGroupNameRejectsDotDotValue}, + {"rejects leading dot", testGroupNameRejectsLeadingDot}, + {"rejects leading underscore", testGroupNameRejectsLeadingUnderscore}, + {"rejects nul byte", testGroupNameRejectsNul}, + } + for _, scenario := range scenarios { + t.Run(scenario.name, scenario.run) + } +} + +func testGroupNameAcceptsPlain(t *testing.T) { + if err := GroupName("work"); err != nil { + t.Fatalf("unexpected error: %v", err) + } +} + +func testGroupNameAcceptsUnicode(t *testing.T) { + if err := GroupName("Mitt hem"); err != nil { + t.Fatalf("spaces and unicode must be accepted: %v", err) + } +} + +func testGroupNameRejectsEmpty(t *testing.T) { + if err := GroupName(""); err == nil { + t.Fatal("empty group name must be rejected") + } +} + +func testGroupNameRejectsSeparator(t *testing.T) { + if err := GroupName("a/b"); err == nil { + t.Fatal("separator in group name must be rejected") + } +} + +func testGroupNameRejectsDotValue(t *testing.T) { + if err := GroupName("."); err == nil { + t.Fatal("dot group name must be rejected") + } +} + +func testGroupNameRejectsDotDotValue(t *testing.T) { + if err := GroupName(".."); err == nil { + t.Fatal("dot-dot group name must be rejected") + } +} + +func testGroupNameRejectsLeadingDot(t *testing.T) { + if err := GroupName(".hidden"); err == nil { + t.Fatal("leading dot must be rejected") + } +} + +func testGroupNameRejectsLeadingUnderscore(t *testing.T) { + if err := GroupName("_control"); err == nil { + t.Fatal("leading underscore must be rejected") + } +} + +func testGroupNameRejectsNul(t *testing.T) { + if err := GroupName("a\x00b"); err == nil { + t.Fatal("nul byte must be rejected") + } +}