From 25ab0e7cbfe48796c78b02cbb2382535e661f551 Mon Sep 17 00:00:00 2001 From: Oluwatobi Ogundimu Date: Fri, 31 Jul 2026 16:08:21 +0100 Subject: [PATCH] SC-7: include/exclude filter rule matching engine Wildcard matching (*, **, ?), directory-only patterns, first-match-wins evaluation, --exclude-from/--include-from, and --filter merge support. Anchoring follows rsync's real rule: leading /, any internal /, or ** all anchor to root; only a bare filename matches at any depth. Filtering runs as a post-pass over Walk()'s output, keeping SC-4's traversal untouched and independently testable. --- README.md | 37 +++- internal/cli/root.go | 49 +++-- internal/sync/filter.go | 351 +++++++++++++++++++++++++++++++ internal/sync/filter_test.go | 395 +++++++++++++++++++++++++++++++++++ 4 files changed, 807 insertions(+), 25 deletions(-) create mode 100644 internal/sync/filter.go create mode 100644 internal/sync/filter_test.go diff --git a/README.md b/README.md index 145845b..5984d2b 100644 --- a/README.md +++ b/README.md @@ -4,9 +4,10 @@ An rsync-inspired file synchronization tool written in Go. ## Status -CLI parsing and file enumeration are implemented; data transfer is not. -`internal/sync` builds a sorted file list (`sync.Walk`), but nothing calls -it yet — the CLI only echoes parsed flags — and `internal/transport` is +CLI parsing, file enumeration, and filter-rule matching are implemented; +data transfer is not. `internal/sync` builds a sorted file list +(`sync.Walk`) and can filter it (`sync.FilterEntries`), but nothing calls +either yet — the CLI only echoes parsed flags — and `internal/transport` is still empty. ## Build @@ -43,10 +44,12 @@ argument is always the destination. | `--exclude PATTERN` | | exclude matching files (repeatable) | | `--include PATTERN` | | include matching files (repeatable) | | `--filter RULE` | | add a filter rule (repeatable) | +| `--exclude-from FILE` | | read exclude patterns from FILE, one per line (repeatable) | +| `--include-from FILE` | | read include patterns from FILE, one per line (repeatable) | -`--exclude`/`--include`/`--filter` share one ordered rule list — their -relative order on the command line is preserved, matching rsync's -first-match-wins semantics. +All five filter-related flags share one ordered rule list — their relative +order on the command line is preserved, matching rsync's first-match-wins +semantics. See [Filter Rules](#filter-rules) below. ## File Enumeration @@ -65,11 +68,31 @@ target). Symlinks are captured via `Lstat`, never followed. On Windows, `UID`/`GID` are always `0` — there's no POSIX ownership concept to read, so `0` means "unavailable," not a real value. +## Filter Rules + +`sync.CompileRules` turns the ordered `--exclude`/`--include`/`--filter`/ +`--exclude-from`/`--include-from` list into ready-to-match rules; +`sync.Included`/`sync.FilterEntries` apply them to `sync.Walk`'s output as +a separate pass, first-match-wins, defaulting to include when nothing +matches. + +Pattern syntax: `*` matches within one path segment, `**` crosses segment +boundaries, `?` matches one character. A trailing `/` makes a pattern match +directories only. `--filter` also accepts `merge FILE` to inline another +rule file at that point in the list (one level deep — a merge file that +itself tries to merge another file is an error, not silently ignored). + +A pattern anchors to the transfer root — matched once against the full +path, not tried at every depth — if it has a leading `/`, contains any +other `/`, or contains `**`. Only a pattern with none of those (a bare +filename like `*.log`) matches at any depth, against the final path +component only. This matches real rsync's actual anchoring rule. + ## Architecture - `cmd/grsync` — CLI entrypoint. - `internal/cli` — flag/argument parsing (built on cobra). -- `internal/sync` — file-list generation today; comparison/delta logic later. +- `internal/sync` — file-list generation and filter matching today; comparison/delta logic later. - `internal/transport` — (placeholder) data movement, local and remote. Goal: full feature parity with upstream rsync, including protocol/format diff --git a/internal/cli/root.go b/internal/cli/root.go index 6e80305..47b23c0 100644 --- a/internal/cli/root.go +++ b/internal/cli/root.go @@ -15,20 +15,24 @@ import ( // FilterRuleType identifies which kind of rule a FilterRule represents. type FilterRuleType string -// The three rule kinds grsync's flags can produce. Kept as their own type -// (rather than a bare string) so callers can't accidentally pass an -// arbitrary value through. +// The rule kinds grsync's filter-related flags can produce. Kept as their +// own type (rather than a bare string) so callers can't accidentally pass +// an arbitrary value through. const ( - FilterRuleInclude FilterRuleType = "include" - FilterRuleExclude FilterRuleType = "exclude" - FilterRuleFilter FilterRuleType = "filter" + FilterRuleInclude FilterRuleType = "include" + FilterRuleExclude FilterRuleType = "exclude" + FilterRuleFilter FilterRuleType = "filter" + FilterRuleExcludeFrom FilterRuleType = "exclude-from" + FilterRuleIncludeFrom FilterRuleType = "include-from" ) -// FilterRule is a single --include/--exclude/--filter rule. rsync treats -// these three flags as one ordered, first-match-wins rule list rather than -// three independent lists, so grsync collects them the same way: Type -// records which flag produced the rule, and relative order across *all* -// three flags is preserved in the order the user supplied them. +// FilterRule is a single --include/--exclude/--filter/--exclude-from/ +// --include-from occurrence. rsync treats all of these as one ordered, +// first-match-wins rule list rather than independent lists, so grsync +// collects them the same way: Type records which flag produced the rule, +// and relative order across *all* of them is preserved in the order the +// user supplied them. For the two "-from" kinds, Pattern is a file path, +// not a filter pattern — internal/sync reads and expands it. type FilterRule struct { Type FilterRuleType Pattern string @@ -49,11 +53,12 @@ type options struct { filterRules []FilterRule } -// filterRuleFlag implements pflag.Value. Each of --exclude/--include/--filter -// gets its own instance, fixed to a single FilterRuleType, but all three -// share the same backing slice — so pflag's normal "call Set once per -// occurrence" behavior naturally builds one ordered rule list regardless of -// which of the three flag names was used at each position. +// filterRuleFlag implements pflag.Value. Each of --exclude/--include/ +// --filter/--exclude-from/--include-from gets its own instance, fixed to a +// single FilterRuleType, but all of them share the same backing slice — so +// pflag's normal "call Set once per occurrence" behavior naturally builds +// one ordered rule list regardless of which flag name was used at each +// position. type filterRuleFlag struct { ruleType FilterRuleType rules *[]FilterRule @@ -67,10 +72,14 @@ func (f *filterRuleFlag) Set(pattern string) error { } func (f *filterRuleFlag) Type() string { - if f.ruleType == FilterRuleFilter { + switch f.ruleType { + case FilterRuleFilter: return "rule" + case FilterRuleExcludeFrom, FilterRuleIncludeFrom: + return "file" + default: + return "pattern" } - return "pattern" } // NewRootCmd builds the root grsync command. It is exported as a @@ -106,6 +115,10 @@ func NewRootCmd() *cobra.Command { "include", "include files matching PATTERN (repeatable, order preserved relative to --exclude/--filter)") flags.Var(&filterRuleFlag{ruleType: FilterRuleFilter, rules: &opts.filterRules}, "filter", "add a file-filtering RULE (repeatable, order preserved relative to --exclude/--include)") + flags.Var(&filterRuleFlag{ruleType: FilterRuleExcludeFrom, rules: &opts.filterRules}, + "exclude-from", "read exclude patterns from FILE, one per line (repeatable, order preserved)") + flags.Var(&filterRuleFlag{ruleType: FilterRuleIncludeFrom, rules: &opts.filterRules}, + "include-from", "read include patterns from FILE, one per line (repeatable, order preserved)") return cmd } diff --git a/internal/sync/filter.go b/internal/sync/filter.go new file mode 100644 index 0000000..30e2415 --- /dev/null +++ b/internal/sync/filter.go @@ -0,0 +1,351 @@ +package sync + +import ( + "bufio" + "fmt" + "os" + "path" + "strings" +) + +// RuleKind identifies which flag produced a raw filter rule, before any +// pattern parsing or file expansion happens. It mirrors the shape of +// internal/cli's FilterRule (Type + Pattern) but is defined independently: +// internal/sync must never import internal/cli, since cli is expected to +// depend on sync (not the other way around) once they're wired together. +type RuleKind string + +const ( + // RuleInclude is a direct --include pattern. + RuleInclude RuleKind = "include" + // RuleExclude is a direct --exclude pattern. + RuleExclude RuleKind = "exclude" + // RuleFilter is a raw --filter rule line, e.g. "+ *.txt", "- .git/", + // or "merge FILE" — see parseFilterLine for the subset of rsync's + // filter-rule syntax this supports. + RuleFilter RuleKind = "filter" + // RuleExcludeFrom has a Pattern that is a file path, not a filter + // pattern itself. CompileRules reads that file and inserts one + // exclude rule per line at this exact position in the list, + // preserving overall command-line order rather than appending + // everything to the end. + RuleExcludeFrom RuleKind = "exclude-from" + // RuleIncludeFrom is RuleExcludeFrom's --include-from counterpart. + RuleIncludeFrom RuleKind = "include-from" +) + +// RawRule is a single --include/--exclude occurrence, in command-line +// order, before any pattern parsing happens. +type RawRule struct { + Kind RuleKind + Pattern string +} + +// Action is what a compiled Rule does when its pattern matches an entry. +type Action int + +const ( + // Include means a matching entry is kept. + Include Action = iota + // Exclude means a matching entry is dropped. + Exclude +) + +// Rule is a single compiled, ready-to-match filter rule. Pattern has +// already had its leading "/" (Anchored) marker stripped by CompileRules; +// it never contains that marker itself. +// +// Anchored matches real rsync's actual rule: a pattern anchors to the +// transfer root if it has a leading "/", contains any other "/", or +// contains "**" — only a pattern with none of those (a bare filename, e.g. +// "*.log") matches at any depth, against the final path component only. +type Rule struct { + Action Action + Pattern string + Anchored bool + // DirOnly means this rule only ever matches directories — set from a + // trailing "/" on the original pattern, stripped by CompileRules just + // like the anchor marker. + DirOnly bool +} + +// matches reports whether entryPath (a directory if isDir) matches r. +func (r Rule) matches(entryPath string, isDir bool) bool { + if r.DirOnly && !isDir { + return false + } + + patternSegs := strings.Split(r.Pattern, "/") + pathSegs := strings.Split(entryPath, "/") + + if r.Anchored { + return matchSegments(patternSegs, pathSegs) + } + + // Unanchored: the pattern may match starting at any depth, as if + // "**/" were implicitly prepended to it. + for start := 0; start <= len(pathSegs); start++ { + if matchSegments(patternSegs, pathSegs[start:]) { + return true + } + } + return false +} + +// matchSegments matches a "/"-split pattern against a "/"-split path, +// segment by segment. Every segment except "**" is matched with +// path.Match, which already gives us "*" (any run of characters within the +// segment) and "?" (exactly one character) for free, without needing a +// hand-rolled matcher of our own. "**" is handled explicitly: it may +// consume zero or more path segments, so both possibilities (consume none +// and keep matching, or consume one and recurse) are tried. +func matchSegments(patternSegs, pathSegs []string) bool { + if len(patternSegs) == 0 { + return len(pathSegs) == 0 + } + + if patternSegs[0] == "**" { + if matchSegments(patternSegs[1:], pathSegs) { + return true + } + if len(pathSegs) == 0 { + return false + } + return matchSegments(patternSegs, pathSegs[1:]) + } + + if len(pathSegs) == 0 { + return false + } + ok, err := path.Match(patternSegs[0], pathSegs[0]) + if err != nil || !ok { + return false + } + return matchSegments(patternSegs[1:], pathSegs[1:]) +} + +// compilePattern parses a single raw pattern's anchor ("/" prefix) and +// dir-only ("/" suffix) markers into a ready-to-match Rule with the given +// action. Shared by direct --include/--exclude rules, each line read from +// an --exclude-from/--include-from file, and --filter rule lines. +// +// A pattern containing any empty "/"-separated segment — a bare "/" or "" +// overall, or an internal "//" typo like "a//b" — is rejected rather than +// silently compiled: no real FileEntry.Path segment is ever empty (Walk() +// never produces one), so a Rule requiring an empty segment could never +// match anything. Compiling it anyway would leave the user with a filter +// rule that silently does nothing forever, which is worse than a clear +// error at compile time. +func compilePattern(action Action, pattern string) (Rule, error) { + leadingSlash := strings.HasPrefix(pattern, "/") + if leadingSlash { + pattern = pattern[1:] + } + dirOnly := strings.HasSuffix(pattern, "/") + if dirOnly { + pattern = pattern[:len(pattern)-1] + } + for _, seg := range strings.Split(pattern, "/") { + if seg == "" { + return Rule{}, fmt.Errorf("filter pattern has an empty path segment (check for a stray \"//\" or a bare %q): %q", "/", pattern) + } + } + + // A leading "/" always anchors. So does any *other* "/" still present + // in the pattern, or a "**" anywhere in it — matching real rsync's + // rule, not just the "leading slash only" simplification this started + // as. Only a genuinely slash-free, "**"-free pattern (a bare filename) + // matches at any depth. + anchored := leadingSlash || strings.Contains(pattern, "/") || strings.Contains(pattern, "**") + + return Rule{Action: action, Pattern: pattern, Anchored: anchored, DirOnly: dirOnly}, nil +} + +// readPatternFile reads one pattern per line from path, for +// --exclude-from/--include-from. Blank lines and lines starting with "#" +// or ";" are skipped, matching rsync's own filter-file comment convention. +func readPatternFile(path string) (patterns []string, err error) { + f, err := os.Open(path) + if err != nil { + return nil, err + } + defer func() { + if cerr := f.Close(); cerr != nil && err == nil { + err = cerr + } + }() + + scanner := bufio.NewScanner(f) + for scanner.Scan() { + line := strings.TrimSpace(scanner.Text()) + if line == "" || strings.HasPrefix(line, "#") || strings.HasPrefix(line, ";") { + continue + } + patterns = append(patterns, line) + } + if serr := scanner.Err(); serr != nil { + return nil, serr + } + return patterns, nil +} + +// parsedFilterLine is the result of parsing one line of --filter RULE +// syntax — whether it came directly from a --filter flag or from a line +// inside a merge file. +type parsedFilterLine struct { + isMerge bool + mergeFile string + rule Rule // valid only when !isMerge +} + +// parseFilterLine implements a deliberately small subset of rsync's +// --filter rule syntax: "+ PATTERN" / "- PATTERN" (and their word-form +// equivalents "include PATTERN" / "exclude PATTERN"), plus "merge FILE". +// Real rsync's filter language is much larger (modifiers like "-C", +// "dir-merge" with per-directory semantics, "!", exclude-if-present rules, +// and more) — none of that is implemented here; anything outside this +// subset is a hard parse error rather than a silent no-op. +func parseFilterLine(line string) (parsedFilterLine, error) { + switch { + case strings.HasPrefix(line, "+ "): + rule, err := compilePattern(Include, strings.TrimSpace(line[2:])) + return parsedFilterLine{rule: rule}, err + case strings.HasPrefix(line, "- "): + rule, err := compilePattern(Exclude, strings.TrimSpace(line[2:])) + return parsedFilterLine{rule: rule}, err + case strings.HasPrefix(line, "include "): + rule, err := compilePattern(Include, strings.TrimSpace(line[len("include "):])) + return parsedFilterLine{rule: rule}, err + case strings.HasPrefix(line, "exclude "): + rule, err := compilePattern(Exclude, strings.TrimSpace(line[len("exclude "):])) + return parsedFilterLine{rule: rule}, err + case strings.HasPrefix(line, "merge "): + return parsedFilterLine{isMerge: true, mergeFile: strings.TrimSpace(line[len("merge "):])}, nil + default: + return parsedFilterLine{}, fmt.Errorf( + "unsupported filter rule syntax: %q (supported: \"+ PATTERN\", \"- PATTERN\", \"include PATTERN\", \"exclude PATTERN\", \"merge FILE\")", + line) + } +} + +// expandMergeFile reads path and parses each non-comment, non-blank line +// with parseFilterLine — the same syntax --filter itself accepts. Nested +// merge (a merge file whose own lines include another "merge OTHER" +// directive) is deliberately unsupported: rather than silently recursing +// (risking an infinite loop on a self-referential merge file) or silently +// dropping the nested directive, it's a hard error. This covers the +// "basic merge case"; full nested/dir-merge support is a larger feature +// left for later if it turns out to be needed. +func expandMergeFile(path string) ([]Rule, error) { + lines, err := readPatternFile(path) + if err != nil { + return nil, err + } + + rules := make([]Rule, 0, len(lines)) + for _, line := range lines { + parsed, err := parseFilterLine(line) + if err != nil { + return nil, err + } + if parsed.isMerge { + return nil, fmt.Errorf("nested merge directives are not supported (found %q inside %q)", line, path) + } + rules = append(rules, parsed.rule) + } + return rules, nil +} + +// CompileRules converts raw rules — --include/--exclude patterns, +// --filter rule lines (including "merge FILE"), and --exclude-from/ +// --include-from file references — into a single ordered, ready-to-match +// Rule list. From-file and merged rules are expanded in place at the +// position their flag occurred, so e.g. a --exclude-from sandwiched +// between two direct --exclude flags on the command line stays sandwiched +// between their compiled Rules, not appended after them. +func CompileRules(raw []RawRule) ([]Rule, error) { + var rules []Rule + for _, r := range raw { + switch r.Kind { + case RuleInclude: + rule, err := compilePattern(Include, r.Pattern) + if err != nil { + return nil, err + } + rules = append(rules, rule) + case RuleExclude: + rule, err := compilePattern(Exclude, r.Pattern) + if err != nil { + return nil, err + } + rules = append(rules, rule) + case RuleFilter: + parsed, err := parseFilterLine(strings.TrimSpace(r.Pattern)) + if err != nil { + return nil, err + } + if !parsed.isMerge { + rules = append(rules, parsed.rule) + continue + } + merged, err := expandMergeFile(parsed.mergeFile) + if err != nil { + return nil, fmt.Errorf("merging filter file %q: %w", parsed.mergeFile, err) + } + rules = append(rules, merged...) + case RuleExcludeFrom, RuleIncludeFrom: + action := Exclude + if r.Kind == RuleIncludeFrom { + action = Include + } + patterns, err := readPatternFile(r.Pattern) + if err != nil { + return nil, fmt.Errorf("reading %s file %q: %w", r.Kind, r.Pattern, err) + } + for _, p := range patterns { + rule, err := compilePattern(action, p) + if err != nil { + return nil, fmt.Errorf("in %s file %q: %w", r.Kind, r.Pattern, err) + } + rules = append(rules, rule) + } + default: + return nil, fmt.Errorf("unsupported rule kind %q", r.Kind) + } + } + return rules, nil +} + +// Included evaluates rules against a single FileEntry using rsync's +// first-match-wins semantics: rules are tried in order, and the action of +// the first one that matches decides the outcome. If no rule matches, the +// entry is included by default — matching rsync's own default behavior of +// transferring anything not explicitly excluded. +// +// entry.Path is never empty and never represents the transfer root itself: +// Walk() (internal/sync/walk.go) deliberately excludes the root from its +// own output, so there's no "path exactly equal to the root" case for this +// function to special-case. +func Included(rules []Rule, entry FileEntry) bool { + for _, r := range rules { + if r.matches(entry.Path, entry.IsDir) { + return r.Action == Include + } + } + return true +} + +// FilterEntries returns the subset of entries that rules includes, in +// their original order. It is a post-pass over an already-collected +// Walk() result rather than a predicate threaded into Walk() itself — see +// the design note on this tradeoff where FilterEntries is introduced in +// the accompanying documentation/commit. +func FilterEntries(entries []FileEntry, rules []Rule) []FileEntry { + kept := make([]FileEntry, 0, len(entries)) + for _, e := range entries { + if Included(rules, e) { + kept = append(kept, e) + } + } + return kept +} diff --git a/internal/sync/filter_test.go b/internal/sync/filter_test.go new file mode 100644 index 0000000..9fd4ef2 --- /dev/null +++ b/internal/sync/filter_test.go @@ -0,0 +1,395 @@ +package sync + +import ( + "path/filepath" + "testing" +) + +func mustCompile(t *testing.T, raw []RawRule) []Rule { + t.Helper() + rules, err := CompileRules(raw) + if err != nil { + t.Fatalf("CompileRules returned error: %v", err) + } + return rules +} + +func TestRule_MatchesLiteralAndSingleSegmentWildcards(t *testing.T) { + tests := []struct { + pattern string + path string + want bool + }{ + {"file.txt", "file.txt", true}, + {"file.txt", "other.txt", false}, + {"*.txt", "file.txt", true}, + {"*.txt", "file.log", false}, + {"file.?xt", "file.txt", true}, + {"file.?xt", "file.text", false}, // ? matches exactly one char + } + + for _, tt := range tests { + r := Rule{Action: Include, Pattern: tt.pattern} + got := r.matches(tt.path, false) + if got != tt.want { + t.Errorf("Rule{Pattern: %q}.matches(%q) = %v, want %v", tt.pattern, tt.path, got, tt.want) + } + } +} + +func TestCompileRules_Anchoring(t *testing.T) { + rules := mustCompile(t, []RawRule{ + {Kind: RuleExclude, Pattern: "/build"}, // anchored: root-level "build" only + {Kind: RuleExclude, Pattern: "cache.tmp"}, // unanchored: matches at any depth + }) + + anchored, unanchored := rules[0], rules[1] + + if !anchored.Anchored { + t.Errorf("anchored.Anchored = false, want true") + } + if anchored.Pattern != "build" { + t.Errorf("anchored.Pattern = %q, want %q (leading / should be stripped)", anchored.Pattern, "build") + } + if !anchored.matches("build", false) { + t.Errorf("anchored pattern should match root-level %q", "build") + } + if anchored.matches("sub/build", false) { + t.Errorf("anchored pattern must NOT match nested %q", "sub/build") + } + + if unanchored.Anchored { + t.Errorf("unanchored.Anchored = true, want false") + } + if !unanchored.matches("cache.tmp", false) { + t.Errorf("unanchored pattern should match root-level %q", "cache.tmp") + } + if !unanchored.matches("a/b/cache.tmp", false) { + t.Errorf("unanchored pattern should match nested %q", "a/b/cache.tmp") + } +} + +func TestCompileRules_InternalSlashAnchorsWithoutLeadingSlash(t *testing.T) { + // Real rsync's actual rule: a pattern anchors to the root if it has a + // leading "/", OR contains any other "/", OR contains "**" — not just + // on an explicit leading "/". "src/main.go" (no leading slash, but an + // internal one) must behave the same as "/src/main.go" here. + rules := mustCompile(t, []RawRule{ + {Kind: RuleExclude, Pattern: "src/main.go"}, + }) + r := rules[0] + + if !r.Anchored { + t.Fatalf("Anchored = false, want true for a pattern with an internal \"/\" but no leading \"/\"") + } + if !r.matches("src/main.go", false) { + t.Errorf("should match root-level %q", "src/main.go") + } + if r.matches("vendor/src/main.go", false) { + t.Errorf("must NOT match nested %q now that internal \"/\" anchors", "vendor/src/main.go") + } +} + +func TestCompileRules_DoubleStarAnchorsWithoutSlash(t *testing.T) { + // Same rsync rule, the "**" half: a pattern containing "**" anchors + // even with no "/" anywhere in it. + rules := mustCompile(t, []RawRule{ + {Kind: RuleExclude, Pattern: "**cache**"}, + }) + if !rules[0].Anchored { + t.Errorf("Anchored = false, want true for a pattern containing \"**\" but no \"/\"") + } +} + +func TestCompileRules_BareFilenameStaysUnanchored(t *testing.T) { + // The one case that must NOT anchor: no "/" at all, no "**" at all. + rules := mustCompile(t, []RawRule{ + {Kind: RuleExclude, Pattern: "*.log"}, + }) + r := rules[0] + if r.Anchored { + t.Fatalf("Anchored = true, want false for a bare filename pattern with no \"/\" or \"**\"") + } + if !r.matches("a/b/debug.log", false) { + t.Errorf("bare filename pattern should still match at any depth") + } +} + +func TestCompileRules_DirOnly(t *testing.T) { + rules := mustCompile(t, []RawRule{ + {Kind: RuleExclude, Pattern: "build/"}, + }) + r := rules[0] + + if !r.DirOnly { + t.Fatalf("DirOnly = false, want true") + } + if r.Pattern != "build" { + t.Errorf("Pattern = %q, want %q (trailing / should be stripped)", r.Pattern, "build") + } + if !r.matches("build", true) { + t.Errorf("dir-only pattern should match a directory named %q", "build") + } + if r.matches("build", false) { + t.Errorf("dir-only pattern must NOT match a regular file named %q", "build") + } +} + +func TestCompileRules_AnchoredAndDirOnlyTogether(t *testing.T) { + rules := mustCompile(t, []RawRule{ + {Kind: RuleExclude, Pattern: "/dist/"}, + }) + r := rules[0] + + if !r.Anchored || !r.DirOnly || r.Pattern != "dist" { + t.Fatalf("got Anchored=%v DirOnly=%v Pattern=%q, want Anchored=true DirOnly=true Pattern=%q", + r.Anchored, r.DirOnly, r.Pattern, "dist") + } + if !r.matches("dist", true) { + t.Errorf("should match root-level directory %q", "dist") + } + if r.matches("sub/dist", true) { + t.Errorf("anchored: must NOT match nested directory %q", "sub/dist") + } +} + +func TestIncluded_NoRulesMatchDefaultsToIncluded(t *testing.T) { + entry := FileEntry{Path: "anything.txt"} + if !Included(nil, entry) { + t.Errorf("Included with no rules = false, want true (default include)") + } +} + +func TestIncluded_FirstMatchWinsOrderMatters(t *testing.T) { + entry := FileEntry{Path: "keep.log"} + + // Same two rules, opposite order: the first one to match should win in + // both cases, so swapping the order must flip the outcome. If it + // didn't, evaluation wouldn't actually be "first match wins" — it'd be + // "last match wins" or "most specific wins" or something else. + includeFirst := mustCompile(t, []RawRule{ + {Kind: RuleInclude, Pattern: "keep.log"}, + {Kind: RuleExclude, Pattern: "*.log"}, + }) + if !Included(includeFirst, entry) { + t.Errorf("include-before-exclude: Included = false, want true") + } + + excludeFirst := mustCompile(t, []RawRule{ + {Kind: RuleExclude, Pattern: "*.log"}, + {Kind: RuleInclude, Pattern: "keep.log"}, + }) + if Included(excludeFirst, entry) { + t.Errorf("exclude-before-include: Included = true, want false") + } +} + +func TestFilterEntries(t *testing.T) { + entries := []FileEntry{ + {Path: "main.go"}, + {Path: "debug.log"}, + {Path: "keep.log"}, + {Path: "build", IsDir: true}, + } + rules := mustCompile(t, []RawRule{ + {Kind: RuleInclude, Pattern: "keep.log"}, + {Kind: RuleExclude, Pattern: "*.log"}, + {Kind: RuleExclude, Pattern: "/build/"}, + }) + + got := FilterEntries(entries, rules) + + want := []string{"main.go", "keep.log"} + if len(got) != len(want) { + t.Fatalf("got %d entries, want %d: %v", len(got), len(want), got) + } + for i, w := range want { + if got[i].Path != w { + t.Errorf("entry %d: Path = %q, want %q", i, got[i].Path, w) + } + } +} + +func TestCompileRules_ExcludeFromPreservesPosition(t *testing.T) { + root := t.TempDir() + excludeFile := filepath.Join(root, "excludes.txt") + mustWriteFile(t, excludeFile, "# a comment, skipped\n\n*.log\n/dist/\n") + + rules := mustCompile(t, []RawRule{ + {Kind: RuleExclude, Pattern: "first.txt"}, + {Kind: RuleExcludeFrom, Pattern: excludeFile}, + {Kind: RuleInclude, Pattern: "last.txt"}, + }) + + // Position matters here: the two patterns read from the file must land + // between "first.txt" and "last.txt", not get appended after + // "last.txt" — that would silently reorder rules relative to what the + // user typed on the command line, breaking first-match-wins semantics. + want := []struct { + action Action + pattern string + dirOnly bool + }{ + {Exclude, "first.txt", false}, + {Exclude, "*.log", false}, + {Exclude, "dist", true}, + {Include, "last.txt", false}, + } + + if len(rules) != len(want) { + t.Fatalf("got %d rules, want %d: %+v", len(rules), len(want), rules) + } + for i, w := range want { + if rules[i].Action != w.action || rules[i].Pattern != w.pattern || rules[i].DirOnly != w.dirOnly { + t.Errorf("rule %d = %+v, want {Action:%v Pattern:%q DirOnly:%v}", i, rules[i], w.action, w.pattern, w.dirOnly) + } + } +} + +func TestCompileRules_IncludeFrom(t *testing.T) { + root := t.TempDir() + includeFile := filepath.Join(root, "includes.txt") + mustWriteFile(t, includeFile, "keep.log\n") + + rules := mustCompile(t, []RawRule{ + {Kind: RuleIncludeFrom, Pattern: includeFile}, + {Kind: RuleExclude, Pattern: "*.log"}, + }) + + entry := FileEntry{Path: "keep.log"} + if !Included(rules, entry) { + t.Errorf("keep.log should be included via --include-from before the *.log exclude") + } +} + +func TestCompileRules_ExcludeFromMissingFileErrors(t *testing.T) { + _, err := CompileRules([]RawRule{ + {Kind: RuleExcludeFrom, Pattern: filepath.Join(t.TempDir(), "does-not-exist.txt")}, + }) + if err == nil { + t.Fatalf("CompileRules with a missing --exclude-from file returned nil error, want an error") + } +} + +func TestCompileRules_FilterRuleSyntax(t *testing.T) { + rules := mustCompile(t, []RawRule{ + {Kind: RuleFilter, Pattern: "+ keep.log"}, + {Kind: RuleFilter, Pattern: "- *.log"}, + {Kind: RuleFilter, Pattern: "exclude /dist/"}, + }) + + if len(rules) != 3 { + t.Fatalf("got %d rules, want 3: %+v", len(rules), rules) + } + if rules[0].Action != Include || rules[0].Pattern != "keep.log" { + t.Errorf("rule 0 = %+v, want Include keep.log", rules[0]) + } + if rules[1].Action != Exclude || rules[1].Pattern != "*.log" { + t.Errorf("rule 1 = %+v, want Exclude *.log", rules[1]) + } + if rules[2].Action != Exclude || rules[2].Pattern != "dist" || !rules[2].Anchored || !rules[2].DirOnly { + t.Errorf("rule 2 = %+v, want anchored dir-only Exclude dist", rules[2]) + } +} + +func TestCompileRules_FilterUnsupportedSyntaxErrors(t *testing.T) { + _, err := CompileRules([]RawRule{ + {Kind: RuleFilter, Pattern: "P *.log"}, // rsync's terse "P" shorthand isn't supported here + }) + if err == nil { + t.Fatalf("CompileRules with unsupported filter syntax returned nil error, want an error") + } +} + +func TestCompileRules_FilterMergeBasic(t *testing.T) { + root := t.TempDir() + mergeFile := filepath.Join(root, "rules.txt") + mustWriteFile(t, mergeFile, "+ keep.log\n- *.log\n") + + rules := mustCompile(t, []RawRule{ + {Kind: RuleFilter, Pattern: "merge " + mergeFile}, + }) + + entry := FileEntry{Path: "keep.log"} + if !Included(rules, entry) { + t.Errorf("keep.log should survive filtering via the merged rules") + } + other := FileEntry{Path: "other.log"} + if Included(rules, other) { + t.Errorf("other.log should be excluded via the merged rules") + } +} + +func TestCompileRules_FilterNestedMergeErrors(t *testing.T) { + root := t.TempDir() + inner := filepath.Join(root, "inner.txt") + outer := filepath.Join(root, "outer.txt") + mustWriteFile(t, inner, "- *.log\n") + mustWriteFile(t, outer, "merge "+inner+"\n") + + _, err := CompileRules([]RawRule{ + {Kind: RuleFilter, Pattern: "merge " + outer}, + }) + if err == nil { + t.Fatalf("CompileRules with a nested merge file returned nil error, want an error") + } +} + +func TestCompileRules_EmptyPatternErrors(t *testing.T) { + tests := []string{ + "", // empty outright + "/", // empty after stripping the anchor marker + "a//b", // internal empty segment from a stray double slash + } + for _, pattern := range tests { + _, err := CompileRules([]RawRule{{Kind: RuleExclude, Pattern: pattern}}) + if err == nil { + t.Errorf("CompileRules with pattern %q returned nil error, want an error (a silent no-op rule is worse than a clear failure)", pattern) + } + } +} + +func TestRule_WildcardOnlyPatternsMatchEverything(t *testing.T) { + // This is intended behavior, matching real rsync: a bare "*" or "**" + // is a deliberate "match everything" rule, not a bug to guard against. + entries := []FileEntry{{Path: "a"}, {Path: "a/b"}, {Path: "a/b/c.txt"}} + + star := mustCompile(t, []RawRule{{Kind: RuleExclude, Pattern: "*"}}) + for _, e := range entries { + if Included(star, e) { + t.Errorf("bare \"*\" should exclude every entry, but %q survived", e.Path) + } + } + + doubleStar := mustCompile(t, []RawRule{{Kind: RuleExclude, Pattern: "**"}}) + for _, e := range entries { + if Included(doubleStar, e) { + t.Errorf("bare \"**\" should exclude every entry, but %q survived", e.Path) + } + } +} + +func TestRule_MatchesDoubleStarCrossesDirectories(t *testing.T) { + tests := []struct { + pattern string + path string + want bool + }{ + {"a/**/z", "a/z", true}, // ** matches zero segments + {"a/**/z", "a/b/z", true}, // ** matches one segment + {"a/**/z", "a/b/c/d/z", true}, // ** matches many segments + {"a/**/z", "x/b/z", false}, // literal "a" segment still required + {"a/*/z", "a/b/c/z", false}, // single "*" must NOT cross "/" + {"**/vendor", "vendor", true}, // leading ** matches zero segments too + {"**/vendor", "a/b/vendor", true}, + {"a/**", "a/b/c", true}, // trailing ** matches everything under a/ + } + + for _, tt := range tests { + r := Rule{Action: Include, Pattern: tt.pattern} + got := r.matches(tt.path, false) + if got != tt.want { + t.Errorf("Rule{Pattern: %q}.matches(%q) = %v, want %v", tt.pattern, tt.path, got, tt.want) + } + } +}