index: add partial runs which only re-index changed directories

After a sync which changed a known set of files there is no need to
walk the whole remote. --changed PATH, --changed-from FILE and
--changed-combined FILE tell rclone index what changed, and it
re-indexes only the directories containing those paths and their
ancestors, each with one non-recursive listing.
This commit is contained in:
Nick Craig-Wood committed 2026-09-29 18:21:23 +01:00
1 parent 72a6d36cb2
commit 666d67f0eb
5 files changed
+470 -22

No files matched your search

+4
View File
@@ -33,6 +33,10 @@ func init() {
flags.FVarP(cmdFlags, &opt.DirTime, "dir-time", "", "How to work out the time shown for a directory", "")
flags.BoolVarP(cmdFlags, &opt.NoModTime, "no-modtime", "", opt.NoModTime, "Don't show modification times in listings", "")
flags.BoolVarP(cmdFlags, &opt.Rewrite, "index-rewrite", "", opt.Rewrite, "Write every listing even if it is unchanged", "")
flags.StringArrayVarP(cmdFlags, &opt.Changed, "changed", "", nil, "Only re-index directories affected by this changed path (directories end in /)", "")
flags.StringArrayVarP(cmdFlags, &opt.ChangedFrom, "changed-from", "", nil, "Read changed paths from file, one per line (use - to read from stdin)", "")
flags.StringArrayVarP(cmdFlags, &opt.ChangedCombined, "changed-combined", "", nil, "Read changed paths from a sync --combined report (use - to read from stdin)", "")
flags.IntVarP(cmdFlags, &opt.ChangedMaxDirs, "changed-max-dirs", "", opt.ChangedMaxDirs, "Do a full index if a partial one would list more than this many directories (0 for no limit)", "")
flags.StringVarP(cmdFlags, &printTemplate, "print-template", "", "", "Print the built-in template for FORMAT and exit", "")
}
+43
View File
@@ -85,6 +85,49 @@ nginx and Apache, and R2 custom domains with a URL rewrite rule. For
hosts that don't (plain S3 REST URLs, B2 friendly URLs) use
`--link-index` to make directory links `dir/index.html` instead.
### Partial runs
After a sync which changed a known set of files there is no need to
walk the whole remote. Tell `rclone index` what changed and it re-indexes
only the directories that could be affected:
rclone index r2:bucket --changed v1.76.0/ --changed version.txt
rclone index r2:bucket --changed-from changes.txt
rclone sync ./public r2:bucket --exclude index.html --combined - | rclone index r2:bucket --changed-combined -
- `--changed PATH` names a changed file or directory relative to
`remote:path` and can be repeated. A directory, given with a trailing
slash or found to be one, is re-indexed completely.
- `--changed-from FILE` reads one path per line exactly as written.
- `--changed-combined FILE` reads the `--combined` report of
`rclone sync`: lines starting with `+`, `-`, `*` and `!` are changes and
`=` lines are ignored.
- Both files accept `-` for standard input and all three flags can be
combined.
A changed file re-indexes its directory and every directory above it,
since their times and their listings can change too, and on bucket
based storage a directory can cease to exist. Each of those is one
directory listing. With `--dir-time newest` the directories above also
need the times of their unchanged subdirectories, which are read from
the modification times of those subdirectories' listings at the cost
of one small request each. With `--use-server-modtime` that is the time
the listing was written rather than the time of the newest file, so
partial runs can drift by the indexing delay until the next full run.
`--dir-time dir` and `none` need no such lookups.
Directories which are neither named nor above a named path are left
alone, as are directories deeper than `--max-depth`. An empty change
list does nothing, and naming the root does a full run. There is no
automatic cutover to a full run since that depends on the size of the
whole remote: a partial run costs one listing per affected directory,
a full run one request per thousand objects on S3-like storage or one
per directory elsewhere. The number
of directories a partial run will list is logged with `-v`, and
`--changed-max-dirs` makes it fall back to a full run above that
number. A good pattern is a partial run after each upload and a
scheduled full run to catch anything missed.
### Site icon
The listings carry no icon or branding of their own. To give a site an
+264 -22
View File
@@ -1,6 +1,7 @@
package operations
import (
"bufio"
"bytes"
"context"
"encoding/json"
@@ -12,6 +13,7 @@ import (
"path"
"sort"
"strings"
"sync"
texttemplate "text/template"
"time"
@@ -56,6 +58,13 @@ type IndexOpt struct {
DirTime IndexDirTime `json:"dirTime"` // how to work out directory times
NoModTime bool `json:"noModTime"` // don't show modification times
Rewrite bool `json:"rewrite"` // write every listing even if it is unchanged
// Partial runs only re-index the directories affected by these
// changed paths and their ancestors. Empty means a full run.
Changed []string `json:"changed"` // changed files or directories, directories with a trailing /
ChangedFrom []string `json:"changedFrom"` // files of changed paths, one per line
ChangedCombined []string `json:"changedCombined"` // files in the sync --combined report format
ChangedMaxDirs int `json:"changedMaxDirs"` // do a full run if a partial run would list more directories than this, 0 for no limit
}
// IndexOptDefault are the default options for Index
@@ -119,6 +128,8 @@ type index struct {
dirTime IndexDirTime
now time.Time
dirs map[string]*indexDir
unwalkedMu sync.Mutex
unwalked map[string]time.Time // newest times of directories which weren't walked
transfers errgroup.Group
ec *errcount.ErrCount
}
@@ -148,6 +159,7 @@ func newIndex(ctx context.Context, f fs.Fs, opt *IndexOpt) (*index, error) {
dirTime: opt.DirTime,
now: time.Now(),
dirs: map[string]*indexDir{},
unwalked: map[string]time.Time{},
ec: errcount.New(),
}
if ix.opt.NoModTime {
@@ -241,11 +253,18 @@ func newIndexTextTemplate(text string) (*texttemplate.Template, error) {
// run does the work of Index
func (ix *index) run(ctx context.Context) error {
ci := fs.GetConfig(ctx)
err := walk.Walk(ctx, ix.f, "", true, ConfigMaxDepth(ctx, true), ix.walkFn(ctx))
changed, err := ix.changedPaths()
if err != nil {
return err
}
if changed == nil {
err = ix.walkAll(ctx)
} else {
err = ix.walkChanged(ctx, changed)
}
if err != nil {
return err
}
ix.prune()
// Directories deepest first so that a directory's subdirectories
// are done before it.
@@ -256,6 +275,7 @@ func (ix *index) run(ctx context.Context) error {
sort.Slice(dirs, func(i, j int) bool {
return dirs[i].depth > dirs[j].depth
})
ix.prune(dirs)
// Read the modification times --checkers at a time as they may
// need a transaction each on some backends, and count them as
@@ -272,6 +292,17 @@ func (ix *index) run(ctx context.Context) error {
return nil
})
}
if ix.dirTime != IndexDirTimeNewest {
continue
}
for _, sub := range d.subdirs {
if ix.dirs[sub.Remote()] == nil {
g.Go(func() error {
ix.readUnwalkedTime(ctx, sub)
return nil
})
}
}
}
_ = g.Wait()
for _, d := range dirs {
@@ -293,9 +324,222 @@ func (ix *index) run(ctx context.Context) error {
return ix.ec.Err("index")
}
// walkAll collects every directory for a full run
func (ix *index) walkAll(ctx context.Context) error {
return walk.Walk(ctx, ix.f, "", true, ConfigMaxDepth(ctx, true), ix.walkFn(ctx))
}
// changedPaths returns the changed paths from the options, or nil
// for a full run. Directories have a trailing slash.
func (ix *index) changedPaths() (changed []string, err error) {
opt := ix.opt
if len(opt.Changed) == 0 && len(opt.ChangedFrom) == 0 && len(opt.ChangedCombined) == 0 {
return nil, nil
}
changed = []string{}
add := func(p string) {
isDir := strings.HasSuffix(p, "/")
p = strings.Trim(path.Clean("/"+p), "/")
if isDir {
p += "/"
}
changed = append(changed, p)
}
for _, p := range opt.Changed {
add(p)
}
for _, name := range opt.ChangedFrom {
err = forEachIndexLine(name, func(line string) error {
if line != "" {
add(line)
}
return nil
})
if err != nil {
return nil, err
}
}
for _, name := range opt.ChangedCombined {
err = forEachIndexLine(name, func(line string) error {
if line == "" {
return nil
}
if len(line) < 2 || line[1] != ' ' || !strings.ContainsRune("+-*!=", rune(line[0])) {
return fmt.Errorf("malformed line %q in combined report %q", line, name)
}
if line[0] != '=' {
add(line[2:])
}
return nil
})
if err != nil {
return nil, err
}
}
return changed, nil
}
// forEachIndexLine calls fn with every line of the file name, or of
// stdin if name is "-", exactly as read
func forEachIndexLine(name string, fn func(string) error) (err error) {
in := os.Stdin
if name != "-" {
in, err = os.Open(name)
if err != nil {
return err
}
defer fs.CheckClose(in, &err)
}
scanner := bufio.NewScanner(in)
for scanner.Scan() {
if err := fn(scanner.Text()); err != nil {
return err
}
}
return scanner.Err()
}
// walkChanged collects the directories a partial run needs: the
// parent of every changed path and all of its ancestors, listed one
// level deep, and every changed directory walked in full.
//
// Directories deeper than --max-depth are left alone as a full run
// wouldn't walk them either.
func (ix *index) walkChanged(ctx context.Context, changed []string) error {
if len(changed) == 0 {
fs.Infof(ix.f, "Nothing changed so nothing to index")
return nil
}
maxDepth := ConfigMaxDepth(ctx, true)
inDepth := func(dir string) bool {
return maxDepth < 0 || indexDepth(dir) < maxDepth
}
var subtrees []string
for _, p := range changed {
if p == "/" {
fs.Infof(ix.f, "Root changed so doing a full index")
return ix.walkAll(ctx)
}
if strings.HasSuffix(p, "/") {
subtrees = append(subtrees, strings.TrimSuffix(p, "/"))
}
}
subtrees = indexOuterDirs(subtrees)
// A changed directory's walk covers everything below it
inSubtree := func(dir string) bool {
for _, sub := range subtrees {
if dir == sub || strings.HasPrefix(dir, sub+"/") {
return true
}
}
return false
}
list := map[string]bool{}
for _, p := range changed {
for dir := indexParent(strings.TrimSuffix(p, "/")); ; dir = indexParent(dir) {
if inDepth(dir) && !inSubtree(dir) {
list[dir] = true
}
if dir == "" {
break
}
}
}
if ix.opt.ChangedMaxDirs > 0 && len(list)+len(subtrees) > ix.opt.ChangedMaxDirs {
fs.Infof(ix.f, "Partial index would list %d directories, more than %d, so doing a full index", len(list)+len(subtrees), ix.opt.ChangedMaxDirs)
return ix.walkAll(ctx)
}
fs.Infof(ix.f, "Partial index: listing %d directories and walking %d changed directories", len(list), len(subtrees))
walkFn := ix.walkFn(ctx)
for dir := range list {
err := walk.Walk(ctx, ix.f, dir, true, 1, walkFn)
if err != nil {
return err
}
}
// Changed paths which turn out to be directories are walked too
for _, p := range changed {
if d := ix.dirs[indexParent(p)]; d != nil && !strings.HasSuffix(p, "/") {
for _, sub := range d.subdirs {
if sub.Remote() == p {
subtrees = append(subtrees, p)
}
}
}
}
for _, dir := range indexOuterDirs(subtrees) {
if !inDepth(dir) {
continue
}
depth := maxDepth
if maxDepth >= 0 {
depth = maxDepth - indexDepth(dir)
}
err := walk.Walk(ctx, ix.f, dir, true, depth, walkFn)
if err != nil {
return err
}
}
return nil
}
// indexOuterDirs sorts dirs and drops any which is the same as, or
// inside, another so that walking the result covers each once
func indexOuterDirs(dirs []string) (outer []string) {
sort.Strings(dirs)
for _, dir := range dirs {
if n := len(outer); n > 0 && (dir == outer[n-1] || strings.HasPrefix(dir, outer[n-1]+"/")) {
continue
}
outer = append(outer, dir)
}
return outer
}
// indexDepth returns how many directories deep remote is, 0 for the root
func indexDepth(remote string) int {
if remote == "" {
return 0
}
return strings.Count(remote, "/") + 1
}
// indexParent returns the parent directory of p, "" for the root
func indexParent(p string) string {
dir := path.Dir(p)
if dir == "." {
return ""
}
return dir
}
// readUnwalkedTime finds the newest time of a directory which wasn't
// walked from the modification time of its listing, which is set to
// the directory's newest time when it is written
func (ix *index) readUnwalkedTime(ctx context.Context, sub fs.Directory) {
o, err := ix.f.NewObject(ctx, path.Join(sub.Remote(), ix.outputs[0].name))
if err != nil {
if !errors.Is(err, fs.ErrorObjectNotFound) {
fs.Debugf(sub, "Failed to read listing modtime: %v", err)
}
return
}
tr := accounting.Stats(ctx).NewCheckingTransfer(o, "reading modtime")
t := o.ModTime(ctx)
tr.Done(ctx, nil)
ix.unwalkedMu.Lock()
ix.unwalked[sub.Remote()] = t
ix.unwalkedMu.Unlock()
}
// walkFn returns the function which collects the directories
func (ix *index) walkFn(ctx context.Context) walk.Func {
return func(remote string, entries fs.DirEntries, err error) error {
if errors.Is(err, fs.ErrorDirNotFound) {
// A changed directory which no longer exists
ix.dirs[remote] = &indexDir{remote: remote, depth: indexDepth(remote), gone: true}
return nil
}
if err != nil {
return err
}
@@ -318,12 +562,10 @@ func (ix *index) walkFn(ctx context.Context) walk.Func {
}
d := &indexDir{
remote: remote,
depth: indexDepth(remote),
outputs: map[string]fs.Object{},
excluded: !included,
}
if remote != "" {
d.depth = strings.Count(remote, "/") + 1
}
for _, entry := range entries {
switch x := entry.(type) {
case fs.Object:
@@ -358,29 +600,22 @@ func (ix *index) walkFn(ctx context.Context) walk.Func {
// On backends which can't have empty directories these will cease to
// exist once their outputs are deleted, so they are removed from their
// parent's listing, which may in turn leave the parent with nothing
// but outputs.
func (ix *index) prune() {
// but outputs. dirs must be deepest first.
func (ix *index) prune(dirs []*indexDir) {
if ix.f.Features().CanHaveEmptyDirectories {
return
}
var pruneDir func(d *indexDir)
pruneDir = func(d *indexDir) {
for _, d := range dirs {
kept := d.subdirs[:0]
for _, sub := range d.subdirs {
if sd := ix.dirs[sub.Remote()]; sd != nil {
pruneDir(sd)
if sd.gone {
continue
}
if sd := ix.dirs[sub.Remote()]; sd != nil && sd.gone {
continue
}
kept = append(kept, sub)
}
d.subdirs = kept
d.gone = !d.hasContent && len(d.subdirs) == 0
}
if root := ix.dirs[""]; root != nil {
pruneDir(root)
}
}
// setNewest sets d.newest from its files and subdirectories, which
@@ -392,12 +627,21 @@ func (ix *index) setNewest(ctx context.Context, d *indexDir) {
}
}
for _, sub := range d.subdirs {
if sd := ix.dirs[sub.Remote()]; sd != nil && sd.newest.After(d.newest) {
d.newest = sd.newest
if t := ix.subdirNewest(sub); t.After(d.newest) {
d.newest = t
}
}
}
// subdirNewest returns the newest time of a subdirectory, which was
// either walked or had its listing's modification time read
func (ix *index) subdirNewest(sub fs.Directory) time.Time {
if sd := ix.dirs[sub.Remote()]; sd != nil {
return sd.newest
}
return ix.unwalked[sub.Remote()]
}
// processDir brings the outputs of d up to date
func (ix *index) processDir(ctx context.Context, d *indexDir) {
if d.gone {
@@ -468,9 +712,7 @@ func (ix *index) render(ctx context.Context, d *indexDir) *serve.Directory {
case IndexDirTimeDir:
t = sub.ModTime(ctx)
case IndexDirTimeNewest:
if sd := ix.dirs[sub.Remote()]; sd != nil {
t = sd.newest
}
t = ix.subdirNewest(sub)
}
// Not sub.Size() as some backends, eg local on macOS, report a
// directory size which changes when the listings are written
+155
View File
@@ -3,6 +3,7 @@ package operations_test
import (
"context"
"encoding/json"
"fmt"
"mime"
"os"
"path/filepath"
@@ -387,3 +388,157 @@ func TestIndexNoHash(t *testing.T) {
assert.Equal(t, int64(1), transfers)
assert.Contains(t, r.read(t, "sub/index.html"), "file4.txt")
}
func TestIndexChanged(t *testing.T) {
r := newIndexRun(t)
r.WriteObject(r.ctx, "other/notes.md", "notes", t1)
r.index(t)
// Add files in sub/deep and in other but only declare the one in sub/deep
r.WriteObject(r.ctx, "sub/deep/file4.txt", "hello4", t1)
r.WriteObject(r.ctx, "other/undeclared.txt", "hello", t1)
r.opt.Changed = []string{"sub/deep/file4.txt"}
transfers, _ := r.index(t)
assert.Equal(t, int64(1), transfers)
assert.Contains(t, r.read(t, "sub/deep/index.html"), "file4.txt")
assert.NotContains(t, r.read(t, "other/index.html"), "undeclared.txt")
// A new directory changes its parent's listing too
r.WriteObject(r.ctx, "sub/new/file5.txt", "hello5", t1)
r.opt.Changed = []string{"sub/new/file5.txt"}
transfers, _ = r.index(t)
assert.Equal(t, int64(2), transfers)
assert.Contains(t, r.read(t, "sub/index.html"), `<a href="new/">new/</a>`)
assert.Contains(t, r.read(t, "sub/new/index.html"), "file5.txt")
// A changed directory is walked whether or not it has a trailing slash
for i, changed := range []string{"other/", "other"} {
name := fmt.Sprintf("other/undeclared%d.txt", i)
r.WriteObject(r.ctx, name, "hello", t1)
r.opt.Changed = []string{changed}
transfers, _ = r.index(t)
assert.Equal(t, int64(1), transfers, changed)
assert.Contains(t, r.read(t, "other/index.html"), name[6:])
}
// An empty change list does nothing
r.WriteObject(r.ctx, "sub/file6.txt", "hello6", t1)
r.opt.Changed = nil
r.opt.ChangedFrom = []string{filepath.Join(t.TempDir(), "empty.txt")}
require.NoError(t, os.WriteFile(r.opt.ChangedFrom[0], nil, 0600))
transfers, _ = r.index(t)
assert.Equal(t, int64(0), transfers)
assert.NotContains(t, r.read(t, "sub/index.html"), "file6.txt")
// Naming the root does a full run
r.opt.ChangedFrom = nil
r.opt.Changed = []string{"/"}
transfers, _ = r.index(t)
assert.Equal(t, int64(1), transfers)
assert.Contains(t, r.read(t, "sub/index.html"), "file6.txt")
// So does exceeding --changed-max-dirs
r.WriteObject(r.ctx, "sub/file7.txt", "hello7", t1)
r.opt.Changed = []string{"other/notes.md"}
r.opt.ChangedMaxDirs = 1
transfers, _ = r.index(t)
assert.Equal(t, int64(1), transfers)
assert.Contains(t, r.read(t, "sub/index.html"), "file7.txt")
r.opt.ChangedMaxDirs = 0
// Paths are cleaned, and nested changed directories are walked once
r.WriteObject(r.ctx, "sub/deep/file8.txt", "hello8", t1)
r.opt.Changed = []string{"./sub/", "sub//deep/", "sub/deep/"}
transfers, _ = r.index(t)
assert.Equal(t, int64(1), transfers)
assert.Contains(t, r.read(t, "sub/deep/index.html"), "file8.txt")
// A partial run doesn't write listings a full run wouldn't
ctx, ci := fs.AddConfig(r.ctx)
ci.MaxDepth = 2
r.WriteObject(r.ctx, "sub/deep/file9.txt", "hello9", t1)
r.WriteObject(r.ctx, "sub/file10.txt", "hello10", t1)
r.opt.Changed = []string{"sub/deep/file9.txt", "sub/file10.txt"}
require.NoError(t, operations.Index(ctx, r.Fremote, &r.opt))
assert.NotContains(t, r.read(t, "sub/deep/index.html"), "file9.txt")
assert.Contains(t, r.read(t, "sub/index.html"), "file10.txt")
}
func TestIndexChangedDelete(t *testing.T) {
r := newIndexRun(t)
r.index(t)
// Delete the only file in sub/deep and declare it
o, err := r.Fremote.NewObject(r.ctx, "sub/deep/file3.txt")
require.NoError(t, err)
require.NoError(t, o.Remove(r.ctx))
r.opt.Changed = []string{"sub/deep/file3.txt"}
transfers, deletes := r.index(t)
if r.Fremote.Features().CanHaveEmptyDirectories {
assert.Equal(t, int64(1), transfers)
assert.Equal(t, int64(0), deletes)
assert.NotContains(t, r.read(t, "sub/deep/index.html"), "file3.txt")
} else {
assert.Equal(t, int64(1), transfers)
assert.Equal(t, int64(1), deletes)
assert.NotContains(t, r.read(t, "sub/index.html"), "deep/")
r.checkFiles(t, "file1.txt", "index.html", "sub/file2.txt", "sub/index.html")
// Declaring a directory which has gone is harmless
r.opt.Changed = []string{"sub/deep/"}
transfers, deletes = r.index(t)
assert.Equal(t, int64(0), transfers)
assert.Equal(t, int64(0), deletes)
}
}
func TestIndexChangedFiles(t *testing.T) {
r := newIndexRun(t)
r.WriteObject(r.ctx, "- odd/file.txt", "odd", t1)
r.index(t)
dir := t.TempDir()
// --changed-from takes paths exactly as written
r.WriteObject(r.ctx, "- odd/file8.txt", "hello8", t1)
r.WriteObject(r.ctx, "sub/file9.txt", "hello9", t1)
from := filepath.Join(dir, "from.txt")
require.NoError(t, os.WriteFile(from, []byte("- odd/file8.txt\n\n"), 0600))
r.opt.ChangedFrom = []string{from}
transfers, _ := r.index(t)
assert.Equal(t, int64(1), transfers)
assert.Contains(t, r.read(t, "- odd/index.html"), "file8.txt")
assert.NotContains(t, r.read(t, "sub/index.html"), "file9.txt")
// --changed-combined strips the prefixes and skips unchanged files
combined := filepath.Join(dir, "combined.txt")
require.NoError(t, os.WriteFile(combined, []byte("= file1.txt\n+ sub/file9.txt\n- sub/deep/gone.txt\n* - odd/file.txt\n! sub/deep/file3.txt\n"), 0600))
r.opt.ChangedFrom = nil
r.opt.ChangedCombined = []string{combined}
transfers, _ = r.index(t)
assert.Equal(t, int64(1), transfers)
assert.Contains(t, r.read(t, "sub/index.html"), "file9.txt")
// Malformed combined lines are an error
require.NoError(t, os.WriteFile(combined, []byte("sub/file9.txt\n"), 0600))
assert.Error(t, operations.Index(r.ctx, r.Fremote, &r.opt))
}
func TestIndexChangedDirTime(t *testing.T) {
r := newIndexRun(t)
r.opt.NoModTime = false
r.opt.DirTime = operations.IndexDirTimeNewest
r.opt.Outputs = []string{"index.json=json"}
r.WriteObject(r.ctx, "other/notes.md", "notes", t2)
r.index(t)
precision := r.Fremote.Precision()
// A partial run gets the time of the unchanged directory other
// from its listing, and the changed directory sub from the walk
r.WriteObject(r.ctx, "sub/deep/file4.txt", "hello4", fstest.Time("2030-01-01T00:00:00Z"))
r.opt.Changed = []string{"sub/deep/file4.txt"}
transfers, _ := r.index(t)
assert.Equal(t, int64(3), transfers)
modTimes := r.indexModTimes(t, "index.json")
fstest.AssertTimeEqualWithPrecision(t, "sub", fstest.Time("2030-01-01T00:00:00Z"), modTimes["sub"], precision)
fstest.AssertTimeEqualWithPrecision(t, "other", t2, modTimes["other"], precision)
}
+4
View File
@@ -712,6 +712,10 @@ func init() {
- dirTime - how to work out directory times: "newest", "dir" or "none"
- noModTime - don't show modification times
- rewrite - write every listing even if it is unchanged
- changed - list of changed paths for a partial run, directories ending in "/"
- changedFrom - list of files of changed paths, one per line
- changedCombined - list of files in the sync --combined report format
- changedMaxDirs - do a full run if a partial one would list more directories than this
See the [index](/commands/rclone_index/) command for more information on the above.
`,