From da41970456110fab0afd66d432db0cc531867774 Mon Sep 17 00:00:00 2001 From: Duc-Tam Nguyen Date: Mon, 10 Aug 2026 13:38:02 +0700 Subject: [PATCH] Use path-component semantics for crawl controls Co-authored-by: SihanTeng --- CHANGELOG.md | 8 +++++++ README.md | 6 ++--- cli/clone.go | 6 ++--- clone/config.go | 2 +- docs/content/guides/scoping-a-crawl.md | 15 ++++++++----- docs/content/reference/cli.md | 6 ++--- docs/content/reference/release-notes.md | 7 ++++++ urlx/urlx.go | 29 ++++++++++++++++++++++--- urlx/urlx_test.go | 11 +++++++++- 9 files changed, 71 insertions(+), 19 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 8bd9768..8529ea2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -13,6 +13,14 @@ All notable changes to kage are recorded here. The format follows - A `robots.txt` reference explains the `kage` agent token, `Crawl-delay`, and the advisory `--no-robots` override ([#8](https://github.com/tamnd/kage/issues/8)). +### Changed + +- `--exclude` and `--scope-prefix` now match complete path prefixes and their + descendants rather than arbitrary substrings. If you relied on the old + `--exclude` behaviour, pass the full path prefix. +- `--max-pages` is documented as attempting at most N page renders; failed + renders count toward the cap. + ### Fixed - `--resume` picks an interrupted crawl back up instead of doing nothing ([#36](https://github.com/tamnd/kage/issues/36)). diff --git a/README.md b/README.md index 76195d2..b998ccf 100644 --- a/README.md +++ b/README.md @@ -114,11 +114,11 @@ The flags you'll actually reach for: | Flag | Default | Meaning | |------|---------|---------| | `-o, --out` | `$HOME/data/kage` | Output root; the mirror lands in `//` | -| `-p, --max-pages` | `0` | Stop after N pages (0 = no limit) | +| `-p, --max-pages` | `0` | Attempt at most N page renders (0 = no limit); failed renders count toward the cap | | `-d, --max-depth` | `0` | How many links deep to follow (0 = no limit) | -| `--scope-prefix` | | Only crawl paths starting with this prefix | +| `--scope-prefix` | | Only crawl this path and its descendants, not similar path names | | `--subdomains` | `false` | Treat subdomains of the seed host as in scope | -| `--exclude` | | Path prefixes to skip (repeatable) | +| `--exclude` | | Paths and their descendants to skip (repeatable), not matching substrings elsewhere | | `--scroll` | `false` | Auto-scroll each page to trigger lazy loading | | `--workers` | `4` | How many pages to render at once | | `--no-robots` | `false` | Ignore `robots.txt` (be nice) | diff --git a/cli/clone.go b/cli/clone.go index a3b402a..8d45eda 100644 --- a/cli/clone.go +++ b/cli/clone.go @@ -69,7 +69,7 @@ func newCloneCmd() *cobra.Command { fs.IntVar(&f.workers, "workers", 4, "concurrent page render workers") fs.IntVar(&f.assetWorkers, "asset-workers", 8, "concurrent asset download workers") fs.IntVar(&f.browserPages, "browser-pages", 4, "Chrome page-pool size") - fs.IntVarP(&f.maxPages, "max-pages", "p", 0, "stop after N pages (0 = unlimited)") + fs.IntVarP(&f.maxPages, "max-pages", "p", 0, "attempt at most N page renders (0 = unlimited)") fs.IntVarP(&f.maxDepth, "max-depth", "d", 0, "link-follow depth cap (0 = unlimited)") fs.StringVar(&f.traversal, "traversal", "bfs", "frontier order: bfs or dfs") fs.Int64Var(&f.maxAssetMB, "max-asset-mb", 25, "skip assets larger than N MB (left on the live web)") @@ -82,8 +82,8 @@ func newCloneCmd() *cobra.Command { fs.BoolVar(&f.scroll, "scroll", false, "auto-scroll each page to trigger lazy loading") fs.StringVar(&f.userAgent, "user-agent", clone.DefaultUserAgent, "User-Agent for asset and robots fetches") fs.BoolVar(&f.subdomains, "subdomains", false, "treat subdomains of the seed host as in scope") - fs.StringVar(&f.scopePrefix, "scope-prefix", "", "only crawl pages whose path starts with this prefix") - fs.StringSliceVar(&f.exclude, "exclude", nil, "path prefixes to skip (repeatable)") + fs.StringVar(&f.scopePrefix, "scope-prefix", "", "only crawl this path and its descendants, e.g. /doc does not match /documentation") + fs.StringSliceVar(&f.exclude, "exclude", nil, "path prefixes to skip, e.g. /archive (repeatable; matches the path and its descendants)") fs.BoolVar(&f.noRobots, "no-robots", false, "ignore robots.txt (be careful and polite)") fs.DurationVar(&f.crawlDelay, "crawl-delay", 0, "override robots.txt Crawl-delay between page starts (0 = use robots.txt)") fs.BoolVar(&f.noSitemap, "no-sitemap", false, "do not seed URLs from sitemap.xml") diff --git a/clone/config.go b/clone/config.go index 97d7eed..3ba256e 100644 --- a/clone/config.go +++ b/clone/config.go @@ -31,7 +31,7 @@ type Config struct { Workers int // page render workers AssetWorkers int // HTTP asset download workers BrowserPages int // Chrome page-pool size - MaxPages int // stop after N pages (0 = unlimited) + MaxPages int // attempt at most N page renders (0 = unlimited) MaxDepth int // BFS/DFS depth cap (0 = unlimited) Traversal string MaxAssetBytes int64 diff --git a/docs/content/guides/scoping-a-crawl.md b/docs/content/guides/scoping-a-crawl.md index 46122a9..270c829 100644 --- a/docs/content/guides/scoping-a-crawl.md +++ b/docs/content/guides/scoping-a-crawl.md @@ -11,7 +11,7 @@ the crawl. ## Limit by count and depth ```bash -# Stop after 200 pages +# Attempt at most 200 page renders kage clone example.com --max-pages 200 # Only follow links three hops from the seed @@ -19,7 +19,8 @@ kage clone example.com --max-depth 3 ``` `--max-depth 0` (the default) means unlimited depth; `--max-pages 0` means -unlimited pages. Combine them to put a hard ceiling on a run. +unlimited attempts. Failed renders count toward the page cap. Combine the flags +to put a hard ceiling on a run. ## Limit by path @@ -29,10 +30,14 @@ To clone just one section of a site, restrict the crawl to a path prefix: kage clone example.com --scope-prefix /docs ``` -Only pages whose path starts with `/docs` are followed. Assets are still fetched -from wherever the page references them, so the section renders correctly. +Only `/docs` and pages below it are followed; a path such as `/documentation` +does not match. Assets are still fetched from wherever the page references +them, so the section renders correctly. -To skip parts of a site, exclude path prefixes (repeatable): +To skip parts of a site, exclude path prefixes (repeatable). An exclude matches +that path and everything under it (`/archive` skips `/archive` and +`/archive/2020`), but not a path containing the same text elsewhere +(`/map/archive-index` is still crawled): ```bash kage clone example.com --exclude /archive --exclude /tags diff --git a/docs/content/reference/cli.md b/docs/content/reference/cli.md index c3b6f4c..9b950e9 100644 --- a/docs/content/reference/cli.md +++ b/docs/content/reference/cli.md @@ -35,11 +35,11 @@ images, and fonts, and writes a browsable mirror to `//`. | Flag | Default | Meaning | |------|---------|---------| -| `-p, --max-pages` | `0` | Stop after N pages (0 = unlimited) | +| `-p, --max-pages` | `0` | Attempt at most N page renders (0 = unlimited); failures count toward the cap | | `-d, --max-depth` | `0` | Link-follow depth cap (0 = unlimited) | -| `--scope-prefix` | | Only crawl pages whose path starts with this prefix | +| `--scope-prefix` | | Only crawl the path prefix and its descendants, not similar path names | | `--subdomains` | `false` | Treat subdomains of the seed host as in scope | -| `--exclude` | | Path prefixes to skip (repeatable) | +| `--exclude` | | Path prefixes to skip (repeatable); matches the path and its descendants, not substrings elsewhere | | `--traversal` | `bfs` | Frontier order: `bfs` or `dfs` | ### Politeness diff --git a/docs/content/reference/release-notes.md b/docs/content/reference/release-notes.md index 8177947..0b5be33 100644 --- a/docs/content/reference/release-notes.md +++ b/docs/content/reference/release-notes.md @@ -6,6 +6,13 @@ weight: 40 The authoritative, commit-level history lives in [`CHANGELOG.md`](https://github.com/tamnd/kage/blob/main/CHANGELOG.md) and on the [releases page](https://github.com/tamnd/kage/releases). This page summarises each version. +## Unreleased + +- **Crawl path controls use path boundaries.** `--exclude` and + `--scope-prefix` match a path and its descendants without catching unrelated + names that merely contain the same text. `--max-pages` now accurately says + that failed render attempts count toward its cap. + ## v0.3.11 - **`go install ...@latest` works again.** The v0.3.9 antivirus fix replaced Rod's leakless dependency with a local stub. That kept the flagged helper out of `kage.exe`, but Go refuses versioned installation of a module containing a dependency-changing `replace` directive ([#72](https://github.com/tamnd/kage/issues/72)). Windows now launches Chrome through a small platform-specific launcher that never imports leakless. Other platforms keep Rod's launcher, the Windows binary remains free of the flagged helper, and the module no longer needs `replace`. diff --git a/urlx/urlx.go b/urlx/urlx.go index de618c4..beecf97 100644 --- a/urlx/urlx.go +++ b/urlx/urlx.go @@ -204,7 +204,7 @@ func Key(u *url.URL) string { return u.String() } type ScopeConfig struct { IncludeSubdomains bool ScopePrefix string // only crawl paths under this prefix, e.g. "/docs/" - ExcludePaths []string // skip any path containing one of these substrings + ExcludePaths []string // skip any path that equals or is under one of these prefixes } // SameSite reports whether u belongs to the seed's site: the same host, or a @@ -253,17 +253,40 @@ func InScope(seed, u *url.URL, cfg ScopeConfig) bool { if !SameSite(seed, u, cfg.IncludeSubdomains) { return false } - if cfg.ScopePrefix != "" && !strings.HasPrefix(u.Path, cfg.ScopePrefix) { + if cfg.ScopePrefix != "" && !pathHasPrefix(u.Path, cfg.ScopePrefix) { return false } for _, ex := range cfg.ExcludePaths { - if ex != "" && strings.Contains(u.Path, ex) { + if ex != "" && pathHasPrefix(u.Path, ex) { return false } } return true } +// pathHasPrefix reports whether path equals prefix or is a descendant of it. +// Both sides are treated as URL paths: a prefix of "/api" matches "/api", +// "/api/", and "/api/v1", but not "/apiv1" or "/map/api". A prefix without a +// leading slash is normalised to one so CLI prefixes such as "api" behave like +// "/api" for both --scope-prefix and --exclude. +func pathHasPrefix(path, prefix string) bool { + if prefix == "" { + return false + } + if !strings.HasPrefix(prefix, "/") { + prefix = "/" + prefix + } + // Exact match, or prefix followed by '/' so "/api" does not match "/apiv1". + if path == prefix || path == strings.TrimSuffix(prefix, "/") { + return true + } + p := prefix + if !strings.HasSuffix(p, "/") { + p += "/" + } + return strings.HasPrefix(path, p) +} + // LikelyPage reports whether an target should be rendered as a page // rather than downloaded as a file. Links ending in a known binary/document // extension are treated as assets. diff --git a/urlx/urlx_test.go b/urlx/urlx_test.go index 265f0b0..071d7e1 100644 --- a/urlx/urlx_test.go +++ b/urlx/urlx_test.go @@ -103,9 +103,18 @@ func TestInScope(t *testing.T) { {"https://other.com/a", ScopeConfig{}, false}, {"https://sub.ex.com/a", ScopeConfig{}, false}, {"https://sub.ex.com/a", ScopeConfig{IncludeSubdomains: true}, true}, + {"https://ex.com/docs", ScopeConfig{ScopePrefix: "/docs/"}, true}, {"https://ex.com/docs/x", ScopeConfig{ScopePrefix: "/docs/"}, true}, + {"https://ex.com/docs/x", ScopeConfig{ScopePrefix: "docs"}, true}, + {"https://ex.com/documentation", ScopeConfig{ScopePrefix: "/docs"}, false}, {"https://ex.com/blog/x", ScopeConfig{ScopePrefix: "/docs/"}, false}, - {"https://ex.com/a/private/x", ScopeConfig{ExcludePaths: []string{"/private/"}}, false}, + // Exclude is a path prefix, not a substring: /private matches /private + // and /private/x, but not /a/private/x or /privatething. + {"https://ex.com/private/x", ScopeConfig{ExcludePaths: []string{"/private"}}, false}, + {"https://ex.com/private", ScopeConfig{ExcludePaths: []string{"/private"}}, false}, + {"https://ex.com/private/x", ScopeConfig{ExcludePaths: []string{"private"}}, false}, + {"https://ex.com/a/private/x", ScopeConfig{ExcludePaths: []string{"/private"}}, true}, + {"https://ex.com/privatething", ScopeConfig{ExcludePaths: []string{"/private"}}, true}, {"https://ex.com/a/public/x", ScopeConfig{ExcludePaths: []string{"/private/"}}, true}, } for _, c := range cases {