feat: 1. add a crawl delay function to honor the Crawl-delay directive parsed from robots.txt during clone. (#57)
2. add --craw-delay flag to specify/override robots directive.
This commit is contained in:
+33
-3
@@ -19,6 +19,7 @@ import (
|
||||
"github.com/tamnd/kage/sanitize"
|
||||
"github.com/tamnd/kage/urlx"
|
||||
"golang.org/x/net/html"
|
||||
"golang.org/x/time/rate"
|
||||
)
|
||||
|
||||
// Logf is an optional sink for human-readable progress lines.
|
||||
@@ -43,9 +44,12 @@ type Cloner struct {
|
||||
mu sync.Mutex
|
||||
seenAssets map[string]bool
|
||||
enqueued int // pages offered to the queue
|
||||
wg sync.WaitGroup
|
||||
pageJobs chan pageItem
|
||||
assetJobs chan assetItem
|
||||
|
||||
crawlLimiter *rate.Limiter
|
||||
|
||||
wg sync.WaitGroup
|
||||
pageJobs chan pageItem
|
||||
assetJobs chan assetItem
|
||||
|
||||
muContent sync.Mutex
|
||||
seenContent map[string]string // sha-256 of page bytes -> first path written
|
||||
@@ -144,6 +148,7 @@ func (c *Cloner) Run(ctx context.Context) (Result, error) {
|
||||
defer func() { _ = c.pool.Close() }()
|
||||
|
||||
c.loadRobots(ctx)
|
||||
c.setupCrawlDelayLimiter()
|
||||
|
||||
// Start workers.
|
||||
var workers sync.WaitGroup
|
||||
@@ -218,6 +223,19 @@ func (c *Cloner) loadRobots(ctx context.Context) {
|
||||
c.robots = robots.Parse(string(data), "kage")
|
||||
}
|
||||
|
||||
func (c *Cloner) setupCrawlDelayLimiter() {
|
||||
delay := c.cfg.CrawlDelay
|
||||
if delay <= 0 && c.cfg.RespectRobots && c.robots != nil {
|
||||
delay = c.robots.CrawlDelay
|
||||
}
|
||||
if delay <= 0 {
|
||||
c.crawlLimiter = nil
|
||||
return
|
||||
}
|
||||
|
||||
c.crawlLimiter = rate.NewLimiter(rate.Every(delay), 1)
|
||||
}
|
||||
|
||||
// seedSitemaps adds in-scope sitemap URLs (from robots and the default path) to
|
||||
// the frontier.
|
||||
func (c *Cloner) seedSitemaps(ctx context.Context) {
|
||||
@@ -252,6 +270,9 @@ func (c *Cloner) processPage(ctx context.Context, j pageItem) {
|
||||
c.stats.skipped.Add(1)
|
||||
return
|
||||
}
|
||||
if !c.waitForCrawlDelay(ctx) {
|
||||
return
|
||||
}
|
||||
|
||||
res, err := c.pool.Render(ctx, j.u.String())
|
||||
if err != nil {
|
||||
@@ -324,6 +345,15 @@ func (c *Cloner) processPage(ctx context.Context, j pageItem) {
|
||||
c.stats.recordPage(c.pagePathKey(j.u), deduped)
|
||||
}
|
||||
|
||||
// waitForCrawlDelay spaces page render starts.
|
||||
func (c *Cloner) waitForCrawlDelay(ctx context.Context) bool {
|
||||
if c.crawlLimiter == nil {
|
||||
return true
|
||||
}
|
||||
|
||||
return c.crawlLimiter.Wait(ctx) == nil
|
||||
}
|
||||
|
||||
// processAsset downloads one asset, rewriting CSS references on the way, and
|
||||
// writes it to its deterministic local path.
|
||||
func (c *Cloner) processAsset(ctx context.Context, j assetItem) {
|
||||
|
||||
@@ -12,6 +12,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/tamnd/kage/browser"
|
||||
"github.com/tamnd/kage/robots"
|
||||
"github.com/tamnd/kage/urlx"
|
||||
)
|
||||
|
||||
@@ -178,6 +179,56 @@ func TestPageKeyCollapsesDuplicates(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCrawlDelaySpacesPageStarts(t *testing.T) {
|
||||
seed, _ := urlx.ParseSeed("https://ex.com")
|
||||
cfg := DefaultConfig()
|
||||
cfg.RespectRobots = true
|
||||
c := New(seed, cfg, nil)
|
||||
c.robots = &robots.Matcher{CrawlDelay: 20 * time.Millisecond}
|
||||
c.setupCrawlDelayLimiter()
|
||||
|
||||
ctx := context.Background()
|
||||
if !c.waitForCrawlDelay(ctx) {
|
||||
t.Fatal("first crawl-delay wait returned false")
|
||||
}
|
||||
|
||||
start := time.Now()
|
||||
if !c.waitForCrawlDelay(ctx) {
|
||||
t.Fatal("second crawl-delay wait returned false")
|
||||
}
|
||||
if elapsed := time.Since(start); elapsed < 15*time.Millisecond {
|
||||
t.Fatalf("second crawl-delay wait = %v, want at least 15ms", elapsed)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCrawlDelayFlagOverridesRobots(t *testing.T) {
|
||||
seed, _ := urlx.ParseSeed("https://ex.com")
|
||||
cfg := DefaultConfig()
|
||||
cfg.RespectRobots = true
|
||||
cfg.CrawlDelay = 20 * time.Millisecond
|
||||
c := New(seed, cfg, nil)
|
||||
c.robots = &robots.Matcher{CrawlDelay: time.Minute}
|
||||
c.setupCrawlDelayLimiter()
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 200*time.Millisecond)
|
||||
defer cancel()
|
||||
if !c.waitForCrawlDelay(ctx) {
|
||||
t.Fatal("first crawl-delay wait returned false")
|
||||
}
|
||||
|
||||
start := time.Now()
|
||||
if !c.waitForCrawlDelay(ctx) {
|
||||
t.Fatal("second crawl-delay wait returned false")
|
||||
}
|
||||
elapsed := time.Since(start)
|
||||
if elapsed < 15*time.Millisecond {
|
||||
t.Fatalf("second crawl-delay wait = %v, want at least 15ms", elapsed)
|
||||
}
|
||||
if elapsed > 150*time.Millisecond {
|
||||
t.Fatalf("second crawl-delay wait = %v, override likely ignored", elapsed)
|
||||
}
|
||||
}
|
||||
|
||||
func mustURL(t *testing.T, raw string) *url.URL {
|
||||
t.Helper()
|
||||
u, err := url.Parse(raw)
|
||||
|
||||
@@ -57,6 +57,7 @@ type Config struct {
|
||||
ExcludePaths []string
|
||||
|
||||
RespectRobots bool
|
||||
CrawlDelay time.Duration // override robots.txt Crawl-delay when > 0
|
||||
FollowSitemap bool
|
||||
Headless bool
|
||||
KeepNoscript bool
|
||||
|
||||
Reference in New Issue
Block a user