PHP code example of commently / crawler

1. Go to this page and download the library: Download commently/crawler library. Choose the download type require.

2. Extract the ZIP file and open the index.php.

3. Add this code to the index.php.
    
        
<?php
require_once('vendor/autoload.php');

/* Start to develop here. Best regards https://php-download.com/ */

    

commently / crawler example snippets


use Commently\Crawler\CrawlRequest;
use Commently\Crawler\Crawler;
use Commently\Crawler\CrawlOutcome;

$response = app(Crawler::class)->fetch(new CrawlRequest('https://example.com/feed.xml'));

if ($response->outcome === CrawlOutcome::Success) {
    file_put_contents('feed.xml', $response->body);
}

use Commently\Crawler\CrawlRequest;
use Commently\Crawler\Crawler;
use Commently\Crawler\CrawlOutcome;

$response = app(Crawler::class)->fetch(
    new CrawlRequest(
        url: 'https://example.com/feed.xml',
        etag: $previousEtag,            // sent as If-None-Match
        lastModified: $previousDate,    // sent as If-Modified-Since
    )
);

match ($response->outcome) {
    CrawlOutcome::Success => $this->store($response->body, $response->etag, $response->lastModified),
    CrawlOutcome::NotModified => $this->keepCachedCopy(),          // status 304
    CrawlOutcome::Error => $this->retryLater($response->retryAfterSeconds, $response->error),
    CrawlOutcome::Throttled => $this->retryLater(30),               // host politeness said "wait"
    CrawlOutcome::Blocked => $this->skip(),                          // robots.txt said "no"
};

use Commently\Crawler\CrawlRequest;
use Commently\Crawler\Crawler;

$responses = app(Crawler::class)->fetchMany([
    new CrawlRequest('https://a.example.com/feed', key: 'a'),
    new CrawlRequest('https://b.example.com/feed', key: 'b'),
    new CrawlRequest('https://c.example.com/feed', key: 'c'),
]);

foreach ($responses as $key => $response) {
    // $responses contains a result only for requests that were actually sent.
    // Requests skipped by politeness (host not eligible, lock busy, robots.txt)
    // are absent — retry them on a later run.
}

$responses = $crawler->fetchMany($requests, permit: fn (string $key, CrawlRequest $request) => $this->isNotAlreadyFetched($key));

use Commently\Crawler\Contracts\Sink;
use Commently\Crawler\CrawlResponse;

class StoreFeedSink implements Sink
{
    public function handle(CrawlResponse $response): void
    {
        // parse $response->body and store it somewhere
    }
}

use App\Sinks\StoreFeedSink;
use Commently\Crawler\Contracts\Sink;

$this->app->bind(Sink::class, StoreFeedSink::class);

use Commently\Crawler\CrawlRequest;
use Commently\Crawler\Jobs\CrawlUrl;

CrawlUrl::dispatch(new CrawlRequest('https://example.com/feed.xml'));

// config/crawler.php
return [

    // Single honest crawler User-Agent. Never rotated, never browser-masqueraded.
    'user_agent' => env('RSS_CRAWLER_USER_AGENT', 'CommentlyBot/1.0 (+https://example.com/bot)'),

    'http' => [
        'connect_timeout' => 5,    // seconds for the TCP/TLS handshake
        'timeout'         => 20,   // seconds for the whole request
        'max_concurrency' => 15,   // max simultaneous requests across all hosts
        'max_redirects'   => 5,
        'http_version'    => '1.1', // '1.1' avoids HTTP/2 resets on some CDNs
    ],

    'rate_limit' => [
        'per_host_delay'          => 30,  // min pause between two requests to the same host (s)
        'max_concurrent_per_host' => 1,   // max concurrent requests to the same host
        'lock_ttl'                => 120, // TTL of the distributed host locks (s)
    ],

    'robots' => [
        'enabled'   => true,
        'cache_ttl' => 86400, // robots.txt is fetched at most once per host per TTL
    ],

];
bash
php artisan vendor:publish --tag=crawler-config