1. Go to this page and download the library: Download commently/crawler library. Choose the download type require.
2. Extract the ZIP file and open the index.php.
3. Add this code to the index.php.
<?php
require_once('vendor/autoload.php');
/* Start to develop here. Best regards https://php-download.com/ */
commently / crawler example snippets
use Commently\Crawler\CrawlRequest;
use Commently\Crawler\Crawler;
use Commently\Crawler\CrawlOutcome;
$response = app(Crawler::class)->fetch(new CrawlRequest('https://example.com/feed.xml'));
if ($response->outcome === CrawlOutcome::Success) {
file_put_contents('feed.xml', $response->body);
}
use Commently\Crawler\CrawlRequest;
use Commently\Crawler\Crawler;
use Commently\Crawler\CrawlOutcome;
$response = app(Crawler::class)->fetch(
new CrawlRequest(
url: 'https://example.com/feed.xml',
etag: $previousEtag, // sent as If-None-Match
lastModified: $previousDate, // sent as If-Modified-Since
)
);
match ($response->outcome) {
CrawlOutcome::Success => $this->store($response->body, $response->etag, $response->lastModified),
CrawlOutcome::NotModified => $this->keepCachedCopy(), // status 304
CrawlOutcome::Error => $this->retryLater($response->retryAfterSeconds, $response->error),
CrawlOutcome::Throttled => $this->retryLater(30), // host politeness said "wait"
CrawlOutcome::Blocked => $this->skip(), // robots.txt said "no"
};
use Commently\Crawler\CrawlRequest;
use Commently\Crawler\Crawler;
$responses = app(Crawler::class)->fetchMany([
new CrawlRequest('https://a.example.com/feed', key: 'a'),
new CrawlRequest('https://b.example.com/feed', key: 'b'),
new CrawlRequest('https://c.example.com/feed', key: 'c'),
]);
foreach ($responses as $key => $response) {
// $responses contains a result only for requests that were actually sent.
// Requests skipped by politeness (host not eligible, lock busy, robots.txt)
// are absent — retry them on a later run.
}
use Commently\Crawler\Contracts\Sink;
use Commently\Crawler\CrawlResponse;
class StoreFeedSink implements Sink
{
public function handle(CrawlResponse $response): void
{
// parse $response->body and store it somewhere
}
}
use App\Sinks\StoreFeedSink;
use Commently\Crawler\Contracts\Sink;
$this->app->bind(Sink::class, StoreFeedSink::class);
use Commently\Crawler\CrawlRequest;
use Commently\Crawler\Jobs\CrawlUrl;
CrawlUrl::dispatch(new CrawlRequest('https://example.com/feed.xml'));
// config/crawler.php
return [
// Single honest crawler User-Agent. Never rotated, never browser-masqueraded.
'user_agent' => env('RSS_CRAWLER_USER_AGENT', 'CommentlyBot/1.0 (+https://example.com/bot)'),
'http' => [
'connect_timeout' => 5, // seconds for the TCP/TLS handshake
'timeout' => 20, // seconds for the whole request
'max_concurrency' => 15, // max simultaneous requests across all hosts
'max_redirects' => 5,
'http_version' => '1.1', // '1.1' avoids HTTP/2 resets on some CDNs
],
'rate_limit' => [
'per_host_delay' => 30, // min pause between two requests to the same host (s)
'max_concurrent_per_host' => 1, // max concurrent requests to the same host
'lock_ttl' => 120, // TTL of the distributed host locks (s)
],
'robots' => [
'enabled' => true,
'cache_ttl' => 86400, // robots.txt is fetched at most once per host per TTL
],
];