Zig
Zig
A Simple Web Crawler
See more Spider Examples
This demonstrates a very simple web crawler using the Chilkat Spider component.Chilkat Zig Downloads
const std = @import("std");
const chilkat = @import("chilkat");
pub fn main(init: std.process.Init) !void {
const alloc = init.arena.allocator();
var success: bool = false;
const spider = try chilkat.Spider.init();
defer spider.deinit();
const seen_domains = try chilkat.StringArray.init();
defer seen_domains.deinit();
const seed_urls = try chilkat.StringArray.init();
defer seed_urls.deinit();
seen_domains.setUnique(true);
seed_urls.setUnique(true);
// You will need to change the start URL to something else...
seed_urls.append("http://something.whateverYouWant.com/") catch {};
// Set outbound URL exclude patterns
// URLs matching any of these patterns will not be added to the
// collection of outbound links.
spider.addAvoidOutboundLinkPattern("*?id=*");
spider.addAvoidOutboundLinkPattern("*.mypages.*");
spider.addAvoidOutboundLinkPattern("*.personal.*");
spider.addAvoidOutboundLinkPattern("*.comcast.*");
spider.addAvoidOutboundLinkPattern("*.aol.*");
spider.addAvoidOutboundLinkPattern("*~*");
// Use a cache so we don't have to re-fetch URLs previously fetched.
spider.setCacheDir("c:/spiderCache/");
spider.setFetchFromCache(true);
spider.setUpdateCache(true);
while (seed_urls.getCount() > 0) {
var url: [:0]const u8 = try seed_urls.pop(alloc);
spider.initialize(url);
// Spider 5 URLs of this domain.
// but first, save the base domain in seenDomains
var domain: [:0]const u8 = try spider.getUrlDomain(alloc, url);
seen_domains.append(try spider.getBaseDomain(alloc, domain)) catch {};
var i: i32 = 0;
var num_crawled: i32 = 0;
success = true;
while ((success) and (num_crawled < 5)) {
if (spider.crawlNext()) {
// Display the URL we just crawled.
std.debug.print("{s}\n", .{try spider.getLastUrl(alloc)});
// If the last URL was retrieved from cache,
// we won't wait. Otherwise we'll wait 1 second
// before fetching the next URL.
if (!spider.getLastFromCache()) {
spider.sleepMs(1000);
}
num_crawled = num_crawled + 1;
} else |_| {}
// If CrawlNext fails (no more URLs to crawl in this domain), success is false and the loop exits.
}
// Add the outbound links to seedUrls, except
// for the domains we've already seen.
i = 0;
while (i < spider.getNumOutboundLinks()) : (i += 1) {
url = try spider.getOutboundLink(alloc, i);
domain = try spider.getUrlDomain(alloc, url);
const base_domain = try spider.getBaseDomain(alloc, domain);
if (!seen_domains.contains(base_domain)) {
// Don't let our list of seedUrls grow too large.
if (seed_urls.getCount() < 1000) {
seed_urls.append(url) catch {};
}
}
}
}
}