Sample code for 30+ languages & platforms
Zig

A Simple Web Crawler

See more Spider Examples

This demonstrates a very simple web crawler using the Chilkat Spider component.

Chilkat Zig Downloads

Zig
const std = @import("std");
const chilkat = @import("chilkat");

pub fn main(init: std.process.Init) !void {
    const alloc = init.arena.allocator();

    var success: bool = false;

    const spider = try chilkat.Spider.init();
    defer spider.deinit();

    const seen_domains = try chilkat.StringArray.init();
    defer seen_domains.deinit();
    const seed_urls = try chilkat.StringArray.init();
    defer seed_urls.deinit();

    seen_domains.setUnique(true);
    seed_urls.setUnique(true);

    // You will need to change the start URL to something else...
    seed_urls.append("http://something.whateverYouWant.com/") catch {};

    // Set outbound URL exclude patterns
    // URLs matching any of these patterns will not be added to the
    // collection of outbound links.
    spider.addAvoidOutboundLinkPattern("*?id=*");
    spider.addAvoidOutboundLinkPattern("*.mypages.*");
    spider.addAvoidOutboundLinkPattern("*.personal.*");
    spider.addAvoidOutboundLinkPattern("*.comcast.*");
    spider.addAvoidOutboundLinkPattern("*.aol.*");
    spider.addAvoidOutboundLinkPattern("*~*");

    // Use a cache so we don't have to re-fetch URLs previously fetched.
    spider.setCacheDir("c:/spiderCache/");
    spider.setFetchFromCache(true);
    spider.setUpdateCache(true);

    while (seed_urls.getCount() > 0) {
        var url: [:0]const u8 = try seed_urls.pop(alloc);
        spider.initialize(url);

        // Spider 5 URLs of this domain.
        // but first, save the base domain in seenDomains
        var domain: [:0]const u8 = try spider.getUrlDomain(alloc, url);
        seen_domains.append(try spider.getBaseDomain(alloc, domain)) catch {};

        var i: i32 = 0;
        var num_crawled: i32 = 0;
        success = true;
        while ((success) and (num_crawled < 5)) {
            if (spider.crawlNext()) {
                // Display the URL we just crawled.
                std.debug.print("{s}\n", .{try spider.getLastUrl(alloc)});

                // If the last URL was retrieved from cache,
                // we won't wait.  Otherwise we'll wait 1 second
                // before fetching the next URL.
                if (!spider.getLastFromCache()) {
                    spider.sleepMs(1000);
                }

                num_crawled = num_crawled + 1;
            } else |_| {}

            // If CrawlNext fails (no more URLs to crawl in this domain), success is false and the loop exits.
        }

        // Add the outbound links to seedUrls, except
        // for the domains we've already seen.
        i = 0;
        while (i < spider.getNumOutboundLinks()) : (i += 1) {
            url = try spider.getOutboundLink(alloc, i);
            domain = try spider.getUrlDomain(alloc, url);
            const base_domain = try spider.getBaseDomain(alloc, domain);
            if (!seen_domains.contains(base_domain)) {
                // Don't let our list of seedUrls grow too large.
                if (seed_urls.getCount() < 1000) {
                    seed_urls.append(url) catch {};
                }
            }
        }
    }
}