Sample code for 30+ languages & platforms
Zig

Avoid URLs Matching Any of a Set of Patterns

See more Spider Examples

Demonstrates how to use "avoid patterns" to prevent spidering any URL that matches a wildcarded pattern. This example avoids URLs containing the substrings "java", "python", or "perl".

Chilkat Zig Downloads

Zig
const std = @import("std");
const chilkat = @import("chilkat");

pub fn main(init: std.process.Init) !void {
    const alloc = init.arena.allocator();

    const spider = try chilkat.Spider.init();
    defer spider.deinit();

    // The spider object crawls a single web site at a time.  As you'll see
    // in later examples, you can collect outbound links and use them to
    // crawl the web.  For now, we'll simply spider 10 pages of chilkatsoft.com
    spider.initialize("www.chilkatsoft.com");

    // Add the 1st URL:
    spider.addUnspidered("http://www.chilkatsoft.com/");

    // Avoid URLs matching these patterns:
    spider.addAvoidPattern("*java*");
    spider.addAvoidPattern("*python*");
    spider.addAvoidPattern("*perl*");

    // Begin crawling the site by calling CrawlNext repeatedly.
    var i: i32 = 0;
    i = 0;
    while (i <= 9) : (i += 1) {
        if (spider.crawlNext()) {
            // Show the URL of the page just spidered.
            std.debug.print("{s}\n", .{try spider.getLastUrl(alloc)});
            // The HTML is available in the LastHtml property
        } else |_| {
            // Did we get an error or are there no more URLs to crawl?
            if (spider.getNumUnspidered() == 0) {
                std.debug.print("No more URLs to spider\n", .{});
            } else {
                std.debug.print("{s}\n", .{try spider.getLastErrorText(alloc)});
            }
        }

        // Sleep 1 second before spidering the next URL.
        spider.sleepMs(1000);
    }
}