1 The Problem
We want a scraper that downloads a web page and pulls out specific pieces — titles, links, prices. It teaches fetching HTML and extracting data from it, plus the responsibility that comes with scraping (respecting sites and their rules).
2 How to Think About It
Strip away the library that usually hides this: an HTTP GET is a socket connection, a few lines of text sent, and a response read back.
getaddrinfo and connect a TCP socket to it. → 2. Write a valid HTTP/1.1 request line and headers, ending in a blank line. → 3. Read the raw response bytes back with recv. → 4. Split it into a status line and body on the blank line HTTP itself defines. → 5. Scan the body for href="..." occurrences.
3 The Build — explained part by part
Here is the complete scraper. C has no standard-library HTTP client at all — not even the bare-bones one Go or Java ship — so writing one by hand over BSD sockets (<sys/socket.h>) is not a workaround here, it is simply what an HTTP client in C looks like underneath any library.
#ifndef WEB_SCRAPER_H
#define WEB_SCRAPER_H
#define MAX_RESPONSE 65536
#define MAX_LINKS 128
#define MAX_LINK_LEN 256
/* A tiny hand-written URL splitter -- just enough for
* "http://host[:port]/path". Returns 1 on success. */
int parse_url(const char *url, char *host, int host_cap, int *port, char *path, int path_cap);
/* A minimal, hand-written HTTP/1.1 GET client over a raw socket. C has no
* standard-library HTTP client at all (unlike Go's net/http or even Java's
* java.net.http), so this project does by hand what any HTTP library does
* underneath: open a socket, write the request line and headers ourselves,
* and read the raw response back. Returns the number of bytes read into
* `out` (capped at out_cap - 1, always NUL-terminated), or -1 on failure. */
int fetch(const char *host, int port, const char *path, char *out, int out_cap);
/* Splits a raw HTTP/1.1 response into a status line and a body, at the
* blank line ("\r\n\r\n") HTTP itself defines as the boundary. Writes into
* caller-supplied buffers; returns 1 if a blank-line boundary was found. */
int split_response(const char *raw, char *status_line, int status_cap, char *body, int body_cap);
/* C's standard library ships no regex engine either, so link extraction is
* a small hand-written scanner: find every href="..." occurrence and pull
* out the quoted text. Less flexible than a real regex, but dependency-free
* and entirely readable. Returns the number of links found. */
int extract_links(const char *html, char links[][MAX_LINK_LEN], int max_links);
#endif
#define _POSIX_C_SOURCE 200809L
#include "WebScraper.h"
#include <string.h>
#include <stdio.h>
#include <stdlib.h>
#include <unistd.h>
#include <sys/socket.h>
#include <netdb.h>
int parse_url(const char *url, char *host, int host_cap, int *port, char *path, int path_cap) {
const char *prefix = "http://";
size_t prefix_len = strlen(prefix);
if (strncmp(url, prefix, prefix_len) != 0) return 0;
const char *rest = url + prefix_len;
const char *slash = strchr(rest, '/');
const char *authority_end = slash ? slash : rest + strlen(rest);
char authority[300];
size_t auth_len = (size_t)(authority_end - rest);
if (auth_len == 0 || auth_len >= sizeof(authority)) return 0;
memcpy(authority, rest, auth_len);
authority[auth_len] = '\0';
char *colon = strchr(authority, ':');
if (colon) {
*colon = '\0';
*port = atoi(colon + 1);
} else {
*port = 80;
}
if (strlen(authority) >= (size_t)host_cap) return 0;
strcpy(host, authority);
if (slash) {
strncpy(path, slash, (size_t)path_cap - 1);
} else {
strncpy(path, "/", (size_t)path_cap - 1);
}
path[path_cap - 1] = '\0';
return 1;
}
int fetch(const char *host, int port, const char *path, char *out, int out_cap) {
char port_str[16];
snprintf(port_str, sizeof(port_str), "%d", port);
struct addrinfo hints, *res;
memset(&hints, 0, sizeof(hints));
hints.ai_family = AF_INET;
hints.ai_socktype = SOCK_STREAM;
if (getaddrinfo(host, port_str, &hints, &res) != 0) return -1;
int fd = socket(res->ai_family, res->ai_socktype, res->ai_protocol);
if (fd < 0) { freeaddrinfo(res); return -1; }
if (connect(fd, res->ai_addr, res->ai_addrlen) < 0) {
close(fd);
freeaddrinfo(res);
return -1;
}
freeaddrinfo(res);
char request[512];
int req_len = snprintf(request, sizeof(request),
"GET %s HTTP/1.1\r\nHost: %s\r\nUser-Agent: codex-scraper/1.0\r\nConnection: close\r\n\r\n",
path, host);
if (send(fd, request, (size_t)req_len, 0) < 0) {
close(fd);
return -1;
}
int total = 0;
ssize_t n;
while (total < out_cap - 1 &&
(n = recv(fd, out + total, (size_t)(out_cap - 1 - total), 0)) > 0) {
total += (int)n;
}
out[total] = '\0';
close(fd);
return total;
}
int split_response(const char *raw, char *status_line, int status_cap, char *body, int body_cap) {
const char *sep = strstr(raw, "\r\n\r\n");
if (!sep) return 0;
const char *line_end = strstr(raw, "\r\n");
if (!line_end || line_end > sep) line_end = sep;
size_t line_len = (size_t)(line_end - raw);
if (line_len >= (size_t)status_cap) line_len = (size_t)status_cap - 1;
memcpy(status_line, raw, line_len);
status_line[line_len] = '\0';
const char *body_start = sep + 4;
size_t body_len = strlen(body_start);
if (body_len >= (size_t)body_cap) body_len = (size_t)body_cap - 1;
memcpy(body, body_start, body_len);
body[body_len] = '\0';
return 1;
}
int extract_links(const char *html, char links[][MAX_LINK_LEN], int max_links) {
int count = 0;
const char *needle = "href=\"";
size_t needle_len = strlen(needle);
const char *p = html;
while (count < max_links) {
const char *found = strstr(p, needle);
if (!found) break;
const char *start = found + needle_len;
const char *end = strchr(start, '"');
if (!end) break;
size_t len = (size_t)(end - start);
if (len >= MAX_LINK_LEN) len = MAX_LINK_LEN - 1;
memcpy(links[count], start, len);
links[count][len] = '\0';
count++;
p = end + 1;
}
return count;
}
#include "WebScraper.h"
#include <stdio.h>
int main(int argc, char **argv) {
const char *url = argc > 1 ? argv[1] : "http://127.0.0.1:8080/";
char host[256], path[512];
int port;
if (!parse_url(url, host, sizeof(host), &port, path, sizeof(path))) {
fprintf(stderr, "Could not parse URL: %s (expected http://host[:port]/path)\n", url);
return 1;
}
static char raw[MAX_RESPONSE];
if (fetch(host, port, path, raw, sizeof(raw)) < 0) {
fprintf(stderr, "Request failed\n");
return 1;
}
char status_line[256];
static char body[MAX_RESPONSE];
if (!split_response(raw, status_line, sizeof(status_line), body, sizeof(body))) {
fprintf(stderr, "Could not parse response\n");
return 1;
}
printf("%s\n", status_line);
static char links[MAX_LINKS][MAX_LINK_LEN];
int n = extract_links(body, links, MAX_LINKS);
printf("Found %d link(s):\n", n);
for (int i = 0; i < n; i++) {
printf(" %s\n", links[i]);
}
return 0;
}
connect() can use, replacing the older gethostbyname. It works identically whether host is a real domain name or a literal IP like 127.0.0.1.send(fd, request, req_len, 0) / recv(fd, ...) — the raw socket primitives every HTTP library in every language is eventually built on. There is no HTTP awareness at this layer at all, just bytes in, bytes out.
snprintf(request, ..., "GET %s HTTP/1.1\r\nHost: %s\r\n...\r\n\r\n", ...) — a valid HTTP/1.1 request is a plain text protocol: a request line, headers each ending in
\r\n, and a blank line marking the end of headers. Writing it by hand is the entire “request” a library builds for you.strstr(raw, "\r\n\r\n") — that same blank line is how you find where headers end and the body begins in the response; HTTP defines this boundary explicitly, so no guessing is involved.
extract_links — a hand-written scanner: repeatedly
strstr for href=", then strchr for the closing quote, and copy what is between them. C’s own library actually does ship a real regex engine (POSIX <regex.h>, unlike Rust or Go’s standard libraries), but this project hand-writes the scanner anyway for the same reason every other language’s version of this project does: it is dependency-free, fully readable, and correct for well-formed HTML — see “Try this next” below for the regex.h version.the local-server test — rather than only testing
extract_links on a hand-written string, one test spawns a real server on a background pthread, bound to an OS-assigned port via bind(..., htons(0)), and confirms fetch genuinely round-trips over a real socket — the same rigor Rust’s and Go’s versions of this project use.
Connection: close in the request headers — without it, a real server may keep the connection open waiting for another request, and the recv loop here would then block forever waiting for the stream to end.Connection: close for a one-shot client like this one, as the code above does.recv’s return value — it can return 0 (connection closed) or a negative number (error), and treating either as “more data arrived” corrupts the buffer or spins forever.while ((n = recv(...)) > 0), as fetch does, and stop cleanly otherwise.4 Test & Prove Each Part
We test link extraction on known strings, the header/body split, the URL parser, and — the important one — a real fetch against a real local server.
#include "WebScraper.h"
#include <assert.h>
#include <stdio.h>
#include <string.h>
#include <unistd.h>
#include <pthread.h>
#include <sys/socket.h>
#include <netinet/in.h>
#define RUN(name) do { name(); printf("PASS: %s\n", #name); } while (0)
static void extracts_links_from_html(void) {
const char *html = "<a href=\"/about\">About</a><a href=\"https://example.com\">Ex</a>";
char links[MAX_LINKS][MAX_LINK_LEN];
int n = extract_links(html, links, MAX_LINKS);
assert(n == 2);
assert(strcmp(links[0], "/about") == 0);
assert(strcmp(links[1], "https://example.com") == 0);
}
static void returns_no_links_for_plain_text(void) {
char links[MAX_LINKS][MAX_LINK_LEN];
int n = extract_links("just some text, no tags here", links, MAX_LINKS);
assert(n == 0);
}
static void splits_headers_from_body_on_the_blank_line(void) {
const char *raw = "HTTP/1.1 200 OK\r\nContent-Type: text/html\r\n\r\n<html>hi</html>";
char status[256], body[256];
int ok = split_response(raw, status, sizeof(status), body, sizeof(body));
assert(ok);
assert(strcmp(status, "HTTP/1.1 200 OK") == 0);
assert(strcmp(body, "<html>hi</html>") == 0);
}
static void parses_a_simple_url(void) {
char host[256], path[512];
int port;
int ok = parse_url("http://example.com:9090/links", host, sizeof(host), &port, path, sizeof(path));
assert(ok);
assert(strcmp(host, "example.com") == 0);
assert(port == 9090);
assert(strcmp(path, "/links") == 0);
}
static void a_url_with_no_path_defaults_to_slash(void) {
char host[256], path[512];
int port;
int ok = parse_url("http://example.com", host, sizeof(host), &port, path, sizeof(path));
assert(ok);
assert(port == 80); /* no port given -> default HTTP port */
assert(strcmp(path, "/") == 0);
}
/* One shared struct so the server thread can hand its assigned port back to
* the test thread once bind() has picked one. */
typedef struct {
int listen_fd;
int port;
} ServerCtx;
static void *serve_one_request(void *arg) {
ServerCtx *ctx = arg;
int client_fd = accept(ctx->listen_fd, NULL, NULL);
char discard[512];
recv(client_fd, discard, sizeof(discard), 0); /* just consume the request */
const char *body =
"<html><body><a href=\"/one\">One</a>"
"<a href=\"/two\">Two</a></body></html>";
char response[512];
int len = snprintf(response, sizeof(response),
"HTTP/1.1 200 OK\r\nContent-Length: %zu\r\nConnection: close\r\n\r\n%s",
strlen(body), body);
send(client_fd, response, (size_t)len, 0);
close(client_fd);
close(ctx->listen_fd);
return NULL;
}
/* End-to-end: starts a real local TCP server on an OS-assigned port, serves
* one canned HTML response, and confirms fetch() + extract_links() pull the
* right links out of an actual network round trip, not just a hand-fed
* string. */
static void fetches_and_extracts_links_from_a_real_local_server(void) {
int listen_fd = socket(AF_INET, SOCK_STREAM, 0);
assert(listen_fd >= 0);
int opt = 1;
setsockopt(listen_fd, SOL_SOCKET, SO_REUSEADDR, &opt, sizeof(opt));
struct sockaddr_in addr;
memset(&addr, 0, sizeof(addr));
addr.sin_family = AF_INET;
addr.sin_addr.s_addr = htonl(INADDR_LOOPBACK);
addr.sin_port = htons(0); /* let the OS pick a free port */
assert(bind(listen_fd, (struct sockaddr *)&addr, sizeof(addr)) == 0);
assert(listen(listen_fd, 1) == 0);
socklen_t len = sizeof(addr);
getsockname(listen_fd, (struct sockaddr *)&addr, &len);
int port = ntohs(addr.sin_port);
ServerCtx ctx = { .listen_fd = listen_fd, .port = port };
pthread_t server_thread;
pthread_create(&server_thread, NULL, serve_one_request, &ctx);
static char raw[MAX_RESPONSE];
int n = fetch("127.0.0.1", port, "/", raw, sizeof(raw));
pthread_join(server_thread, NULL);
assert(n > 0);
char status[256], body[MAX_RESPONSE];
assert(split_response(raw, status, sizeof(status), body, sizeof(body)));
char links[MAX_LINKS][MAX_LINK_LEN];
int link_count = extract_links(body, links, MAX_LINKS);
assert(link_count == 2);
assert(strcmp(links[0], "/one") == 0);
assert(strcmp(links[1], "/two") == 0);
}
int main(void) {
RUN(extracts_links_from_html);
RUN(returns_no_links_for_plain_text);
RUN(splits_headers_from_body_on_the_blank_line);
RUN(parses_a_simple_url);
RUN(a_url_with_no_path_defaults_to_slash);
RUN(fetches_and_extracts_links_from_a_real_local_server);
printf("All tests passed.\n");
return 0;
}
Compile and run with gcc -std=c17 -Wall -Wextra -Wpedantic -pthread -o test_run WebScraper.c test_WebScraper.c && ./test_run. The last test is the one worth reading closely: it starts a background thread running a real socket server, serves one hand-built HTTP response, and lets fetch connect to it exactly as it would to a real website — proving the socket code works over an actual network round trip, not just against a string.
5 The Interface
What it expects
http://127.0.0.1:8099/index.htmlWhat it returns
HTTP/1.0 200 OK
Found 2 link(s):
/about
https://example.com6 Run It & Automate It
Save the code as WebScraper.h / WebScraper.c / main.c and compile it with gcc — that turns your source directly into a native executable for your machine. No separate runtime needed: the compiled binary runs on its own.
gcc -o scrape main.c WebScraper.c && ./scrape http://127.0.0.1:8099/index.htmlPoint it at any plain HTTP (not HTTPS — this client has no TLS) server, including one you started locally with
python3 -m http.server.A CI tool like Jenkins runs the same compile-then-test-then-check-for-leaks steps automatically whenever the code changes — every line below has a plain explanation.
$ python3 -m http.server 8099 &
$ ./scrape http://127.0.0.1:8099/index.html
HTTP/1.0 200 OK
Found 2 link(s):
/about
https://example.comgetaddrinfo could not resolve the host. Start a server there first, or check the port number.http:// URLs. An https:// URL needs TLS, which this hand-rolled client deliberately does not implement.Connection: close — without it, the recv loop can block forever waiting for a server that keeps the connection open.// Jenkinsfile — compiles, tests, and checks for leaks on every change.
pipeline {
agent any
stages {
stage('Get the code') {
// download the latest code
steps { checkout scm }
}
stage('Compile') {
steps {
// confirm a compiler is installed
sh 'gcc --version'
// compile with strict warnings on
sh 'gcc -std=c17 -Wall -Wextra -o app *.c -pthread'
}
}
stage('Run the tests') {
steps {
// prints PASS/FAIL, exits non-zero on failure
sh './app'
}
}
stage('Check for memory leaks') {
steps {
// fails the build on any leak or invalid access
sh 'valgrind --error-exitcode=1 --leak-check=full ./app'
}
}
}
post {
success { echo 'All tests passed, no leaks found.' }
failure { echo 'A test or Valgrind check failed — see above.' }
}
}
- Follow redirects. Check for a 3xx status and a
Locationheader, then fetch again. (Teaches: reading response headers, not just the body.) - Use POSIX
regex.h. Replaceextract_linkswithregcomp/regexecand compare how the code reads. (Teaches: what C’s own standard regex engine buys you over hand-rolled scanning.) - Add a timeout. Use
setsockoptwithSO_RCVTIMEOso a hung server cannot block forever. (Teaches: socket options.)
<regex.h>), and tested it against a real local server on a background thread rather than only against strings. Related: The Standard Library, Concurrency in C.