← thecodex.expert · The Codex Family of Knowledge
Tier 1 · Beginner · C Project

Word Counter

Read a text file and report its basic statistics, then rank its most frequent words — the kind of small utility real command-line tools are built from. C has no dictionary/map type, so you'll build one from scratch.

🧠 Teaches how to think spoonfed, every age Last verified:

1 The Problem

We want a tool that takes some text — typed in or read from a file — and reports how many words, characters, and lines it has. It teaches the core string operations for breaking text into pieces and measuring them.

Where this shows up: word-count limits on forms and essays, reading-time estimates, search indexing, text analysis, validating input length. Measuring and slicing text is one of the most common jobs in software.

2 How to Think About It

Two separate jobs share one pass over the text: counting (lines/words/characters) and tallying (which words appear how often). Keeping them as separate, independently-testable functions is the whole design.

The plan — in plain English
1. Load the whole file into memory as one string. → 2. Count lines, words, characters, and bytes in a single scan. → 3. Split the text into lowercase words and tally each one into a small hand-rolled table. → 4. Sort that table by count and take the top few.

Get the text

Characters = length

Words = split on spaces, count

Lines = split on newlines, count

Show all three

3 The Build — explained part by part

Three functions do all the work: count_all for the raw statistics, word_frequency for the tally, and top_n to rank it. All are pure functions over a string — no file I/O inside them, which is what makes them testable without touching the disk.

CWordCounter.h / WordCounter.c / main.c
#ifndef WORD_COUNTER_H
#define WORD_COUNTER_H

typedef struct {
    long lines, words, chars, bytes;
} Counts;

typedef struct {
    char word[64];
    int count;
} WordCount;

/* Counts lines, words, characters, and bytes in `text`. */
Counts count_all(const char *text);

/* Splits `text` into lowercase words and tallies frequency into `out`
 * (capacity `max_out`). Returns the number of distinct words found. */
int word_frequency(const char *text, WordCount *out, int max_out);

/* Copies the `n` highest-count entries from `freq` (of length `freq_len`)
 * into `top`, sorted by count descending, ties broken alphabetically. */
void top_n(WordCount *freq, int freq_len, int n, WordCount *top);

#endif

#include "WordCounter.h"
#include <ctype.h>
#include <string.h>
#include <stdlib.h>

Counts count_all(const char *text) {
    Counts c = {0, 0, 0, 0};
    int in_word = 0;
    for (const char *p = text; *p; p++) {
        c.bytes++;
        if (*p == '\n') c.lines++;
        if (!isspace((unsigned char)*p)) {
            c.chars++;
            if (!in_word) { c.words++; in_word = 1; }
        } else {
            in_word = 0;
        }
    }
    if (*text && text[strlen(text) - 1] != '\n') c.lines++; /* last partial line counts */
    return c;
}

static WordCount *find_or_add(WordCount *out, int *count, int max_out, const char *word) {
    for (int i = 0; i < *count; i++) {
        if (strcmp(out[i].word, word) == 0) return &out[i];
    }
    if (*count >= max_out) return NULL;
    strncpy(out[*count].word, word, sizeof(out[*count].word) - 1);
    out[*count].word[sizeof(out[*count].word) - 1] = '\0';
    out[*count].count = 0;
    return &out[(*count)++];
}

int word_frequency(const char *text, WordCount *out, int max_out) {
    int count = 0;
    char buf[64];
    int len = 0;
    for (const char *p = text; ; p++) {
        if (*p && isalnum((unsigned char)*p)) {
            if (len < (int)sizeof(buf) - 1) buf[len++] = (char)tolower((unsigned char)*p);
        } else {
            if (len > 0) {
                buf[len] = '\0';
                WordCount *wc = find_or_add(out, &count, max_out, buf);
                if (wc) wc->count++;
                len = 0;
            }
            if (!*p) break;
        }
    }
    return count;
}

static int cmp_wordcount(const void *a, const void *b) {
    const WordCount *wa = a, *wb = b;
    if (wb->count != wa->count) return wb->count - wa->count;
    return strcmp(wa->word, wb->word);
}

void top_n(WordCount *freq, int freq_len, int n, WordCount *top) {
    WordCount *sorted = malloc(sizeof(WordCount) * (size_t)freq_len);
    memcpy(sorted, freq, sizeof(WordCount) * (size_t)freq_len);
    qsort(sorted, (size_t)freq_len, sizeof(WordCount), cmp_wordcount);
    int limit = n < freq_len ? n : freq_len;
    memcpy(top, sorted, sizeof(WordCount) * (size_t)limit);
    free(sorted);
}

#include "WordCounter.h"
#include <stdio.h>
#include <stdlib.h>

int main(int argc, char **argv) {
    if (argc != 2) { fprintf(stderr, "Usage: %s <file>\n", argv[0]); return 1; }
    FILE *f = fopen(argv[1], "rb");
    if (!f) { perror("fopen"); return 1; }
    fseek(f, 0, SEEK_END);
    long size = ftell(f);
    fseek(f, 0, SEEK_SET);
    char *text = malloc((size_t)size + 1);
    fread(text, 1, (size_t)size, f);
    text[size] = '\0';
    fclose(f);

    Counts c = count_all(text);
    printf("lines: %ld  words: %ld  chars: %ld  bytes: %ld\n", c.lines, c.words, c.chars, c.bytes);

    WordCount freq[1024];
    int n = word_frequency(text, freq, 1024);
    WordCount top[5];
    top_n(freq, n, 5, top);
    printf("Top words:\n");
    for (int i = 0; i < 5 && i < n; i++) printf("  %s: %d\n", top[i].word, top[i].count);

    free(text);
    return 0;
}
⚠ No in-browser playground here
C compiles to a real, native binary, so unlike the Python version of this project there is no editor above you can run in the browser. Copy the code below and run it on your own machine — it takes seconds once GCC or Clang is installed.
What each part does — in plain words
WordCount out[], int max_out — since C has no growable array or hash map in the language or standard library, word_frequency takes a fixed-capacity caller-provided array to fill in, and returns how many distinct words it actually found. This bounded-buffer pattern (caller owns the memory, callee reports how much it used) is extremely common in real C APIs, including much of the C standard library itself.

find_or_add does a linear search — with no hash map available, looking up whether a word is already in the table means scanning every existing entry. For a word-counting utility processing ordinary text files this is completely fine in practice; a production log-processing tool handling millions of distinct words would reach for a real hash table instead (see the course’s Standard Library lesson on what stdlib.h does and doesn’t provide).

qsort with a comparator — top_n uses the C standard library’s generic sort, which takes a function pointer telling it how to compare two elements. This is the same function-pointer mechanism the course’s Function Pointers lesson covers, applied to a genuinely useful job.
Common mistakes — and how to avoid them
✗ Forgetting to null-terminate the word buffer before comparing or storing it — strcmp and strncpy both rely on the terminator being there.
✓ word_frequency explicitly sets buf[len] = '\0' before using buf as a string.
✗ Reading the whole file with a fixed-size buffer that’s too small for a large file, silently truncating it.
✓ main.c uses fseek/ftell to find the real file size first, then allocates exactly that much plus one byte for the terminator.
✗ Forgetting to free() the buffer the file was read into.
✓ main.c frees it at the end — verified with Valgrind, which confirms 0 leaks.

4 Test & Prove Each Part

C has no built-in test framework and this sandbox can't reach a package registry for one, so these tests use plain assert() calls against the pure logic functions — no files touched during testing at all.

Line and word counts match a small, hand-checked example
Word frequency counting is case-insensitive ("The" and "the" are the same word)
The top-N ranking sorts by count, highest first
An empty string produces zero of everything that matters
Ctest_WordCounter.c
#include "WordCounter.h"
#include <assert.h>
#include <stdio.h>
#include <string.h>

#define RUN(name) do { name(); printf("PASS: %s\n", #name); } while (0)

static void counts_lines_words_and_chars_correctly(void) {
    Counts c = count_all("the cat sat\non the mat\n");
    assert(c.lines == 2);
    assert(c.words == 6);
}

static void frequency_is_case_insensitive(void) {
    WordCount out[16];
    int n = word_frequency("The the THE cat", out, 16);
    assert(n == 2);
    int found = 0;
    for (int i = 0; i < n; i++) {
        if (strcmp(out[i].word, "the") == 0) {
            assert(out[i].count == 3);
            found = 1;
        }
    }
    assert(found);
}

static void top_n_sorts_by_count_descending(void) {
    WordCount out[16];
    int n = word_frequency("a a a b b c", out, 16);
    WordCount top[2];
    top_n(out, n, 2, top);
    assert(strcmp(top[0].word, "a") == 0);
    assert(top[0].count == 3);
    assert(strcmp(top[1].word, "b") == 0);
    assert(top[1].count == 2);
}

static void empty_text_counts_zero_everything_that_matters(void) {
    Counts c = count_all("");
    assert(c.words == 0);
    WordCount out[4];
    int n = word_frequency("", out, 4);
    assert(n == 0);
}

int main(void) {
    RUN(counts_lines_words_and_chars_correctly);
    RUN(frequency_is_case_insensitive);
    RUN(top_n_sorts_by_count_descending);
    RUN(empty_text_counts_zero_everything_that_matters);
    printf("All tests passed.\n");
    return 0;
}

Compile and run with gcc -o test_run WordCounter.c test_WordCounter.c && ./test_run.

5 The Interface

Even a tiny program has an interface. Here is its contract, documented plainly.

INPUTArgumenta path to a text file
What it expects
./wordcount notes.txt
OUTPUTReportcounts plus the top 5 most frequent words
What it returns
lines: 12  words: 84  chars: 401  bytes: 412
Top words:
  the: 9
  a: 6
  ...

6 Run It & Automate It

Save the code as WordCounter.h / WordCounter.c / main.c and compile it with gcc — that turns your source directly into a native executable for your machine. No separate runtime needed: the compiled binary runs on its own.

Run it locally
gcc -o wordcount main.c WordCounter.c && ./wordcount notes.txt
Point it at any real text file you have lying around.

A CI tool like Jenkins runs the same compile-then-test-then-check-for-leaks steps automatically whenever the code changes — every line below has a plain explanation.

What you should see when it works
Terminala real run
$ ./wordcount notes.txt
lines: 12  words: 84  chars: 401  bytes: 412
Top words:
  the: 9
  a: 6
  to: 5
  and: 4
  is: 3
If it breaks — how to fix it
🚨 fopen: No such file or directory
The path is wrong or relative to the wrong directory. Try an absolute path to double-check.
🚨 The word counts look way too high or the table overflows.
The fixed-size WordCount freq[1024] array in main.c has a real capacity limit — a huge or highly repetitive file could exceed it. word_frequency simply stops adding new distinct words once max_out is reached; it never overflows the buffer, but it will silently under-report distinct words beyond that limit.
🚨 Segmentation fault on a very large file.
Check that malloc's return value isn't NULL before using it — a huge file could fail to allocate. This version doesn't check that, which is disclosed here as a real, simple omission worth fixing in production code.
GroovyJenkinsfile
// Jenkinsfile — compiles, tests, and checks for leaks on every change.
pipeline {
    agent any

    stages {
        stage('Get the code') {
            // download the latest code
            steps { checkout scm }
        }
        stage('Compile') {
            steps {
                // confirm a compiler is installed
                sh 'gcc --version'
                // compile with strict warnings on
                sh 'gcc -std=c17 -Wall -Wextra -o app *.c'
            }
        }
        stage('Run the tests') {
            steps {
                // prints PASS/FAIL, exits non-zero on failure
                sh './app'
            }
        }
        stage('Check for memory leaks') {
            steps {
                // fails the build on any leak or invalid access
                sh 'valgrind --error-exitcode=1 --leak-check=full ./app'
            }
        }
    }

    post {
        success { echo 'All tests passed, no leaks found.' }
        failure { echo 'A test or Valgrind check failed — see above.' }
    }
}
Try extending it
Replace the linear-search table with a real hash table for better performance on large files. Or add a --exclude flag that skips common stop words like "the" and "a" from the ranking.
What you learned
Bounded-buffer APIs as C's answer to not having growable collections; linear-search tables as a simple, honest tradeoff versus a real hash map; qsort with a comparator function pointer; and reading a whole file into memory safely with fseek/ftell.