1 The Problem
We want a log analyser: read a server log file, count how many entries are errors versus normal, find which hour had the most traffic, and list the most frequent error messages. It teaches parsing semi-structured text at scale and aggregating it into useful insight — a daily task in operations.
2 How to Think About It
Think about turning lines into counts, before any code:
fgets. → 2. Parse each line into its parts: hour, level (INFO/ERROR), message. → 3. Count as you go: errors, entries per hour, message frequencies, in a small hand-rolled table. → 4. Report the totals and the top items with qsort.
3 The Build — explained part by part
Here is the complete analyser. C’s standard library has no built-in equivalent of Python’s collections.Counter, no hash map, and (in the strict sense used here) no dependency-free regex engine either, so this project builds its counting and “top N” logic explicitly with a small linear-search table and a hand-written line parser — the same tradeoff word-counter made for its frequency table.
#ifndef LOG_ANALYSER_H
#define LOG_ANALYSER_H
#define MAX_LINE 512
#define TOP_N_ERRORS 3
typedef struct {
char hour[3]; /* "00".."23" */
char level[16];
char message[MAX_LINE];
} Entry;
typedef struct {
char key[MAX_LINE];
int count;
} Counted;
typedef struct {
int total;
int errors;
int has_busiest_hour;
char busiest_hour[3];
int busiest_hour_count;
Counted top_errors[TOP_N_ERRORS];
int top_errors_count;
} Report;
/* Parses one line shaped like "2026-06-24 14:30:00 ERROR Database timeout".
* Splits on whitespace but caps it at 4 fields so the message (which may
* itself contain spaces) stays whole as the last field. Lines that do not
* look like "date time LEVEL message" are rejected. Returns 1 on success. */
int parse_line(const char *line, Entry *out);
/* Counts occurrences of `key` in a hand-rolled counter table (no built-in
* hash map in C, the same tradeoff word-counter's frequency table makes),
* inserting a new entry if `key` has not been seen. */
void bump_count(Counted *table, int *table_count, int table_cap, const char *key);
/* Sorts a copy of `table` by count descending (ties broken alphabetically)
* and copies the top `n` into `out`. Returns how many were copied. */
int top_n(const Counted *table, int table_count, int n, Counted *out);
/* Runs parse_line over every line in `lines` and builds the full report:
* total valid entries, error count, the busiest hour, and the top 3 most
* common error messages. */
Report analyse(const char *const *lines, int line_count);
#endif
#include "LogAnalyser.h"
#include <string.h>
#include <stdlib.h>
#define MAX_HOURS 24
#define MAX_MESSAGES 4096
int parse_line(const char *line, Entry *out) {
const char *p1 = strchr(line, ' ');
if (!p1) return 0;
const char *p2 = strchr(p1 + 1, ' ');
if (!p2) return 0;
const char *p3 = strchr(p2 + 1, ' ');
if (!p3) return 0;
size_t date_len = (size_t)(p1 - line);
size_t time_len = (size_t)(p2 - (p1 + 1));
size_t level_len = (size_t)(p3 - (p2 + 1));
if (date_len >= 32 || time_len < 2 || time_len >= 32) return 0;
if (level_len >= sizeof(out->level)) return 0;
char date[32], time_s[32];
memcpy(date, line, date_len); date[date_len] = '\0';
memcpy(time_s, p1 + 1, time_len); time_s[time_len] = '\0';
if (!strchr(date, '-') || !strchr(time_s, ':')) return 0; /* not a "date time" pair */
const char *msg_start = p3 + 1;
size_t msg_len = strlen(msg_start);
while (msg_len > 0 && (msg_start[msg_len - 1] == '\n' || msg_start[msg_len - 1] == '\r')) msg_len--;
if (msg_len >= sizeof(out->message)) msg_len = sizeof(out->message) - 1;
out->hour[0] = time_s[0];
out->hour[1] = time_s[1];
out->hour[2] = '\0';
memcpy(out->level, p2 + 1, level_len);
out->level[level_len] = '\0';
memcpy(out->message, msg_start, msg_len);
out->message[msg_len] = '\0';
return 1;
}
void bump_count(Counted *table, int *table_count, int table_cap, const char *key) {
for (int i = 0; i < *table_count; i++) {
if (strcmp(table[i].key, key) == 0) { table[i].count++; return; }
}
if (*table_count < table_cap) {
strncpy(table[*table_count].key, key, sizeof(table[*table_count].key) - 1);
table[*table_count].key[sizeof(table[*table_count].key) - 1] = '\0';
table[*table_count].count = 1;
(*table_count)++;
}
}
static int cmp_count_desc(const void *a, const void *b) {
const Counted *ca = a, *cb = b;
if (cb->count != ca->count) return cb->count - ca->count;
return strcmp(ca->key, cb->key); /* ties broken alphabetically, same as word-counter */
}
int top_n(const Counted *table, int table_count, int n, Counted *out) {
Counted *sorted = malloc(sizeof(Counted) * (size_t)(table_count > 0 ? table_count : 1));
memcpy(sorted, table, sizeof(Counted) * (size_t)table_count);
qsort(sorted, (size_t)table_count, sizeof(Counted), cmp_count_desc);
int copied = table_count < n ? table_count : n;
memcpy(out, sorted, sizeof(Counted) * (size_t)copied);
free(sorted);
return copied;
}
Report analyse(const char *const *lines, int line_count) {
Report report = {0};
static Counted by_hour[MAX_HOURS];
static Counted by_message[MAX_MESSAGES];
int hour_count = 0, message_count = 0;
for (int i = 0; i < line_count; i++) {
Entry entry;
if (!parse_line(lines[i], &entry)) continue;
report.total++;
bump_count(by_hour, &hour_count, MAX_HOURS, entry.hour);
if (strcmp(entry.level, "ERROR") == 0) {
report.errors++;
bump_count(by_message, &message_count, MAX_MESSAGES, entry.message);
}
}
Counted busiest[1];
if (top_n(by_hour, hour_count, 1, busiest) == 1) {
report.has_busiest_hour = 1;
strncpy(report.busiest_hour, busiest[0].key, sizeof(report.busiest_hour) - 1);
report.busiest_hour[sizeof(report.busiest_hour) - 1] = '\0';
report.busiest_hour_count = busiest[0].count;
}
report.top_errors_count = top_n(by_message, message_count, TOP_N_ERRORS, report.top_errors);
return report;
}
#include "LogAnalyser.h"
#include <stdio.h>
#include <stdlib.h>
#define MAX_LINES 100000
int main(int argc, char **argv) {
const char *path = argc > 1 ? argv[1] : "server.log";
FILE *f = fopen(path, "r");
if (!f) {
fprintf(stderr, "Could not open %s\n", path);
return 1;
}
static char *lines[MAX_LINES];
static char buf[MAX_LINES][MAX_LINE];
int line_count = 0;
while (line_count < MAX_LINES && fgets(buf[line_count], MAX_LINE, f)) {
lines[line_count] = buf[line_count];
line_count++;
}
fclose(f);
Report report = analyse((const char *const *)lines, line_count);
printf("Total entries: %d\n", report.total);
printf("Errors: %d\n", report.errors);
if (report.has_busiest_hour) {
printf("Busiest hour: (\"%s\", %d)\n", report.busiest_hour, report.busiest_hour_count);
}
printf("Top errors: [");
for (int i = 0; i < report.top_errors_count; i++) {
printf("%s(\"%s\", %d)", i > 0 ? ", " : "", report.top_errors[i].key, report.top_errors[i].count);
}
printf("]\n");
return 0;
}
strchr calls, so the message (which may itself contain spaces, like “Database timeout”) stays whole as everything after the third space, instead of being split further the way a naive strtok on every space would split it.if (!strchr(date, '-') || !strchr(time_s, ':')) return 0; — a cheap sanity check that the line actually looks like a log line before trusting its shape further, since there is no regex validation involved.
Counted table[] used as a hand-rolled counter, with bump_count doing a linear search for an existing key before appending a new one — honest and simple for a bounded number of distinct hours/messages, but
O(n) per line where a real hash map would be O(1), disclosed plainly as the same tradeoff word-counter’s frequency table makes.int top_n(const Counted *table, int table_count, int n, Counted *out) — since the table has no defined order and C has no built-in “most common” function, this copies the counts and sorts them explicitly with
qsort and a comparator, exactly what Python’s Counter.most_common(n) does automatically — one small function, reused for both the busiest hour and the top errors.static Counted by_hour[MAX_HOURS]; static Counted by_message[MAX_MESSAGES]; inside
analyse — static local arrays keep a potentially large table (4096 entries here) out of the stack frame, the same reason main.c keeps its line buffer static too.
strtok calls on every space — a multi-word error message gets chopped into extra pieces and any field-count check breaks.parse_line does) and take everything after the third as the message, unsplit.static) local arrays — at 4096 entries each holding a 512-byte message, that is over 2MB on the stack, comfortably enough to crash the program with a stack overflow.static, moving them to the data segment instead of the stack.4 Test & Prove Each Part
We test parsing a log line and the aggregation, using a few known lines, plus an explicit check that ranking is stable.
#include "LogAnalyser.h"
#include <assert.h>
#include <stdio.h>
#include <string.h>
#define RUN(name) do { name(); printf("PASS: %s\n", #name); } while (0)
static void a_line_parses_into_hour_level_and_message(void) {
Entry e;
int ok = parse_line("2026-06-24 14:30:00 ERROR Database timeout", &e);
assert(ok);
assert(strcmp(e.hour, "14") == 0);
assert(strcmp(e.level, "ERROR") == 0);
assert(strcmp(e.message, "Database timeout") == 0);
}
static void errors_are_counted_correctly(void) {
const char *lines[] = {
"2026-06-24 14:30:00 INFO Server started",
"2026-06-24 14:31:00 ERROR Database timeout",
"2026-06-24 14:32:00 ERROR Disk full",
};
Report r = analyse(lines, 3);
assert(r.total == 3);
assert(r.errors == 2);
}
static void the_busiest_hour_is_identified(void) {
const char *lines[] = {
"2026-06-24 09:00:00 INFO a",
"2026-06-24 14:00:00 INFO b",
"2026-06-24 14:05:00 INFO c",
"2026-06-24 14:10:00 INFO d",
};
Report r = analyse(lines, 4);
assert(r.has_busiest_hour);
assert(strcmp(r.busiest_hour, "14") == 0);
assert(r.busiest_hour_count == 3);
}
static void a_malformed_line_is_skipped_not_a_crash(void) {
const char *lines[] = {
"not a real log line",
"2026-06-24 14:00:00 INFO fine",
};
Report r = analyse(lines, 2);
assert(r.total == 1);
}
static void top_errors_are_ranked_by_count_descending(void) {
const char *lines[] = {
"2026-06-24 09:00:00 ERROR Disk full",
"2026-06-24 09:01:00 ERROR Disk full",
"2026-06-24 09:02:00 ERROR Disk full",
"2026-06-24 09:03:00 ERROR Timeout",
"2026-06-24 09:04:00 ERROR Timeout",
"2026-06-24 09:05:00 ERROR Rare",
};
Report r = analyse(lines, 6);
assert(r.top_errors_count == 3);
assert(strcmp(r.top_errors[0].key, "Disk full") == 0 && r.top_errors[0].count == 3);
assert(strcmp(r.top_errors[1].key, "Timeout") == 0 && r.top_errors[1].count == 2);
}
static void an_empty_log_reports_all_zeros(void) {
Report r = analyse(NULL, 0);
assert(r.total == 0);
assert(r.errors == 0);
assert(!r.has_busiest_hour);
assert(r.top_errors_count == 0);
}
int main(void) {
RUN(a_line_parses_into_hour_level_and_message);
RUN(errors_are_counted_correctly);
RUN(the_busiest_hour_is_identified);
RUN(a_malformed_line_is_skipped_not_a_crash);
RUN(top_errors_are_ranked_by_count_descending);
RUN(an_empty_log_reports_all_zeros);
printf("All tests passed.\n");
return 0;
}
Compile and run with gcc -std=c17 -Wall -Wextra -Wpedantic -o test_run LogAnalyser.c test_LogAnalyser.c && ./test_run. We feed analyse a few known log lines so every count can be checked by hand. This is how you trust an analyser before running it on millions of real lines.
5 The Interface
What it expects
2026-06-24 14:15:44 ERROR Database timeoutWhat it returns
Total entries: 6
Errors: 3
Busiest hour: ("14", 4)
Top errors: [("Database timeout", 2), ...]6 Run It & Automate It
Save the code as LogAnalyser.h / LogAnalyser.c / main.c and compile it with gcc — that turns your source directly into a native executable for your machine. No separate runtime needed: the compiled binary runs on its own.
gcc -o analyse main.c LogAnalyser.c && ./analyse server.logPoint it at a
server.log file to get a full report; defaults to server.log in the current directory if no argument is given.A CI tool like Jenkins runs the same compile-then-test-then-check-for-leaks steps automatically whenever the code changes — every line below has a plain explanation.
$ ./analyse server.log
Total entries: 6
Errors: 3
Busiest hour: ("14", 4)
Top errors: [("Database timeout", 2), ("Disk full", 1)]server.log file in the same directory, or pass a path as the first argument.- and the time field a :, or parse_line rejects the whole line.// Jenkinsfile — compiles, tests, and checks for leaks on every change.
pipeline {
agent any
stages {
stage('Get the code') {
// download the latest code
steps { checkout scm }
}
stage('Compile') {
steps {
// confirm a compiler is installed
sh 'gcc --version'
// compile with strict warnings on
sh 'gcc -std=c17 -Wall -Wextra -o app *.c'
}
}
stage('Run the tests') {
steps {
// prints PASS/FAIL, exits non-zero on failure
sh './app'
}
}
stage('Check for memory leaks') {
steps {
// fails the build on any leak or invalid access
sh 'valgrind --error-exitcode=1 --leak-check=full ./app'
}
}
}
post {
success { echo 'All tests passed, no leaks found.' }
failure { echo 'A test or Valgrind check failed — see above.' }
}
}
- Date filtering. Only include entries from one specific day. (Teaches: string comparison on the date field.)
- Use POSIX
regex.h. Replace the hand-written parser with a real regular expression viaregcomp/regexec— unlike Rust or Go, C’s own C library genuinely ships one. (Teaches: what a standard-library dependency buys you over hand-rolled parsing, and when it is worth reaching for.) - Live tail. Keep reading a log file as new lines arrive, like
tail -f. (Teaches: following a growing file withsleepand re-reading, from<unistd.h>.)
qsort-based “most common” helper, since C has no built-in Counter or hash map either. Turning raw logs into insight is a vital operations skill in any language. Related: Structs and Arrays, File I/O.