typestar

word_freq.zig in Zig

A word-frequency table from tokenize, a hash map, a sort, and a bar chart.

// Count word frequencies in a passage and chart the leaders.
const std = @import("std");

const passage = "the rain in maine falls mainly on the plain and " ++
    "the rain stays on the plain";

const Entry = struct {
    word: []const u8,
    count: u32,
};

fn moreFirst(_: void, a: Entry, b: Entry) bool {
    return a.count > b.count;
}

pub fn main() !void {
    var gpa = std.heap.DebugAllocator(.{}){};
    defer _ = gpa.deinit();
    const alloc = gpa.allocator();

    // Tally every word into a map keyed by its slice of the passage.
    var counts = std.StringHashMap(u32).init(alloc);
    defer counts.deinit();
    var words = std.mem.tokenizeScalar(u8, passage, ' ');
    while (words.next()) |w| {
        const slot = try counts.getOrPut(w);
        if (!slot.found_existing) slot.value_ptr.* = 0;
        slot.value_ptr.* += 1;
    }

    // Pull the entries out and sort by count, biggest first.
    var ranked = std.ArrayList(Entry).empty;
    defer ranked.deinit(alloc);
    var it = counts.iterator();
    while (it.next()) |kv| {
        try ranked.append(alloc, .{
            .word = kv.key_ptr.*,
            .count = kv.value_ptr.*,
        });
    }
    std.mem.sort(Entry, ranked.items, {}, moreFirst);

    std.debug.print("{d} distinct words\n", .{ranked.items.len});
    std.debug.print("word    ct bar\n", .{});
    const bars = "####################";
    for (ranked.items) |e| {
        std.debug.print("{s:<7} {d:>2} {s}\n", .{
            e.word, e.count, bars[0..e.count],
        });
    }
}

How it works

  1. tokenizeScalar walks the passage; getOrPut tallies each word in one probe.
  2. The entries move into an ArrayList and std.mem.sort ranks them, biggest first.
  3. Slicing a bar of hashes to e.count draws each word's line in the chart.

Keywords and builtins used here

The run, in numbers

Lines
51
Characters to type
1340
Tokens
378
Three-star pace
65 tpm

At the three-star pace of 65 tokens a minute, this run takes about 349 seconds.

Type this snippet

Step 1 of 3 in Encore, step 25 of 27 in Language basics.

← Previous Next →