From ecb731b91fcda571c507044b316de25c9b52ffd7 Mon Sep 17 00:00:00 2001 From: Eric Willigers Date: Fri, 11 Sep 2026 13:09:38 +1000 Subject: [PATCH 1/2] rna-transcription: approaches --- CONTRIBUTING.md | 30 ++++++++ .../rna-transcription/.approaches/config.json | 39 +++++++++++ .../.approaches/introduction.md | 68 +++++++++++++++++++ .../.approaches/lookup-table/content.md | 34 ++++++++++ .../.approaches/lookup-table/snippet.txt | 8 +++ .../.approaches/simd/content.md | 68 +++++++++++++++++++ .../.approaches/simd/snippet.txt | 8 +++ .../.approaches/switch/content.md | 29 ++++++++ .../.approaches/switch/snippet.txt | 7 ++ 9 files changed, 291 insertions(+) create mode 100644 exercises/practice/rna-transcription/.approaches/config.json create mode 100644 exercises/practice/rna-transcription/.approaches/introduction.md create mode 100644 exercises/practice/rna-transcription/.approaches/lookup-table/content.md create mode 100644 exercises/practice/rna-transcription/.approaches/lookup-table/snippet.txt create mode 100644 exercises/practice/rna-transcription/.approaches/simd/content.md create mode 100644 exercises/practice/rna-transcription/.approaches/simd/snippet.txt create mode 100644 exercises/practice/rna-transcription/.approaches/switch/content.md create mode 100644 exercises/practice/rna-transcription/.approaches/switch/snippet.txt diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 7e4f604f..33ae3529 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -26,8 +26,38 @@ Exercises can be tested against the test runner using bin/verify-exercises-in-docker ``` +## Formatting and linting + +CI runs the following checks on every pull request, so please run them locally before submitting. + +All Zig code — exercise stubs, example solutions and test files — must be formatted with `zig fmt`: + +```bash +zig fmt --check . +``` + +The track configuration must pass configlet's linter, and the JSON files must be formatted with configlet's formatter: + +```bash +bin/configlet lint +bin/configlet fmt --update +``` + +When adding an exercise or an approach, generate any new UUIDs with configlet: + +```bash +bin/configlet uuid +``` + +Practice-exercise test files are generated from the canonical data in [`problem-specifications`][problem-specifications]; edit the generator in `generators/exercises/` and regenerate: + +```bash +bin/generate +``` + [guidelines]: https://exercism.org/blog/contribution-guidelines-nov-2023 [zig-forum]: https://forum.exercism.org/c/programming/zig/199 [building-tracks]: https://exercism.org/docs/building/tracks [style]: https://exercism.org/docs/building/markdown/style-guide [unnecessary-test-runs]: https://exercism.org/docs/building/tracks#h-avoiding-triggering-unnecessary-test-runs +[problem-specifications]: https://github.com/exercism/problem-specifications diff --git a/exercises/practice/rna-transcription/.approaches/config.json b/exercises/practice/rna-transcription/.approaches/config.json new file mode 100644 index 00000000..cb0a22fa --- /dev/null +++ b/exercises/practice/rna-transcription/.approaches/config.json @@ -0,0 +1,39 @@ +{ + "introduction": { + "authors": [ + "keiravillekode" + ] + }, + "approaches": [ + { + "uuid": "ff1e1b6a-5e9d-488f-b002-7cb4b7c25a82", + "slug": "switch", + "title": "Switch", + "blurb": "Map each nucleotide to its complement with a switch expression.", + "authors": [ + "massivelivefun" + ], + "contributors": [ + "keiravillekode" + ] + }, + { + "uuid": "b458705a-ded6-479a-bcb8-2a9bae34d499", + "slug": "lookup-table", + "title": "Lookup table", + "blurb": "Precompute a 256-byte table and index it by each nucleotide.", + "authors": [ + "keiravillekode" + ] + }, + { + "uuid": "34c0f10e-3d03-4b18-8fca-cd769d852618", + "slug": "simd", + "title": "SIMD", + "blurb": "Transcribe a whole block of nucleotides at once with vector operations.", + "authors": [ + "keiravillekode" + ] + } + ] +} diff --git a/exercises/practice/rna-transcription/.approaches/introduction.md b/exercises/practice/rna-transcription/.approaches/introduction.md new file mode 100644 index 00000000..5eae7bfa --- /dev/null +++ b/exercises/practice/rna-transcription/.approaches/introduction.md @@ -0,0 +1,68 @@ +# Introduction + +Transcription replaces each DNA nucleotide with its RNA complement: `G`↔`C` and `A`→`U`, `T`→`A`. The result is caller-owned memory. + +Since the output is the same length as the input and every nucleotide maps independently, every approach allocates a result buffer of `dna.len` bytes and fills it; they differ only in how a single nucleotide is mapped. + +## Approach: switch + +```zig +pub fn toRna(allocator: mem.Allocator, dna: []const u8) mem.Allocator.Error![]const u8 { + const rna = try allocator.alloc(u8, dna.len); + for (dna, rna) |nucleotide, *out| { + out.* = switch (nucleotide) { + 'A' => 'U', + 'C' => 'G', + 'G' => 'C', + 'T' => 'A', + else => unreachable, + }; + } + return rna; +} +``` + +A `switch` maps each nucleotide to its complement, one character at a time. +For details, see the [switch approach][approach-switch]. + +## Approach: lookup table + +```zig +const complement = blk: { + var table: [256]u8 = undefined; + table['A'] = 'U'; + table['C'] = 'G'; + table['G'] = 'C'; + table['T'] = 'A'; + break :blk table; +}; + +pub fn toRna(allocator: mem.Allocator, dna: []const u8) mem.Allocator.Error![]const u8 { + const rna = try allocator.alloc(u8, dna.len); + for (dna, rna) |nucleotide, *out| out.* = complement[nucleotide]; + return rna; +} +``` + +A 256-byte table computed at compile time turns each mapping into a single array index, with no branches in the loop. +For details, see the [lookup table approach][approach-lookup-table]. + +## Approach: SIMD + +```zig +fn transcribeBlock(block: Block) Block { + var rna: Block = block; + rna = @select(u8, block == @as(Block, @splat('A')), @as(Block, @splat('U')), rna); + rna = @select(u8, block == @as(Block, @splat('C')), @as(Block, @splat('G')), rna); + rna = @select(u8, block == @as(Block, @splat('G')), @as(Block, @splat('C')), rna); + rna = @select(u8, block == @as(Block, @splat('T')), @as(Block, @splat('A')), rna); + return rna; +} +``` + +A whole block of nucleotides is loaded into a `@Vector` and transcribed at once with vector compares and selects, processing many characters per iteration. +For details, see the [SIMD approach][approach-simd]. + +[approach-switch]: https://exercism.org/tracks/zig/exercises/rna-transcription/approaches/switch +[approach-lookup-table]: https://exercism.org/tracks/zig/exercises/rna-transcription/approaches/lookup-table +[approach-simd]: https://exercism.org/tracks/zig/exercises/rna-transcription/approaches/simd diff --git a/exercises/practice/rna-transcription/.approaches/lookup-table/content.md b/exercises/practice/rna-transcription/.approaches/lookup-table/content.md new file mode 100644 index 00000000..3b3db3b9 --- /dev/null +++ b/exercises/practice/rna-transcription/.approaches/lookup-table/content.md @@ -0,0 +1,34 @@ +# Lookup table + +```zig +const std = @import("std"); +const mem = std.mem; + +const complement = blk: { + var table: [256]u8 = undefined; + table['A'] = 'U'; + table['C'] = 'G'; + table['G'] = 'C'; + table['T'] = 'A'; + break :blk table; +}; + +pub fn toRna(allocator: mem.Allocator, dna: []const u8) mem.Allocator.Error![]const u8 { + const rna = try allocator.alloc(u8, dna.len); + for (dna, rna) |nucleotide, *out| { + out.* = complement[nucleotide]; + } + return rna; +} +``` + +Instead of deciding the complement with control flow, this approach precomputes it. +A 256-entry table — one slot per possible byte value — is built once at compile time, with the four DNA bases filled in. +The [labeled block][labeled-block] `blk: { ... break :blk table; }` runs at `comptime` because it initializes a `const`, so the table is baked into the binary as data rather than constructed at run time. + +The loop body is then a single array index, `complement[nucleotide]`, with no branches at all. +Iterating over `dna` and `rna` together binds `out` as a pointer into the result, so the transcribed byte is written straight through it. + +The table is deliberately `[256]u8` rather than something smaller: indexing by the raw byte value needs no range check or offset subtraction, and the untouched entries are simply never read for valid input. + +[labeled-block]: https://ziglang.org/documentation/master/#Blocks diff --git a/exercises/practice/rna-transcription/.approaches/lookup-table/snippet.txt b/exercises/practice/rna-transcription/.approaches/lookup-table/snippet.txt new file mode 100644 index 00000000..93e0ba38 --- /dev/null +++ b/exercises/practice/rna-transcription/.approaches/lookup-table/snippet.txt @@ -0,0 +1,8 @@ +const complement = blk: { + var table: [256]u8 = undefined; + table['A'] = 'U'; + table['C'] = 'G'; + table['G'] = 'C'; + table['T'] = 'A'; + break :blk table; +}; diff --git a/exercises/practice/rna-transcription/.approaches/simd/content.md b/exercises/practice/rna-transcription/.approaches/simd/content.md new file mode 100644 index 00000000..6f9ebebb --- /dev/null +++ b/exercises/practice/rna-transcription/.approaches/simd/content.md @@ -0,0 +1,68 @@ +# SIMD + +```zig +const std = @import("std"); +const mem = std.mem; + +const block_len = std.simd.suggestVectorLength(u8) orelse 8; +const Block = @Vector(block_len, u8); + +/// Transcribes a whole block of DNA nucleotides at once. +fn transcribeBlock(block: Block) Block { + // Every base is replaced; the seed only survives for bytes that are not + // a base, which valid DNA never contains. + var rna: Block = block; + rna = @select(u8, block == @as(Block, @splat('A')), @as(Block, @splat('U')), rna); + rna = @select(u8, block == @as(Block, @splat('C')), @as(Block, @splat('G')), rna); + rna = @select(u8, block == @as(Block, @splat('G')), @as(Block, @splat('C')), rna); + rna = @select(u8, block == @as(Block, @splat('T')), @as(Block, @splat('A')), rna); + return rna; +} + +fn transcribe(nucleotide: u8) u8 { + return switch (nucleotide) { + 'A' => 'U', + 'C' => 'G', + 'G' => 'C', + 'T' => 'A', + else => unreachable, + }; +} + +pub fn toRna(allocator: mem.Allocator, dna: []const u8) mem.Allocator.Error![]const u8 { + const rna = try allocator.alloc(u8, dna.len); + var i: usize = 0; + while (i + block_len <= dna.len) : (i += block_len) { + rna[i..][0..block_len].* = transcribeBlock(dna[i..][0..block_len].*); + } + // Transcribe the remaining nucleotides one at a time. + for (dna[i..], rna[i..]) |nucleotide, *out| { + out.* = transcribe(nucleotide); + } + return rna; +} +``` + +Zig's [`@Vector`][vectors] type provides portable SIMD: an operation on a vector is applied to all of its elements at once, compiling to the target's vector instructions where available. +Rather than transcribe one nucleotide per iteration, this approach loads a whole block of them into a vector and transcribes the block in one pass, following the pattern in [Everyone Should Know SIMD][mitchellh-simd]. + +[`std.simd.suggestVectorLength`][suggest-vector-length] picks a block length the target's vector registers can hold — for example 32 on x86-64 with AVX2 — and returns `null` on targets without SIMD, so a fallback of `8` is supplied. + +`transcribeBlock` handles all four bases the same way: one [`@select`][select] per base, each comparing the whole block against that base and, in the lanes that match, replacing the running result with its complement (`'A'`→`'U'`, `'C'`→`'G'`, `'G'`→`'C'`, `'T'`→`'A'`). +A lane matches at most one base, so the order of the selects does not matter. +The seed value is the input block itself; for valid DNA every lane matches exactly one base and is overwritten, so the seed only shows through for bytes that are not a base. + +The main loop copies `block_len` bytes out of `dna`, transcribes them, and stores the vector straight into the matching slice of `rna` — `dna[i..][0..block_len].*` dereferences a slice as a fixed-size array, which coerces to the vector, and the store on the left does the reverse. + +Any nucleotides past the last full block are handled by a scalar `switch`, reusing the same mapping as the [switch approach][approach-switch]. +Unlike the isogram-style problems, transcription has no notion of an "invalid" block, so the tail needs no special padding — just a short remainder loop. + +The loop body is branch-free and consumes many nucleotides at once, so this is the fastest approach on long sequences. +The trade-off is the most code and the need to reason about lane-wise selection and the scalar tail; for short input the simpler [switch][approach-switch] or [lookup table][approach-lookup-table] approaches are preferable. + +[vectors]: https://ziglang.org/documentation/master/#Vectors +[suggest-vector-length]: https://ziglang.org/documentation/master/std/#std.simd.suggestVectorLength +[select]: https://ziglang.org/documentation/master/#select +[mitchellh-simd]: https://mitchellh.com/writing/everyone-should-know-simd +[approach-switch]: https://exercism.org/tracks/zig/exercises/rna-transcription/approaches/switch +[approach-lookup-table]: https://exercism.org/tracks/zig/exercises/rna-transcription/approaches/lookup-table diff --git a/exercises/practice/rna-transcription/.approaches/simd/snippet.txt b/exercises/practice/rna-transcription/.approaches/simd/snippet.txt new file mode 100644 index 00000000..70a67ffc --- /dev/null +++ b/exercises/practice/rna-transcription/.approaches/simd/snippet.txt @@ -0,0 +1,8 @@ +fn transcribeBlock(block: Block) Block { + var rna: Block = block; + rna = @select(u8, block == @as(Block, @splat('A')), @as(Block, @splat('U')), rna); + rna = @select(u8, block == @as(Block, @splat('C')), @as(Block, @splat('G')), rna); + rna = @select(u8, block == @as(Block, @splat('G')), @as(Block, @splat('C')), rna); + rna = @select(u8, block == @as(Block, @splat('T')), @as(Block, @splat('A')), rna); + return rna; +} diff --git a/exercises/practice/rna-transcription/.approaches/switch/content.md b/exercises/practice/rna-transcription/.approaches/switch/content.md new file mode 100644 index 00000000..24300343 --- /dev/null +++ b/exercises/practice/rna-transcription/.approaches/switch/content.md @@ -0,0 +1,29 @@ +# Switch + +```zig +const std = @import("std"); +const mem = std.mem; + +pub fn toRna(allocator: mem.Allocator, dna: []const u8) mem.Allocator.Error![]const u8 { + var rna_slice = try allocator.alloc(u8, dna.len); + for (dna, 0..) |dna_nucleotide, i| { + switch (dna_nucleotide) { + 'A' => rna_slice[i] = 'U', + 'C' => rna_slice[i] = 'G', + 'G' => rna_slice[i] = 'C', + 'T' => rna_slice[i] = 'A', + else => unreachable, + } + } + return rna_slice; +} +``` + +A [`switch`][switch] maps each of the four DNA bases to its RNA complement. + +The `else => unreachable` prong states that no other byte can occur. +[`unreachable`][unreachable] is an assertion: in safe build modes reaching it panics, and in `ReleaseFast`/`ReleaseSmall` it is undefined behavior the optimizer may assume never happens. +The exercise's inputs only ever contain `A`, `C`, `G`, and `T`, so the prong is never taken. + +[switch]: https://ziglang.org/documentation/master/#switch +[unreachable]: https://ziglang.org/documentation/master/#unreachable diff --git a/exercises/practice/rna-transcription/.approaches/switch/snippet.txt b/exercises/practice/rna-transcription/.approaches/switch/snippet.txt new file mode 100644 index 00000000..9822856c --- /dev/null +++ b/exercises/practice/rna-transcription/.approaches/switch/snippet.txt @@ -0,0 +1,7 @@ +switch (dna_nucleotide) { + 'A' => rna_slice[i] = 'U', + 'C' => rna_slice[i] = 'G', + 'G' => rna_slice[i] = 'C', + 'T' => rna_slice[i] = 'A', + else => unreachable, +} From 9027196c82f85dd374def9843dd2b82c8a6f356a Mon Sep 17 00:00:00 2001 From: Eric Willigers Date: Fri, 11 Sep 2026 19:23:47 +1000 Subject: [PATCH 2/2] omit CONTRIBUTING.md changes --- CONTRIBUTING.md | 30 ------------------------------ 1 file changed, 30 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 33ae3529..7e4f604f 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -26,38 +26,8 @@ Exercises can be tested against the test runner using bin/verify-exercises-in-docker ``` -## Formatting and linting - -CI runs the following checks on every pull request, so please run them locally before submitting. - -All Zig code — exercise stubs, example solutions and test files — must be formatted with `zig fmt`: - -```bash -zig fmt --check . -``` - -The track configuration must pass configlet's linter, and the JSON files must be formatted with configlet's formatter: - -```bash -bin/configlet lint -bin/configlet fmt --update -``` - -When adding an exercise or an approach, generate any new UUIDs with configlet: - -```bash -bin/configlet uuid -``` - -Practice-exercise test files are generated from the canonical data in [`problem-specifications`][problem-specifications]; edit the generator in `generators/exercises/` and regenerate: - -```bash -bin/generate -``` - [guidelines]: https://exercism.org/blog/contribution-guidelines-nov-2023 [zig-forum]: https://forum.exercism.org/c/programming/zig/199 [building-tracks]: https://exercism.org/docs/building/tracks [style]: https://exercism.org/docs/building/markdown/style-guide [unnecessary-test-runs]: https://exercism.org/docs/building/tracks#h-avoiding-triggering-unnecessary-test-runs -[problem-specifications]: https://github.com/exercism/problem-specifications