From 803360f1ff300fc3ef60aa3316875a1a7d642afd Mon Sep 17 00:00:00 2001 From: MohammadRokib Date: Fri, 10 Jul 2026 22:58:04 +0600 Subject: [PATCH 1/4] refactor: extract Options and Counter classes - Extracted counting logic into a dedicated Counter class and argument parsing into an Options class. - The Counter class now handles all metrics counting in a single pass, while Options handles flag parsing with support for multiple concurrent flags. - All metrics are enabled by default, when no flags are specified. - Added header documentation on each method Signed-off-by: MohammadRokib --- src/ccwc/Counter.java | 71 +++++++++++++++++++++++++ src/ccwc/Main.java | 120 +++++++++--------------------------------- src/ccwc/Options.java | 55 +++++++++++++++++++ 3 files changed, 152 insertions(+), 94 deletions(-) create mode 100644 src/ccwc/Counter.java create mode 100644 src/ccwc/Options.java diff --git a/src/ccwc/Counter.java b/src/ccwc/Counter.java new file mode 100644 index 0000000..fe9e343 --- /dev/null +++ b/src/ccwc/Counter.java @@ -0,0 +1,71 @@ +package ccwc; + +import java.io.BufferedReader; +import java.io.IOException; +import java.io.InputStreamReader; +import java.nio.charset.StandardCharsets; +import java.nio.file.Files; +import java.nio.file.Path; + +/** + * Accumulates byte, line, word, and character counts for a file. + * A single pass is made over the file and all requested counts are + * collected concurrently. + */ +public class Counter { + /** Number of lines counted in the file. */ + public long lines = 0; + /** Number of words counted in the file. */ + public long words = 0; + /** Number of bytes counted in the file. */ + public long bytes = 0; + /** Number of characters counted in the file. */ + public long chars = 0; + + /** + * Counts the specified metrics from the given file. Which metrics + * are collected is controlled by the flags set on {@code opts}. + * The byte count is obtained directly from the file system; the + * remaining metrics are gathered by streaming UTF-8 decoded + * characters through a {@code BufferedReader}. + * + * @param path the path to the file to count + * @param opts specifies which counters to enable + * @throws IOException if an I/O error occurs + */ + public void count(Path path, Options opts) throws IOException { + lines = words = bytes = chars = 0; + + if (opts.countBytes) { + bytes = Files.size(path); + } + + boolean needCharStream = opts.countLines || opts.countWords || opts.countChars; + if (needCharStream) { + try (BufferedReader reader = new BufferedReader( + new InputStreamReader(Files.newInputStream(path), StandardCharsets.UTF_8))) { + + boolean inWord = false; + int ch; + while((ch = reader.read())!= -1) { + if (opts.countChars) { + chars++; + } + + if (opts.countLines && ch == '\n') { + lines++; + } + + if (opts.countWords) { + if (Character.isWhitespace(ch)) { + inWord = false; + } else if (!inWord) { + words++; + inWord = true; + } + } + } + } + } + } +} diff --git a/src/ccwc/Main.java b/src/ccwc/Main.java index 8ffb773..fdf6bfc 100644 --- a/src/ccwc/Main.java +++ b/src/ccwc/Main.java @@ -10,109 +10,41 @@ import java.nio.file.Paths; public class Main { - public static void main(String[] args) throws IOException { - if (args.length < 2) { - System.err.println("Usage: ccwc (-c | -l) "); - System.exit(1); - } - - String flag = args[0]; - String fileName = args[1]; - Path path = Paths.get(fileName); - - switch (flag) { - case "-c": - System.out.println(countBytes(path) + " " + fileName); - break; - case "-l": - System.out.println(countLines(path) + " " + fileName); - break; - case "-w": - System.out.println(countWords(path) + " " + fileName); - break; - case "-m": - System.out.println(countChars(path) + " " + fileName); - break; - default: - System.err.println("Unknown flag: " + flag); - System.exit(1); - } - } /** - * Counts the number of bytes in a file by streaming it through a buffer, - * without ever loading the whole file into memory - * */ - private static long countBytes(Path path) throws IOException { - long count = 0; - - try (InputStream in = Files.newInputStream(path)) { - byte[] buffer = new byte[8192]; - int bytesRead; + * Entry point for the ccwc word count utility. Parses command-line arguments + * and delegates to the {@link Counter} class to perform the requested counts. + * + * @param args command-line arguments, expected to contain a flag and a filename + * @throws IOException if an I/O error occurs reading the file + */ + public static void main(String[] args) throws IOException { + Options opts = Options.parse(args); - while((bytesRead = in.read(buffer)) != -1) { - count += bytesRead; - } + if (opts.fileName == null) { + System.err.println("Usage: ccwc [-c] [-l] [-w] [-m] "); + System.exit(1); } - return count; - } - /** - * Counts newline characters by streaming decoded characters through a BufferedReader - * */ - private static long countLines(Path path) throws IOException { - long count = 0; + Path path = Paths.get(opts.fileName); + Counter counter = new Counter(); + counter.count(path, opts); - try (BufferedReader reader = new BufferedReader( - new InputStreamReader(Files.newInputStream(path), StandardCharsets.UTF_8))) { - int ch; - while ((ch = reader.read()) != -1) { - if (ch == '\n') { - count++; - } + if (opts.countBytes && opts.countLines && opts.countWords && !opts.countChars) { + System.out.printf("%8d %8d %8d %s\n", counter.lines, counter.words, counter.bytes, opts.fileName); + } else { + if (opts.countBytes) { + System.out.println(counter.bytes + " " + opts.fileName); } - } - return count; - } - - /** - * Counts words by streaming decoded characters and detecting transitions - * from whitespace to non-whitespace. - * */ - private static long countWords(Path path) throws IOException { - long count = 0; - boolean inWord = false; - - try (BufferedReader reader = new BufferedReader( - new InputStreamReader(Files.newInputStream(path), StandardCharsets.UTF_8))) { - int ch; - while ((ch = reader.read()) != -1) { - if (Character.isWhitespace(ch)) { - inWord = false; - } else if (!inWord) { - count++; - inWord = true; - } + if (opts.countLines) { + System.out.println(counter.lines + " " + opts.fileName); } - } - return count; - } - - /** - * Counts characters by streaming decoded UTF-8 characters. The InputStreamReader's - * internal decoder handles multibyte sequences, so each read() returns one logical - * character regardless of how many bytes it occupies on disk. - * */ - private static long countChars(Path path) throws IOException { - long count = 0; - - try (BufferedReader reader = new BufferedReader( - new InputStreamReader(Files.newInputStream(path), StandardCharsets.UTF_8))) { - int ch; - while((ch = reader.read()) != -1) { - count++; + if (opts.countWords) { + System.out.println(counter.words + " " + opts.fileName); + } + if (opts.countChars) { + System.out.println(counter.chars + " " + opts.fileName); } } - return count; } } diff --git a/src/ccwc/Options.java b/src/ccwc/Options.java new file mode 100644 index 0000000..4449ddf --- /dev/null +++ b/src/ccwc/Options.java @@ -0,0 +1,55 @@ +package ccwc; + +/** + * Parses and stores command-line flag options for the ccwc utility. + * If no flag is specified, all metrics (bytes, lines, words, chars) + * are enabled by default. + */ +public class Options { + /** Whether to count bytes ({@code -c}). */ + public boolean countBytes = false; + /** Whether to count lines ({@code -l}). */ + public boolean countLines = false; + /** Whether to count words ({@code -w}). */ + public boolean countWords = false; + /** Whether to count characters ({@code -m}). */ + public boolean countChars = false; + /** The filename argument, or {@code null} if not provided. */ + public String fileName = null; + + /** + * Parses the command-line arguments and returns an {@code Options} + * instance with the appropriate flags set. + *

+ * Recognized flags are {@code -c}, {@code -l}, {@code -w}, and + * {@code -m}. The first non-flag argument is treated as the filename. + * If no flags are specified, all four metrics are enabled. + * + * @param args the command-line arguments to parse + * @return an {@code Options} instance reflecting the parsed flags + */ + public static Options parse(String[] args) { + Options opts = new Options(); + for (String arg : args) { + switch (arg) { + case "-c": opts.countBytes = true; break; + case "-l": opts.countLines = true; break; + case "-w": opts.countWords = true; break; + case "-m": opts.countChars = true; break; + default: + if (opts.fileName == null) { + opts.fileName = arg; + } + break; + } + } + + boolean anyFlagSet = opts.countBytes || opts.countLines || opts.countWords || opts.countChars; + if (!anyFlagSet) { + opts.countBytes = true; + opts.countLines = true; + opts.countWords = true; + } + return opts; + } +} From 0826ec42d2717a6bc61b5537bff2fd7187d8c828 Mon Sep 17 00:00:00 2001 From: MohammadRokib Date: Sat, 11 Jul 2026 20:25:46 +0600 Subject: [PATCH 2/4] feat: support reading from standard input - Implement support for reading from standard input when no filename is provided. - Refactor Counter to use a new CountingInputStream wrapper that tracks bytes read, enabling all four metrics (bytes, lines, words, chars) to be counted in a single pass regardless of input source. - Fix null filename output in printResults when reading from stdin Signed-off-by: MohammadRokib --- src/ccwc/Counter.java | 88 ++++++++++++++++++++++++------- src/ccwc/CountingInputStream.java | 71 +++++++++++++++++++++++++ src/ccwc/Main.java | 42 +++++++++------ 3 files changed, 168 insertions(+), 33 deletions(-) create mode 100644 src/ccwc/CountingInputStream.java diff --git a/src/ccwc/Counter.java b/src/ccwc/Counter.java index fe9e343..dff2350 100644 --- a/src/ccwc/Counter.java +++ b/src/ccwc/Counter.java @@ -2,6 +2,7 @@ import java.io.BufferedReader; import java.io.IOException; +import java.io.InputStream; import java.io.InputStreamReader; import java.nio.charset.StandardCharsets; import java.nio.file.Files; @@ -42,27 +43,78 @@ public void count(Path path, Options opts) throws IOException { boolean needCharStream = opts.countLines || opts.countWords || opts.countChars; if (needCharStream) { - try (BufferedReader reader = new BufferedReader( - new InputStreamReader(Files.newInputStream(path), StandardCharsets.UTF_8))) { + try (InputStream in = Files.newInputStream(path)) { + countFromStream(in, opts); + } + } + } - boolean inWord = false; - int ch; - while((ch = reader.read())!= -1) { - if (opts.countChars) { - chars++; - } + /** + * Counts the specified metrics from the given input stream. Which + * metrics are collected is controlled by the flags set on {@code opts}. + *

+ * When only byte counting is requested the stream is read directly + * in a raw byte loop. Otherwise, the stream is wrapped in a + * {@link CountingInputStream} so that line, word, and character + * counting can share the single read pass while bytes are still + * accumulated. + * + * @param in the input stream to read from + * @param opts specifies which counters to enable + * @throws IOException if an I/O error occurs + */ + public void count(InputStream in, Options opts) throws IOException { + lines = words = chars = bytes = 0; - if (opts.countLines && ch == '\n') { - lines++; - } + if (opts.countBytes && !opts.countLines && !opts.countWords && !opts.countChars) { + byte[] buffer = new byte[8192]; + int bytesRead; + while((bytesRead = in.read(buffer)) != -1) { + bytes += bytesRead; + } + return; + } + + CountingInputStream countingIn = new CountingInputStream(in); + countFromStream(countingIn, opts); + + if (opts.countBytes) { + bytes = countingIn.getBytesRead(); + } + } + + /** + * Reads UTF-8 decoded characters from the given input stream and + * increments line, word, and character counters as requested by + * {@code opts}. If a {@link CountingInputStream} is passed, its + * byte counter is also accumulated into {@link #bytes}. + * + * @param in the input stream to read from (may be a + * {@code CountingInputStream} for byte tracking) + * @param opts specifies which counters to enable + * @throws IOException if an I/O error occurs + */ + private void countFromStream(InputStream in, Options opts) throws IOException { + try (BufferedReader reader = new BufferedReader( + new InputStreamReader(in, StandardCharsets.UTF_8))) { + + boolean inWord = false; + int ch; + while((ch = reader.read())!= -1) { + if (opts.countChars) { + chars++; + } + + if (opts.countLines && ch == '\n') { + lines++; + } - if (opts.countWords) { - if (Character.isWhitespace(ch)) { - inWord = false; - } else if (!inWord) { - words++; - inWord = true; - } + if (opts.countWords) { + if (Character.isWhitespace(ch)) { + inWord = false; + } else if (!inWord) { + words++; + inWord = true; } } } diff --git a/src/ccwc/CountingInputStream.java b/src/ccwc/CountingInputStream.java new file mode 100644 index 0000000..6a210cf --- /dev/null +++ b/src/ccwc/CountingInputStream.java @@ -0,0 +1,71 @@ +package ccwc; + +import java.io.FilterInputStream; +import java.io.IOException; +import java.io.InputStream; + +/** + * An {@link java.io.InputStream} wrapper that tracks the total number of bytes + * read from the underlying stream. Useful for counting bytes without an + * additional pass over the data. + */ +public class CountingInputStream extends FilterInputStream { + private long bytesRead = 0; + + /** + * Constructs a {@code CountingInputStream} wrapping the given input stream. + * + * @param in the input stream to wrap + */ + CountingInputStream(InputStream in) { + super(in); + } + + /** + * Reads a single byte from the underlying stream and increments the + * byte counter if a byte was successfully read. + * + * @return the next byte of data, or {@code -1} if the end of the stream + * has been reached + * @throws IOException if an I/O error occurs + */ + @Override + public int read() throws IOException { + int b = super.read(); + if (b != -1) { + bytesRead++; + } + return b; + } + + /** + * Reads up to {@code len} bytes from the underlying stream into the + * given buffer and increments the byte counter by the number of bytes + * actually read. + * + * @param b the buffer into which the data is read + * @param off the start offset in the buffer at which the data is written + * @param len the maximum number of bytes to read + * @return the total number of bytes read into the buffer, or {@code -1} + * if there is no more data + * @throws IOException if an I/O error occurs + */ + @Override + public int read(byte[] b, int off, int len) throws IOException { + int n = super.read(b, off, len); + if (n != -1) { + bytesRead += n; + } + return n; + } + + /** + * Returns the total number of bytes that have been read from the + * underlying stream since this object was created. + * + * @return the total bytes read + */ + long getBytesRead() { + return bytesRead; + } +} diff --git a/src/ccwc/Main.java b/src/ccwc/Main.java index fdf6bfc..ee0c693 100644 --- a/src/ccwc/Main.java +++ b/src/ccwc/Main.java @@ -1,11 +1,6 @@ package ccwc; -import java.io.BufferedReader; import java.io.IOException; -import java.io.InputStream; -import java.io.InputStreamReader; -import java.nio.charset.StandardCharsets; -import java.nio.file.Files; import java.nio.file.Path; import java.nio.file.Paths; @@ -20,30 +15,47 @@ public class Main { */ public static void main(String[] args) throws IOException { Options opts = Options.parse(args); + Counter counter = new Counter(); if (opts.fileName == null) { - System.err.println("Usage: ccwc [-c] [-l] [-w] [-m] "); - System.exit(1); + counter.count(System.in, opts); + printResults(counter, opts, null); + } else { + Path path = Paths.get(opts.fileName); + counter.count(path, opts); + printResults(counter, opts, opts.fileName); } + } - Path path = Paths.get(opts.fileName); - Counter counter = new Counter(); - counter.count(path, opts); + /** + * Prints the results from the counter to standard output. When all of + * {@code -c}, {@code -l}, {@code -w} are active and {@code -m} is not, + * the output is formatted as {@code "%8d %8d %8d [filename]"} matching + * the classic {@code wc} default format. Otherwise, each enabled metric + * is printed on its own line. + * + * @param counter the counter whose results to print + * @param opts the options indicating which metrics are enabled + * @param fileName the file name to include in the output, or {@code null} + * when reading from standard input + */ + private static void printResults(Counter counter, Options opts, String fileName) { + String suffix = fileName == null ? "" : fileName; if (opts.countBytes && opts.countLines && opts.countWords && !opts.countChars) { - System.out.printf("%8d %8d %8d %s\n", counter.lines, counter.words, counter.bytes, opts.fileName); + System.out.printf("%8d %8d %8d %s\n", counter.lines, counter.words, counter.bytes, suffix); } else { if (opts.countBytes) { - System.out.println(counter.bytes + " " + opts.fileName); + System.out.println(counter.bytes + " " + suffix); } if (opts.countLines) { - System.out.println(counter.lines + " " + opts.fileName); + System.out.println(counter.lines + " " + suffix); } if (opts.countWords) { - System.out.println(counter.words + " " + opts.fileName); + System.out.println(counter.words + " " + suffix); } if (opts.countChars) { - System.out.println(counter.chars + " " + opts.fileName); + System.out.println(counter.chars + " " + suffix); } } } From ce9b8f0a21d444570c6ff34a1171937d656a8d4e Mon Sep 17 00:00:00 2001 From: MohammadRokib Date: Sun, 12 Jul 2026 12:07:14 +0600 Subject: [PATCH 3/4] docs: add README Adds a detailed README.md covering features, installation, usage examples, architecture, and environment notes. Includes guides for compiling with javac, running the tool with various flags, and notes for PowerShell users on Windows. Signed-off-by: MohammadRokib --- README.md | 165 ++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 165 insertions(+) create mode 100644 README.md diff --git a/README.md b/README.md new file mode 100644 index 0000000..b6db668 --- /dev/null +++ b/README.md @@ -0,0 +1,165 @@ + + +

+ +# ccwc - A Java Implementation of the Unix `wc` Tool + +[![Java](https://img.shields.io/badge/Java-17+-ED8B00?style=for-the-badge&logo=java&logoColor=white)](https://www.oracle.com/java/) + +`ccwc` (Coding Challenges Word Count) is a custom implementation of the classic Unix `wc` (word count) command-line utility, written entirely in Java. It is built to be memory-efficient, scalable for massive files, and fully compatible with standard Unix pipelines. + +This project was built as part of the [Build Your Own wc Tool Challenge](https://codingchallenges.fyi/challenges/challenge-wc). + +
+ +

+ Explore the docs · + Report Bug · + LinkedIn · + Email +

+ +--- + +## Features + +- **`-c`**: Count bytes in a file or stream. +- **`-l`**: Count lines (newline characters). +- **`-w`**: Count words (sequences of characters delimited by whitespace). +- **`-m`**: Count characters (correctly handles multi-byte UTF-8 encoded text). +
+ +- **Default Mode**: Output lines, words, and bytes simultaneously when no flag is provided. +- **Standard Input (stdin)**: Supports Unix piping (e.g., `cat file.txt | ccwc -l`). +- **Memory Safe**: Uses a streaming single-pass architecture. It can process files larger than available RAM without crashing. + +

Back to top ⬆️

+ +--- + +## Prerequisites + +- **Java Development Kit (JDK) 17** or higher. +- A terminal/command prompt environment. + +

Back to top ⬆️

+ +--- + +## Installation & Building + +Because this project uses standard Java libraries with no external dependencies, you can compile it directly using `javac`. + +1. Clone the repository: + ```bash + git clone https://github.com/MohammadRokib/wc-tool-java.git + cd ccwc + ``` + +2. Compile the Java source files into an `out` directory: + ```bash + # On Linux / macOS / Git Bash / Windows CMD + javac -d out src/ccwc/*.java + ``` + +

Back to top ⬆️

+ +--- + +## Usage + +The application is run via the `java` command, pointing to the `out` directory as the classpath. + +### Syntax +```bash +java -cp out ccwc.Main [-c] [-l] [-w] [-m] [filename] +``` + +*If no `filename` is provided, the tool automatically reads from standard input (`stdin`).* + +### Examples + +**1. Count bytes in a file:** +```bash +$ java -cp out ccwc.Main -c test.txt +342190 test.txt +``` + +**2. Count lines in a file:** +```bash +$ java -cp out ccwc.Main -l test.txt +7145 test.txt +``` + +**3. Count words in a file:** +```bash +$ java -cp out ccwc.Main -w test.txt +58164 test.txt +``` + +**4. Count characters in a file (UTF-8 aware):** +```bash +$ java -cp out ccwc.Main -m test.txt +339292 test.txt +``` + +**5. Default mode (lines, words, bytes):** +```bash +$ java -cp out ccwc.Main test.txt + 7145 58164 342190 test.txt +``` + +**6. Reading from Standard Input (Piping):** +When reading from `stdin`, the filename is omitted from the output. +```bash +$ cat test.txt | java -cp out ccwc.Main -l +7145 +``` + +

Back to top ⬆️

+ +--- + +## Architecture & Design + +Instead of reading entire files into memory (which causes `OutOfMemoryError` on large files), `ccwc` uses a **single-pass, streaming architecture**. + +The project is divided into four main components: + +1. **`Main.java`**: The entry point. Delegates argument parsing to `Options`, invokes the `Counter`, and formats the standard output using `System.out.printf`. +2. **`Options.java`**: A Data Transfer Object (DTO) that parses command-line arguments into boolean flags. If no flags are provided, it automatically enables the default metrics (lines, words, bytes). +3. **`Counter.java`**: The core engine. It features two entry points: + - `count(Path path, Options)`: For file inputs. Uses `Files.size()` for an instant O(1) byte count, avoiding unnecessary disk reads. + - `count(InputStream in, Options)`: For standard input. Uses a shared `countFromStream` method that loops through decoded characters exactly once, checking for line breaks, word boundaries, and character counts simultaneously. +4. **`CountingInputStream.java`**: Extends `FilterInputStream` (Decorator Pattern). When reading from `stdin`, this class sits at the bottom of the stream stack, intercepting raw bytes to tally the total byte count as they flow up to the character decoder. + +

Back to top ⬆️

+ +--- + +## Environment Notes (PowerShell Users) + +If you are testing the standard input byte count (`-c`) on Windows using **PowerShell**, you may notice a 3-byte discrepancy compared to reading the file directly (e.g., `342187` instead of `342190`). + +**Why?** PowerShell's `cat` alias (`Get-Content`) decodes files into .NET strings and silently strips the 3-byte UTF-8 Byte Order Mark (BOM) before piping the data to external executables like `java.exe`. + +Your Java code is correct. To verify raw byte piping on Windows, use **Command Prompt (`cmd.exe`)** with the `type` command, or use **Git Bash**: +```cmd +:: In cmd.exe +type test.txt | java -cp out ccwc.Main -c +``` + +

Back to top ⬆️

+ +--- + +## Contact + +Mohammad Rokib + +- **[LinkedIn](https://www.linkedin.com/in/m0hammadrokib/)** +- **[Email](mailto:mohammadrokibkhan@gmail.com)** +- **[GitHub](https://github.com/MohammadRokib)** +- **[Project Link: wc-tool-java](https://github.com/MohammadRokib/wc-tool-java)** + +

Back to top ⬆️

From 8d4ae38ee0055c2b720f3f21a73c4d5b36a98132 Mon Sep 17 00:00:00 2001 From: MohammadRokib Date: Wed, 15 Jul 2026 14:30:29 +0600 Subject: [PATCH 4/4] ci: add GitHub Actions smoke-test workflow --- .github/scripts/smoke-test.sh | 44 ++++++++++++++ .github/workflows/ci.yml | 30 ++++++++++ AGENTS.md | 106 ++++++++++++++++++++++++++++++++++ CLAUDE.md | 66 +++++++++++++++++++++ README.md | 1 + 5 files changed, 247 insertions(+) create mode 100644 .github/scripts/smoke-test.sh create mode 100644 .github/workflows/ci.yml create mode 100644 AGENTS.md create mode 100644 CLAUDE.md diff --git a/.github/scripts/smoke-test.sh b/.github/scripts/smoke-test.sh new file mode 100644 index 0000000..a10fc65 --- /dev/null +++ b/.github/scripts/smoke-test.sh @@ -0,0 +1,44 @@ +#!/usr/bin/env bash +# Smoke test for ccwc. Compiles must already have produced ./out (javac -d out src/ccwc/*.java). +# Verifies every distinct code path in Counter against the known-good values for test.txt +# documented in AGENTS.md / README.md: 7145 lines, 58164 words, 342190 bytes, 339292 chars. +# +# Runnable locally from the repo root: bash .github/scripts/smoke-test.sh +set -euo pipefail +cd "$(dirname "$0")/../.." + +fail=0 +check() { # desc expected actual + if [ "$3" = "$2" ]; then + echo "OK $1" + else + echo "FAIL $1 -- expected [$2], got [$3]" + fail=1 + fi +} + +# --- File input, single flag: exercises the Files.size() -c shortcut (Counter.java:41) --- +check "-c file" 342190 "$(java -cp out ccwc.Main -c test.txt | awk '{print $1}')" +check "-l file" 7145 "$(java -cp out ccwc.Main -l test.txt | awk '{print $1}')" +check "-w file" 58164 "$(java -cp out ccwc.Main -w test.txt | awk '{print $1}')" +check "-m file" 339292 "$(java -cp out ccwc.Main -m test.txt | awk '{print $1}')" + +# --- File input, default combo: the columnar printf branch (Main.java:45-46) --- +check "default file" "7145 58164 342190" "$(java -cp out ccwc.Main test.txt | awk '{print $1, $2, $3}')" + +# --- Stdin input, single flag: -c alone exercises the raw 8KB-loop shortcut (Counter.java:69-76) --- +check "-c stdin" 342190 "$(cat test.txt | java -cp out ccwc.Main -c | awk '{print $1}')" +check "-l stdin" 7145 "$(cat test.txt | java -cp out ccwc.Main -l | awk '{print $1}')" +check "-w stdin" 58164 "$(cat test.txt | java -cp out ccwc.Main -w | awk '{print $1}')" +check "-m stdin" 339292 "$(cat test.txt | java -cp out ccwc.Main -m | awk '{print $1}')" + +# --- Stdin input, default combo --- +check "default stdin" "7145 58164 342190" "$(cat test.txt | java -cp out ccwc.Main | awk '{print $1, $2, $3}')" + +# --- Stdin, -c combined with another flag: the ONLY path that exercises the +# CountingInputStream decorator (Counter.java:78-79) instead of either shortcut --- +result="$(cat test.txt | java -cp out ccwc.Main -c -l)" +check "-c -l stdin bytes (decorator path)" 342190 "$(sed -n 1p <<< "$result" | awk '{print $1}')" +check "-c -l stdin lines (decorator path)" 7145 "$(sed -n 2p <<< "$result" | awk '{print $1}')" + +exit $fail diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..ca46538 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,30 @@ +name: CI + +on: + push: + branches: [master] + pull_request: + +jobs: + build-and-verify: + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, windows-latest] + java: ['17', '24'] + runs-on: ${{ matrix.os }} + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-java@v4 + with: + distribution: temurin + java-version: ${{ matrix.java }} + + - name: Compile + shell: bash + run: javac -d out src/ccwc/*.java + + - name: Smoke test + shell: bash + run: bash .github/scripts/smoke-test.sh diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..643c655 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,106 @@ +# Repository Guidelines + +## Project Overview +`ccwc` ("Coding Challenges Word Count") is a Java clone of the Unix `wc` command, built for the [Coding Challenges "Build Your Own wc Tool"](https://codingchallenges.fyi/challenges/challenge-wc) exercise. It counts bytes (`-c`), lines (`-l`), words (`-w`), and characters (`-m`) in a file or from stdin. No flags defaults to `-c -l -w` (bytes+lines+words), matching real `wc`. Core design goal: memory-safe single-pass streaming so it scales to files larger than available RAM, rather than loading input into memory. + +## Architecture & Data Flow +Four classes, all in package `ccwc`, all under `src/ccwc/`. Pure JDK — no third-party libraries, no `java.util` collections anywhere. + +Call flow from `Main.main` (`src/ccwc/Main.java:16`): +1. `Options.parse(args)` (`Options.java:31`) turns argv into an `Options` DTO: four public `boolean` flags (`countBytes`, `countLines`, `countWords`, `countChars`) plus `fileName`. The first non-flag argument becomes `fileName` (`Options.java:39-43`); further non-flag arguments are silently dropped, and unrecognized flags are silently treated as a filename candidate too (no validation/error path). If no flag was set at all, `countBytes`/`countLines`/`countWords` are turned on — deliberately **not** `countChars` (`Options.java:47-52`), matching real `wc`'s default. +2. `Main` branches on `opts.fileName == null` (`Main.java:20`): stdin → `Counter.count(InputStream, Options)` (`Counter.java:66`); file → `Counter.count(Path, Options)` (`Counter.java:37`). +3. `Counter` accumulates results into four public `long` fields (`lines`, `words`, `bytes`, `chars`), reset to `0` at the start of every `count()` call (`Counter.java:38`, `:67`) — the object is reusable but destructive/stateful, not safe for concurrent use. +4. `Main.printResults` (`Main.java:42`) formats output: the classic aligned `"%8d %8d %8d %s\n"` columnar format only when exactly `-c -l -w` are active and `-m` is not (`Main.java:45-46`); otherwise one metric per line via `println`, one `if` block per flag (`Main.java:47-60`). + +**Single-pass streaming is the central design decision.** Lines/words/chars are all derived from one loop, `while ((ch = reader.read()) != -1)`, over a `BufferedReader(InputStreamReader(in, StandardCharsets.UTF_8))` (`Counter.java:97-120`) — each decoded char is inspected once for newline (`ch == '\n'`), word-boundary (`Character.isWhitespace(ch)`, Unicode-aware, not just ASCII), and char-count, all in the same iteration (`Counter.java:104-119`). Byte counting normally can't ride along in that loop (the reader consumes decoded chars, not raw bytes), so when bytes must be counted from a stream, `Counter` wraps the raw `InputStream` in `CountingInputStream` — a `FilterInputStream` decorator (`CountingInputStream.java:12`) — *underneath* the reader (`Counter.java:78`), tallying bytes as they're consumed. Still one physical read pass. + +Two shortcuts skip the full read entirely: +- File + `-c` only → `bytes = Files.size(path)` (`Counter.java:41`), zero bytes actually read. Note: if `-c` is combined with any of `-l`/`-w`/`-m` on a file, the code both stats the file for size *and* streams it for the other metrics (`Counter.java:40-49`) — two filesystem operations, not a bug but not optimal. +- Stdin + `-c` only → raw 8 KB-buffer byte loop (`Counter.java:69-76`), no UTF-8 decoding, no `CountingInputStream` involved. + +```mermaid +flowchart LR + A[String args] --> B[Options.parse] + B --> C{fileName == null?} + C -- yes --> D[Counter.count InputStream] + C -- no --> E[Counter.count Path] + D --> F[CountingInputStream wraps stdin] + E --> G[Files.size shortcut when only -c] + F --> H[BufferedReader UTF-8 single-pass loop] + E --> H + H --> I[Counter fields: lines words bytes chars] + I --> J[Main.printResults] +``` + +## Key Directories +- `src/ccwc/` — all source, 4 files, package `ccwc`. Flat, no sub-packages, no test source root declared anywhere. +- `out/ccwc/` — `javac`/IntelliJ compiler output, mirrors `src/ccwc/` 1:1 (`Main.class`, `Options.class`, `Counter.class`, `CountingInputStream.class`). Gitignored, fully regenerated by every build — never hand-edit or commit anything under `out/`. +- `.idea/` + `ccwc.iml` (repo root) — IntelliJ project metadata; see Runtime/Tooling Preferences below. Only `.idea/misc.xml`, `.idea/modules.xml`, `.idea/vcs.xml` are tracked in git; `.idea/workspace.xml` and `.idea/shelf/` are user-local and gitignored (via `.idea/.gitignore`). +- `.github/workflows/ci.yml` + `.github/scripts/smoke-test.sh` — GitHub Actions CI (see Testing & QA). Repo root docs (`README.md`, `CLAUDE.md`, this file) and manual sample-input fixtures (`test.txt`, `test2.txt`). No `tests/` directory, no test source root anywhere. + +## Development Commands +No build tool — plain `javac`/`java`, run every command from the repo root. + +```bash +# Compile (mirrors what IntelliJ's own build does) +javac -d out src/ccwc/*.java + +# Run against a file +java -cp out ccwc.Main -l test.txt + +# Run against stdin +cat test.txt | java -cp out ccwc.Main -w + +# Windows cmd.exe — avoids PowerShell's BOM-stripping `cat` alias (see Runtime/Tooling Preferences) +type test.txt | java -cp out ccwc.Main -c +``` +That's the complete workflow — there is no lint step, no formatter, no packaging/release step. Flags: `-c` bytes, `-l` lines, `-w` words, `-m` characters; no flags defaults to `-c -l -w`. The first non-flag argument is the filename; if omitted, input is read from stdin. + +## Code Conventions & Common Patterns +- **Formatting**: standard Java style — 4-space indents, opening braces on the same line, full Javadoc (`/** … */`) on every public class/field/method including `@param`/`@return`/`@throws`. Match this exactly for any new public member. +- **Naming**: camelCase throughout. Abbreviated names for widely-scoped DTO variables (`opts`), full words elsewhere (`counter`, `bytesRead`). Boolean flags are prefixed `count*` (`countBytes`, `countLines`, `countWords`, `countChars`). +- **State management**: flat DTOs with **public mutable fields, no getters/setters, no builders, no immutability** — `Options` (flags + filename) and `Counter` (result counts) are both plain data holders read directly by callers, e.g. `Main.printResults` reads `counter.lines` / `counter.bytes` straight off the field (`Main.java:46,49,52,55,58`). `Counter`'s fields are reset at the top of `count()` (`Counter.java:38,67`) rather than the object being recreated — carry this pattern forward rather than introducing immutable value objects. +- **Object creation / no DI**: no dependency-injection framework or container of any kind. Dependencies are wired with plain `new` at the point of use (`new Counter()` — `Main.java:18`; `new CountingInputStream(in)` — `Counter.java:78`) or via a **static factory** instead of a constructor for parsing: `Options.parse(String[] args)` (`Options.java:31`), not `new Options(args)`. Follow the static-factory convention for any new "build me an X from raw input" need. +- **Decorator pattern** for cross-cutting byte counting: `CountingInputStream extends FilterInputStream`, overriding both `read()` (`CountingInputStream.java:33`) and `read(byte[], int, int)` (`CountingInputStream.java:54`), always delegating to `super.read(...)` first and only incrementing the counter when the result isn't `-1`. This is the pattern to copy if another transparent stream-tap is ever needed (e.g. hashing, progress tracking). +- **Error handling**: checked exceptions propagate, they are never caught-and-handled. Every method up the call chain declares `throws IOException` and lets it bubble all the way to the JVM — `main` itself declares `throws IOException` (`Main.java:16`) and there is no error-message/exit-code layer; a missing file surfaces as a raw stack trace. Resource cleanup is done exclusively via try-with-resources (`Counter.java:46-48`, `98-99`), never a manual `close()` in a `finally`. Don't add a `catch` that swallows `IOException` — if you need friendlier error output, that's new scope, wire it consistently through `main`, not ad hoc inside `Counter`. +- **Async patterns**: none. Everything is synchronous, single-threaded, blocking I/O — no `Thread`, `ExecutorService`, or `CompletableFuture` anywhere in the codebase. Keep new code synchronous unless there's a specific reason to change that (and treat that as a deliberate, separate design decision, not an incidental addition). +- **Flag parsing**: a `switch` on string literals maps `-c`/`-l`/`-w`/`-m` to fields (`Options.java:34-38`); the `default` case silently captures the first non-flag argument as the filename and drops everything else without any validation or error message (`Options.java:39-43`). +- **Explicit UTF-8, always**: charset is never left to the platform default — `StandardCharsets.UTF_8` is passed explicitly wherever bytes become chars (`Counter.java:99`). Match this in any new stream-decoding code; never rely on the platform default charset. +- **Verified current output quirks** (compiled and ran the exact source in an isolated scratch dir to confirm — this supersedes `CLAUDE.md`'s older "Known gotcha" wording, which claims stdin output literally prints the word `null`; that is **not** what the current code does): when reading from **stdin**, `Main.java:43` computes `suffix = fileName == null ? "" : fileName`, so `suffix` is an empty string, never the literal text `"null"`. The real, verified defects are (a) a **trailing space** before the line ends in every output branch when `suffix` is empty (e.g. `4 ` instead of `4`), present in both the columnar branch and the per-metric branch equally; and (b) a **line-terminator inconsistency**: the columnar default branch (`Main.java:46`) always emits a literal `\n` from the `printf` format string, while every individual-metric branch (`Main.java:49,52,55,58`) uses `println`, which appends the JVM's platform line separator (`\r\n` on Windows) — so on Windows, default-format output and per-metric output end their lines differently. Don't copy either inconsistency into new output code; if you touch `printResults`, fix the suffix/trailing-space handling and standardize the line terminator in the same change. +- **No multi-file support**: real `wc` accepts multiple filenames and prints a totals line; this implementation only ever takes the first non-flag argument as a filename (`Options.java:39-43`, `Main.java:20-27`) — additional filenames are silently ignored, there's no totals row. + +## Important Files +- `src/ccwc/Main.java` — entry point; `main` (`:16`) and output formatting `printResults` (`:42`). +- `src/ccwc/Options.java` — CLI flag/filename parsing, `Options.parse` (`:31`). +- `src/ccwc/Counter.java` — counting engine; `count(Path, …)` (`:37`), `count(InputStream, …)` (`:66`), single-pass loop in private `countFromStream` (`:97`). +- `src/ccwc/CountingInputStream.java` — byte-counting stream decorator, used only when stdin needs both a byte count and at least one of line/word/char count simultaneously. +- `.idea/misc.xml` — authoritative JDK/language-level declaration (`languageLevel="JDK_24"`, `project-jdk-name="openjdk-24"`) and compiler output path (`out/`). +- `ccwc.iml` — module definition; single source root `src/` (`isTestSource="false"`), no test source root, inherits JDK and output path from `.idea/misc.xml`. +- `README.md` — user-facing usage/build docs, flag examples with expected output numbers, and the PowerShell BOM-stripping gotcha (lines 140-150). +- `CLAUDE.md` — earlier AI-assistant guidance covering similar architecture/build ground as this file. Its "Known gotcha" section is stale (see the Code Conventions note above) — treat this `AGENTS.md` as canonical where the two disagree. +- `test.txt` — 7146-line / 342190-byte UTF-8-with-BOM sample fixture (Project Gutenberg's *The Art of War*) used in every README usage example; **not** an automated test, just sample input. `test2.txt` — a second, unreferenced 165-line/13.0 KB sample file (a YouTube script draft), not mentioned anywhere in source or docs. + +## Runtime/Tooling Preferences +- **JDK 24** (`openjdk-24`, language level `JDK_24`) per `.idea/misc.xml:3` — treat this as the authoritative target version. `README.md` advertises a looser "JDK 17+" minimum; any modern JDK 17+ does compile the code, but match JDK 24 conventions/APIs when in doubt, and don't rely on anything newer than 24. +- **No package manager, no dependency manifest** — confirmed repo-wide absent: no `pom.xml` (Maven), no `build.gradle`/`build.gradle.kts` (Gradle), no `package.json` (npm), no `Makefile`, no `CMakeLists.txt`. The project is intentionally zero-dependency, pure JDK (`java.io` + `java.nio.file` + `java.nio.charset` only). Do not introduce a build tool or third-party dependency without that being an explicit, separate ask. +- **IntelliJ IDEA project** (`ccwc.iml` + `.idea/`), single module, single source root (`src/`). Compile output goes to `out/` (gitignored, regenerated every build). If project settings need to change, edit the tracked `.idea/misc.xml` / `.idea/modules.xml` / `.idea/vcs.xml` — never `.idea/workspace.xml` (user-local, gitignored, not shared). +- **Windows/PowerShell cross-platform note**: PowerShell's `cat`/`Get-Content` strips the UTF-8 BOM before piping to `java.exe`, causing a 3-byte discrepancy in stdin `-c` byte counts versus reading the same file directly. Use `cmd.exe`'s `type` or Git Bash's `cat` instead when verifying stdin byte counts on Windows (`README.md:140-150`). Separately (verified by direct testing, not documented anywhere before this file), `println`-based output lines end with `\r\n` on Windows while the columnar `printf` branch always ends with a literal `\n` — see the Code Conventions note above. + +## Testing & QA +There is still no unit-test framework (confirmed zero matches repo-wide for `junit|testng|mockito|@Test`) and `ccwc.iml` declares no test source root — this remains a deliberate gap, not something to "fix" as a side effect of an unrelated change; adding real unit tests is new scope. + +There **is** CI: `.github/workflows/ci.yml` runs on every push to `master` (the repo's actual default branch — verify with `git ls-remote --symref origin HEAD` if that ever changes) and every PR, on a 2×2 matrix (`ubuntu-latest`/`windows-latest` × JDK `17`/`24` via Temurin — deliberately spanning README's claimed "17+" floor and `.idea/misc.xml`'s configured `24`, and both OSes since `Main.java`'s output has a verified Windows-specific `\r\n`-vs-`\n` quirk, see Code Conventions). Each job compiles (`javac -d out src/ccwc/*.java`) then runs `.github/scripts/smoke-test.sh`, which turns the manual checks below into automated assertions — 12 checks covering every distinct branch in `Counter` (the `Files.size` shortcut, the raw-byte stdin shortcut, the `CountingInputStream` decorator path, and both the columnar and per-metric output branches, for both file and stdin input). This is whole-program/CLI-level coverage driven by `test.txt`, not per-class unit tests. + +To verify a change locally, compile then either run the smoke-test script or repeat its checks by hand — both compare against the same values documented in `README.md`: +```bash +javac -d out src/ccwc/*.java +bash .github/scripts/smoke-test.sh # automated: all 12 checks, exits non-zero on any mismatch + +# equivalent by hand, one flag at a time: +java -cp out ccwc.Main -c test.txt # expect 342190 +java -cp out ccwc.Main -l test.txt # expect 7145 +java -cp out ccwc.Main -w test.txt # expect 58164 +java -cp out ccwc.Main -m test.txt # expect 339292 +java -cp out ccwc.Main test.txt # expect " 7145 58164 342190 test.txt" +``` +`test.txt` / `test2.txt` are sample input fixtures the smoke test happens to assert against, not a hand-written test corpus — there's still no per-class edge-case coverage (empty input, unknown flags, multi-byte UTF-8 boundaries). When changing anything in `Counter` or `CountingInputStream`, run `smoke-test.sh` (or let CI run it on the PR) and confirm all 12 checks pass before considering the change done. diff --git a/CLAUDE.md b/CLAUDE.md new file mode 100644 index 0000000..b38fdb1 --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1,66 @@ +# CLAUDE.md + +This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository. + +## What this is + +`ccwc` is a Java clone of the Unix `wc` command (a Coding Challenges exercise). It +counts bytes, lines, words, and characters in a file or from standard input. + +## Build & run + +There is no build tool (no Maven/Gradle) and no test suite. It is an IntelliJ IDEA +project that compiles with `javac` to `out/`. Run all commands from the repo root. + +```bash +# Compile (mirrors what IntelliJ does) +javac -d out src/ccwc/*.java + +# Run against a file +java -cp out ccwc.Main -l test.txt + +# Run against stdin +cat test.txt | java -cp out ccwc.Main -w +``` + +`test.txt` is a sample input fixture (not a test). The project targets Java language +level 24 (`.idea/misc.xml`); a newer JDK also compiles it. + +## Flags + +`-c` bytes, `-l` lines, `-w` words, `-m` characters. With no flags, the default is +`-c -l -w` (matching `wc`). The first non-flag argument is the filename; if absent, +input is read from stdin. + +## Architecture + +Four classes in package `ccwc` (`src/ccwc/`): + +- **`Main`** — entry point. Delegates parsing to `Options`, invokes `Counter`, then + formats output in `printResults`. Output has two modes: the classic aligned `wc` + format (`%8d %8d %8d filename`) only when exactly `-c -l -w` are active (and not + `-m`); otherwise one metric per line. +- **`Options`** — parses flags and holds them as public fields; applies the + "no flags → `-c -l -w`" default. +- **`Counter`** — the counting engine. Accumulates results in public fields + (`bytes`, `lines`, `words`, `chars`). +- **`CountingInputStream`** — a `FilterInputStream` that tallies bytes read. + +**Key design — single pass (the reason `CountingInputStream` exists):** line, word, +and char counts are gathered by decoding the stream as UTF-8 through a +`BufferedReader` in one pass. Byte counting normally can't share that pass (the +reader consumes decoded chars, not raw bytes), so `Counter` wraps the raw stream in +`CountingInputStream` to tally bytes *underneath* the reader — one read pass yields +all four counts. Two shortcuts avoid reading data when possible: +- File + only `-c` → `Files.size(path)`, no read at all. +- Stdin + only `-c` → a raw byte loop, no character decoding. + +`Counter` has two `count()` overloads (one taking a `Path`, one taking an +`InputStream`) that funnel into the private `countFromStream`. + +## Known gotcha + +When reading from **stdin** with any flag other than the default combination, +`printResults` still appends `opts.fileName`, which is `null` — so output looks like +`58164 null`. The aligned default-format branch handles the null filename correctly; +the per-metric branch does not. diff --git a/README.md b/README.md index b6db668..b1e81e5 100644 --- a/README.md +++ b/README.md @@ -5,6 +5,7 @@ # ccwc - A Java Implementation of the Unix `wc` Tool [![Java](https://img.shields.io/badge/Java-17+-ED8B00?style=for-the-badge&logo=java&logoColor=white)](https://www.oracle.com/java/) +[![CI](https://github.com/MohammadRokib/wc-tool-java/actions/workflows/ci.yml/badge.svg)](https://github.com/MohammadRokib/wc-tool-java/actions/workflows/ci.yml) `ccwc` (Coding Challenges Word Count) is a custom implementation of the classic Unix `wc` (word count) command-line utility, written entirely in Java. It is built to be memory-efficient, scalable for massive files, and fully compatible with standard Unix pipelines.