diff --git a/.github/scripts/smoke-test.sh b/.github/scripts/smoke-test.sh new file mode 100644 index 0000000..a10fc65 --- /dev/null +++ b/.github/scripts/smoke-test.sh @@ -0,0 +1,44 @@ +#!/usr/bin/env bash +# Smoke test for ccwc. Compiles must already have produced ./out (javac -d out src/ccwc/*.java). +# Verifies every distinct code path in Counter against the known-good values for test.txt +# documented in AGENTS.md / README.md: 7145 lines, 58164 words, 342190 bytes, 339292 chars. +# +# Runnable locally from the repo root: bash .github/scripts/smoke-test.sh +set -euo pipefail +cd "$(dirname "$0")/../.." + +fail=0 +check() { # desc expected actual + if [ "$3" = "$2" ]; then + echo "OK $1" + else + echo "FAIL $1 -- expected [$2], got [$3]" + fail=1 + fi +} + +# --- File input, single flag: exercises the Files.size() -c shortcut (Counter.java:41) --- +check "-c file" 342190 "$(java -cp out ccwc.Main -c test.txt | awk '{print $1}')" +check "-l file" 7145 "$(java -cp out ccwc.Main -l test.txt | awk '{print $1}')" +check "-w file" 58164 "$(java -cp out ccwc.Main -w test.txt | awk '{print $1}')" +check "-m file" 339292 "$(java -cp out ccwc.Main -m test.txt | awk '{print $1}')" + +# --- File input, default combo: the columnar printf branch (Main.java:45-46) --- +check "default file" "7145 58164 342190" "$(java -cp out ccwc.Main test.txt | awk '{print $1, $2, $3}')" + +# --- Stdin input, single flag: -c alone exercises the raw 8KB-loop shortcut (Counter.java:69-76) --- +check "-c stdin" 342190 "$(cat test.txt | java -cp out ccwc.Main -c | awk '{print $1}')" +check "-l stdin" 7145 "$(cat test.txt | java -cp out ccwc.Main -l | awk '{print $1}')" +check "-w stdin" 58164 "$(cat test.txt | java -cp out ccwc.Main -w | awk '{print $1}')" +check "-m stdin" 339292 "$(cat test.txt | java -cp out ccwc.Main -m | awk '{print $1}')" + +# --- Stdin input, default combo --- +check "default stdin" "7145 58164 342190" "$(cat test.txt | java -cp out ccwc.Main | awk '{print $1, $2, $3}')" + +# --- Stdin, -c combined with another flag: the ONLY path that exercises the +# CountingInputStream decorator (Counter.java:78-79) instead of either shortcut --- +result="$(cat test.txt | java -cp out ccwc.Main -c -l)" +check "-c -l stdin bytes (decorator path)" 342190 "$(sed -n 1p <<< "$result" | awk '{print $1}')" +check "-c -l stdin lines (decorator path)" 7145 "$(sed -n 2p <<< "$result" | awk '{print $1}')" + +exit $fail diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..ca46538 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,30 @@ +name: CI + +on: + push: + branches: [master] + pull_request: + +jobs: + build-and-verify: + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, windows-latest] + java: ['17', '24'] + runs-on: ${{ matrix.os }} + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-java@v4 + with: + distribution: temurin + java-version: ${{ matrix.java }} + + - name: Compile + shell: bash + run: javac -d out src/ccwc/*.java + + - name: Smoke test + shell: bash + run: bash .github/scripts/smoke-test.sh diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..643c655 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,106 @@ +# Repository Guidelines + +## Project Overview +`ccwc` ("Coding Challenges Word Count") is a Java clone of the Unix `wc` command, built for the [Coding Challenges "Build Your Own wc Tool"](https://codingchallenges.fyi/challenges/challenge-wc) exercise. It counts bytes (`-c`), lines (`-l`), words (`-w`), and characters (`-m`) in a file or from stdin. No flags defaults to `-c -l -w` (bytes+lines+words), matching real `wc`. Core design goal: memory-safe single-pass streaming so it scales to files larger than available RAM, rather than loading input into memory. + +## Architecture & Data Flow +Four classes, all in package `ccwc`, all under `src/ccwc/`. Pure JDK — no third-party libraries, no `java.util` collections anywhere. + +Call flow from `Main.main` (`src/ccwc/Main.java:16`): +1. `Options.parse(args)` (`Options.java:31`) turns argv into an `Options` DTO: four public `boolean` flags (`countBytes`, `countLines`, `countWords`, `countChars`) plus `fileName`. The first non-flag argument becomes `fileName` (`Options.java:39-43`); further non-flag arguments are silently dropped, and unrecognized flags are silently treated as a filename candidate too (no validation/error path). If no flag was set at all, `countBytes`/`countLines`/`countWords` are turned on — deliberately **not** `countChars` (`Options.java:47-52`), matching real `wc`'s default. +2. `Main` branches on `opts.fileName == null` (`Main.java:20`): stdin → `Counter.count(InputStream, Options)` (`Counter.java:66`); file → `Counter.count(Path, Options)` (`Counter.java:37`). +3. `Counter` accumulates results into four public `long` fields (`lines`, `words`, `bytes`, `chars`), reset to `0` at the start of every `count()` call (`Counter.java:38`, `:67`) — the object is reusable but destructive/stateful, not safe for concurrent use. +4. `Main.printResults` (`Main.java:42`) formats output: the classic aligned `"%8d %8d %8d %s\n"` columnar format only when exactly `-c -l -w` are active and `-m` is not (`Main.java:45-46`); otherwise one metric per line via `println`, one `if` block per flag (`Main.java:47-60`). + +**Single-pass streaming is the central design decision.** Lines/words/chars are all derived from one loop, `while ((ch = reader.read()) != -1)`, over a `BufferedReader(InputStreamReader(in, StandardCharsets.UTF_8))` (`Counter.java:97-120`) — each decoded char is inspected once for newline (`ch == '\n'`), word-boundary (`Character.isWhitespace(ch)`, Unicode-aware, not just ASCII), and char-count, all in the same iteration (`Counter.java:104-119`). Byte counting normally can't ride along in that loop (the reader consumes decoded chars, not raw bytes), so when bytes must be counted from a stream, `Counter` wraps the raw `InputStream` in `CountingInputStream` — a `FilterInputStream` decorator (`CountingInputStream.java:12`) — *underneath* the reader (`Counter.java:78`), tallying bytes as they're consumed. Still one physical read pass. + +Two shortcuts skip the full read entirely: +- File + `-c` only → `bytes = Files.size(path)` (`Counter.java:41`), zero bytes actually read. Note: if `-c` is combined with any of `-l`/`-w`/`-m` on a file, the code both stats the file for size *and* streams it for the other metrics (`Counter.java:40-49`) — two filesystem operations, not a bug but not optimal. +- Stdin + `-c` only → raw 8 KB-buffer byte loop (`Counter.java:69-76`), no UTF-8 decoding, no `CountingInputStream` involved. + +```mermaid +flowchart LR + A[String args] --> B[Options.parse] + B --> C{fileName == null?} + C -- yes --> D[Counter.count InputStream] + C -- no --> E[Counter.count Path] + D --> F[CountingInputStream wraps stdin] + E --> G[Files.size shortcut when only -c] + F --> H[BufferedReader UTF-8 single-pass loop] + E --> H + H --> I[Counter fields: lines words bytes chars] + I --> J[Main.printResults] +``` + +## Key Directories +- `src/ccwc/` — all source, 4 files, package `ccwc`. Flat, no sub-packages, no test source root declared anywhere. +- `out/ccwc/` — `javac`/IntelliJ compiler output, mirrors `src/ccwc/` 1:1 (`Main.class`, `Options.class`, `Counter.class`, `CountingInputStream.class`). Gitignored, fully regenerated by every build — never hand-edit or commit anything under `out/`. +- `.idea/` + `ccwc.iml` (repo root) — IntelliJ project metadata; see Runtime/Tooling Preferences below. Only `.idea/misc.xml`, `.idea/modules.xml`, `.idea/vcs.xml` are tracked in git; `.idea/workspace.xml` and `.idea/shelf/` are user-local and gitignored (via `.idea/.gitignore`). +- `.github/workflows/ci.yml` + `.github/scripts/smoke-test.sh` — GitHub Actions CI (see Testing & QA). Repo root docs (`README.md`, `CLAUDE.md`, this file) and manual sample-input fixtures (`test.txt`, `test2.txt`). No `tests/` directory, no test source root anywhere. + +## Development Commands +No build tool — plain `javac`/`java`, run every command from the repo root. + +```bash +# Compile (mirrors what IntelliJ's own build does) +javac -d out src/ccwc/*.java + +# Run against a file +java -cp out ccwc.Main -l test.txt + +# Run against stdin +cat test.txt | java -cp out ccwc.Main -w + +# Windows cmd.exe — avoids PowerShell's BOM-stripping `cat` alias (see Runtime/Tooling Preferences) +type test.txt | java -cp out ccwc.Main -c +``` +That's the complete workflow — there is no lint step, no formatter, no packaging/release step. Flags: `-c` bytes, `-l` lines, `-w` words, `-m` characters; no flags defaults to `-c -l -w`. The first non-flag argument is the filename; if omitted, input is read from stdin. + +## Code Conventions & Common Patterns +- **Formatting**: standard Java style — 4-space indents, opening braces on the same line, full Javadoc (`/** … */`) on every public class/field/method including `@param`/`@return`/`@throws`. Match this exactly for any new public member. +- **Naming**: camelCase throughout. Abbreviated names for widely-scoped DTO variables (`opts`), full words elsewhere (`counter`, `bytesRead`). Boolean flags are prefixed `count*` (`countBytes`, `countLines`, `countWords`, `countChars`). +- **State management**: flat DTOs with **public mutable fields, no getters/setters, no builders, no immutability** — `Options` (flags + filename) and `Counter` (result counts) are both plain data holders read directly by callers, e.g. `Main.printResults` reads `counter.lines` / `counter.bytes` straight off the field (`Main.java:46,49,52,55,58`). `Counter`'s fields are reset at the top of `count()` (`Counter.java:38,67`) rather than the object being recreated — carry this pattern forward rather than introducing immutable value objects. +- **Object creation / no DI**: no dependency-injection framework or container of any kind. Dependencies are wired with plain `new` at the point of use (`new Counter()` — `Main.java:18`; `new CountingInputStream(in)` — `Counter.java:78`) or via a **static factory** instead of a constructor for parsing: `Options.parse(String[] args)` (`Options.java:31`), not `new Options(args)`. Follow the static-factory convention for any new "build me an X from raw input" need. +- **Decorator pattern** for cross-cutting byte counting: `CountingInputStream extends FilterInputStream`, overriding both `read()` (`CountingInputStream.java:33`) and `read(byte[], int, int)` (`CountingInputStream.java:54`), always delegating to `super.read(...)` first and only incrementing the counter when the result isn't `-1`. This is the pattern to copy if another transparent stream-tap is ever needed (e.g. hashing, progress tracking). +- **Error handling**: checked exceptions propagate, they are never caught-and-handled. Every method up the call chain declares `throws IOException` and lets it bubble all the way to the JVM — `main` itself declares `throws IOException` (`Main.java:16`) and there is no error-message/exit-code layer; a missing file surfaces as a raw stack trace. Resource cleanup is done exclusively via try-with-resources (`Counter.java:46-48`, `98-99`), never a manual `close()` in a `finally`. Don't add a `catch` that swallows `IOException` — if you need friendlier error output, that's new scope, wire it consistently through `main`, not ad hoc inside `Counter`. +- **Async patterns**: none. Everything is synchronous, single-threaded, blocking I/O — no `Thread`, `ExecutorService`, or `CompletableFuture` anywhere in the codebase. Keep new code synchronous unless there's a specific reason to change that (and treat that as a deliberate, separate design decision, not an incidental addition). +- **Flag parsing**: a `switch` on string literals maps `-c`/`-l`/`-w`/`-m` to fields (`Options.java:34-38`); the `default` case silently captures the first non-flag argument as the filename and drops everything else without any validation or error message (`Options.java:39-43`). +- **Explicit UTF-8, always**: charset is never left to the platform default — `StandardCharsets.UTF_8` is passed explicitly wherever bytes become chars (`Counter.java:99`). Match this in any new stream-decoding code; never rely on the platform default charset. +- **Verified current output quirks** (compiled and ran the exact source in an isolated scratch dir to confirm — this supersedes `CLAUDE.md`'s older "Known gotcha" wording, which claims stdin output literally prints the word `null`; that is **not** what the current code does): when reading from **stdin**, `Main.java:43` computes `suffix = fileName == null ? "" : fileName`, so `suffix` is an empty string, never the literal text `"null"`. The real, verified defects are (a) a **trailing space** before the line ends in every output branch when `suffix` is empty (e.g. `4 ` instead of `4`), present in both the columnar branch and the per-metric branch equally; and (b) a **line-terminator inconsistency**: the columnar default branch (`Main.java:46`) always emits a literal `\n` from the `printf` format string, while every individual-metric branch (`Main.java:49,52,55,58`) uses `println`, which appends the JVM's platform line separator (`\r\n` on Windows) — so on Windows, default-format output and per-metric output end their lines differently. Don't copy either inconsistency into new output code; if you touch `printResults`, fix the suffix/trailing-space handling and standardize the line terminator in the same change. +- **No multi-file support**: real `wc` accepts multiple filenames and prints a totals line; this implementation only ever takes the first non-flag argument as a filename (`Options.java:39-43`, `Main.java:20-27`) — additional filenames are silently ignored, there's no totals row. + +## Important Files +- `src/ccwc/Main.java` — entry point; `main` (`:16`) and output formatting `printResults` (`:42`). +- `src/ccwc/Options.java` — CLI flag/filename parsing, `Options.parse` (`:31`). +- `src/ccwc/Counter.java` — counting engine; `count(Path, …)` (`:37`), `count(InputStream, …)` (`:66`), single-pass loop in private `countFromStream` (`:97`). +- `src/ccwc/CountingInputStream.java` — byte-counting stream decorator, used only when stdin needs both a byte count and at least one of line/word/char count simultaneously. +- `.idea/misc.xml` — authoritative JDK/language-level declaration (`languageLevel="JDK_24"`, `project-jdk-name="openjdk-24"`) and compiler output path (`out/`). +- `ccwc.iml` — module definition; single source root `src/` (`isTestSource="false"`), no test source root, inherits JDK and output path from `.idea/misc.xml`. +- `README.md` — user-facing usage/build docs, flag examples with expected output numbers, and the PowerShell BOM-stripping gotcha (lines 140-150). +- `CLAUDE.md` — earlier AI-assistant guidance covering similar architecture/build ground as this file. Its "Known gotcha" section is stale (see the Code Conventions note above) — treat this `AGENTS.md` as canonical where the two disagree. +- `test.txt` — 7146-line / 342190-byte UTF-8-with-BOM sample fixture (Project Gutenberg's *The Art of War*) used in every README usage example; **not** an automated test, just sample input. `test2.txt` — a second, unreferenced 165-line/13.0 KB sample file (a YouTube script draft), not mentioned anywhere in source or docs. + +## Runtime/Tooling Preferences +- **JDK 24** (`openjdk-24`, language level `JDK_24`) per `.idea/misc.xml:3` — treat this as the authoritative target version. `README.md` advertises a looser "JDK 17+" minimum; any modern JDK 17+ does compile the code, but match JDK 24 conventions/APIs when in doubt, and don't rely on anything newer than 24. +- **No package manager, no dependency manifest** — confirmed repo-wide absent: no `pom.xml` (Maven), no `build.gradle`/`build.gradle.kts` (Gradle), no `package.json` (npm), no `Makefile`, no `CMakeLists.txt`. The project is intentionally zero-dependency, pure JDK (`java.io` + `java.nio.file` + `java.nio.charset` only). Do not introduce a build tool or third-party dependency without that being an explicit, separate ask. +- **IntelliJ IDEA project** (`ccwc.iml` + `.idea/`), single module, single source root (`src/`). Compile output goes to `out/` (gitignored, regenerated every build). If project settings need to change, edit the tracked `.idea/misc.xml` / `.idea/modules.xml` / `.idea/vcs.xml` — never `.idea/workspace.xml` (user-local, gitignored, not shared). +- **Windows/PowerShell cross-platform note**: PowerShell's `cat`/`Get-Content` strips the UTF-8 BOM before piping to `java.exe`, causing a 3-byte discrepancy in stdin `-c` byte counts versus reading the same file directly. Use `cmd.exe`'s `type` or Git Bash's `cat` instead when verifying stdin byte counts on Windows (`README.md:140-150`). Separately (verified by direct testing, not documented anywhere before this file), `println`-based output lines end with `\r\n` on Windows while the columnar `printf` branch always ends with a literal `\n` — see the Code Conventions note above. + +## Testing & QA +There is still no unit-test framework (confirmed zero matches repo-wide for `junit|testng|mockito|@Test`) and `ccwc.iml` declares no test source root — this remains a deliberate gap, not something to "fix" as a side effect of an unrelated change; adding real unit tests is new scope. + +There **is** CI: `.github/workflows/ci.yml` runs on every push to `master` (the repo's actual default branch — verify with `git ls-remote --symref origin HEAD` if that ever changes) and every PR, on a 2×2 matrix (`ubuntu-latest`/`windows-latest` × JDK `17`/`24` via Temurin — deliberately spanning README's claimed "17+" floor and `.idea/misc.xml`'s configured `24`, and both OSes since `Main.java`'s output has a verified Windows-specific `\r\n`-vs-`\n` quirk, see Code Conventions). Each job compiles (`javac -d out src/ccwc/*.java`) then runs `.github/scripts/smoke-test.sh`, which turns the manual checks below into automated assertions — 12 checks covering every distinct branch in `Counter` (the `Files.size` shortcut, the raw-byte stdin shortcut, the `CountingInputStream` decorator path, and both the columnar and per-metric output branches, for both file and stdin input). This is whole-program/CLI-level coverage driven by `test.txt`, not per-class unit tests. + +To verify a change locally, compile then either run the smoke-test script or repeat its checks by hand — both compare against the same values documented in `README.md`: +```bash +javac -d out src/ccwc/*.java +bash .github/scripts/smoke-test.sh # automated: all 12 checks, exits non-zero on any mismatch + +# equivalent by hand, one flag at a time: +java -cp out ccwc.Main -c test.txt # expect 342190 +java -cp out ccwc.Main -l test.txt # expect 7145 +java -cp out ccwc.Main -w test.txt # expect 58164 +java -cp out ccwc.Main -m test.txt # expect 339292 +java -cp out ccwc.Main test.txt # expect " 7145 58164 342190 test.txt" +``` +`test.txt` / `test2.txt` are sample input fixtures the smoke test happens to assert against, not a hand-written test corpus — there's still no per-class edge-case coverage (empty input, unknown flags, multi-byte UTF-8 boundaries). When changing anything in `Counter` or `CountingInputStream`, run `smoke-test.sh` (or let CI run it on the PR) and confirm all 12 checks pass before considering the change done. diff --git a/CLAUDE.md b/CLAUDE.md new file mode 100644 index 0000000..b38fdb1 --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1,66 @@ +# CLAUDE.md + +This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository. + +## What this is + +`ccwc` is a Java clone of the Unix `wc` command (a Coding Challenges exercise). It +counts bytes, lines, words, and characters in a file or from standard input. + +## Build & run + +There is no build tool (no Maven/Gradle) and no test suite. It is an IntelliJ IDEA +project that compiles with `javac` to `out/`. Run all commands from the repo root. + +```bash +# Compile (mirrors what IntelliJ does) +javac -d out src/ccwc/*.java + +# Run against a file +java -cp out ccwc.Main -l test.txt + +# Run against stdin +cat test.txt | java -cp out ccwc.Main -w +``` + +`test.txt` is a sample input fixture (not a test). The project targets Java language +level 24 (`.idea/misc.xml`); a newer JDK also compiles it. + +## Flags + +`-c` bytes, `-l` lines, `-w` words, `-m` characters. With no flags, the default is +`-c -l -w` (matching `wc`). The first non-flag argument is the filename; if absent, +input is read from stdin. + +## Architecture + +Four classes in package `ccwc` (`src/ccwc/`): + +- **`Main`** — entry point. Delegates parsing to `Options`, invokes `Counter`, then + formats output in `printResults`. Output has two modes: the classic aligned `wc` + format (`%8d %8d %8d filename`) only when exactly `-c -l -w` are active (and not + `-m`); otherwise one metric per line. +- **`Options`** — parses flags and holds them as public fields; applies the + "no flags → `-c -l -w`" default. +- **`Counter`** — the counting engine. Accumulates results in public fields + (`bytes`, `lines`, `words`, `chars`). +- **`CountingInputStream`** — a `FilterInputStream` that tallies bytes read. + +**Key design — single pass (the reason `CountingInputStream` exists):** line, word, +and char counts are gathered by decoding the stream as UTF-8 through a +`BufferedReader` in one pass. Byte counting normally can't share that pass (the +reader consumes decoded chars, not raw bytes), so `Counter` wraps the raw stream in +`CountingInputStream` to tally bytes *underneath* the reader — one read pass yields +all four counts. Two shortcuts avoid reading data when possible: +- File + only `-c` → `Files.size(path)`, no read at all. +- Stdin + only `-c` → a raw byte loop, no character decoding. + +`Counter` has two `count()` overloads (one taking a `Path`, one taking an +`InputStream`) that funnel into the private `countFromStream`. + +## Known gotcha + +When reading from **stdin** with any flag other than the default combination, +`printResults` still appends `opts.fileName`, which is `null` — so output looks like +`58164 null`. The aligned default-format branch handles the null filename correctly; +the per-metric branch does not. diff --git a/README.md b/README.md new file mode 100644 index 0000000..b1e81e5 --- /dev/null +++ b/README.md @@ -0,0 +1,166 @@ + + +
+ +# ccwc - A Java Implementation of the Unix `wc` Tool + +[![Java](https://img.shields.io/badge/Java-17+-ED8B00?style=for-the-badge&logo=java&logoColor=white)](https://www.oracle.com/java/) +[![CI](https://github.com/MohammadRokib/wc-tool-java/actions/workflows/ci.yml/badge.svg)](https://github.com/MohammadRokib/wc-tool-java/actions/workflows/ci.yml) + +`ccwc` (Coding Challenges Word Count) is a custom implementation of the classic Unix `wc` (word count) command-line utility, written entirely in Java. It is built to be memory-efficient, scalable for massive files, and fully compatible with standard Unix pipelines. + +This project was built as part of the [Build Your Own wc Tool Challenge](https://codingchallenges.fyi/challenges/challenge-wc). + +
+ +

+ Explore the docs · + Report Bug · + LinkedIn · + Email +

+ +--- + +## Features + +- **`-c`**: Count bytes in a file or stream. +- **`-l`**: Count lines (newline characters). +- **`-w`**: Count words (sequences of characters delimited by whitespace). +- **`-m`**: Count characters (correctly handles multi-byte UTF-8 encoded text). +
+ +- **Default Mode**: Output lines, words, and bytes simultaneously when no flag is provided. +- **Standard Input (stdin)**: Supports Unix piping (e.g., `cat file.txt | ccwc -l`). +- **Memory Safe**: Uses a streaming single-pass architecture. It can process files larger than available RAM without crashing. + +

Back to top ⬆️

+ +--- + +## Prerequisites + +- **Java Development Kit (JDK) 17** or higher. +- A terminal/command prompt environment. + +

Back to top ⬆️

+ +--- + +## Installation & Building + +Because this project uses standard Java libraries with no external dependencies, you can compile it directly using `javac`. + +1. Clone the repository: + ```bash + git clone https://github.com/MohammadRokib/wc-tool-java.git + cd ccwc + ``` + +2. Compile the Java source files into an `out` directory: + ```bash + # On Linux / macOS / Git Bash / Windows CMD + javac -d out src/ccwc/*.java + ``` + +

Back to top ⬆️

+ +--- + +## Usage + +The application is run via the `java` command, pointing to the `out` directory as the classpath. + +### Syntax +```bash +java -cp out ccwc.Main [-c] [-l] [-w] [-m] [filename] +``` + +*If no `filename` is provided, the tool automatically reads from standard input (`stdin`).* + +### Examples + +**1. Count bytes in a file:** +```bash +$ java -cp out ccwc.Main -c test.txt +342190 test.txt +``` + +**2. Count lines in a file:** +```bash +$ java -cp out ccwc.Main -l test.txt +7145 test.txt +``` + +**3. Count words in a file:** +```bash +$ java -cp out ccwc.Main -w test.txt +58164 test.txt +``` + +**4. Count characters in a file (UTF-8 aware):** +```bash +$ java -cp out ccwc.Main -m test.txt +339292 test.txt +``` + +**5. Default mode (lines, words, bytes):** +```bash +$ java -cp out ccwc.Main test.txt + 7145 58164 342190 test.txt +``` + +**6. Reading from Standard Input (Piping):** +When reading from `stdin`, the filename is omitted from the output. +```bash +$ cat test.txt | java -cp out ccwc.Main -l +7145 +``` + +

Back to top ⬆️

+ +--- + +## Architecture & Design + +Instead of reading entire files into memory (which causes `OutOfMemoryError` on large files), `ccwc` uses a **single-pass, streaming architecture**. + +The project is divided into four main components: + +1. **`Main.java`**: The entry point. Delegates argument parsing to `Options`, invokes the `Counter`, and formats the standard output using `System.out.printf`. +2. **`Options.java`**: A Data Transfer Object (DTO) that parses command-line arguments into boolean flags. If no flags are provided, it automatically enables the default metrics (lines, words, bytes). +3. **`Counter.java`**: The core engine. It features two entry points: + - `count(Path path, Options)`: For file inputs. Uses `Files.size()` for an instant O(1) byte count, avoiding unnecessary disk reads. + - `count(InputStream in, Options)`: For standard input. Uses a shared `countFromStream` method that loops through decoded characters exactly once, checking for line breaks, word boundaries, and character counts simultaneously. +4. **`CountingInputStream.java`**: Extends `FilterInputStream` (Decorator Pattern). When reading from `stdin`, this class sits at the bottom of the stream stack, intercepting raw bytes to tally the total byte count as they flow up to the character decoder. + +

Back to top ⬆️

+ +--- + +## Environment Notes (PowerShell Users) + +If you are testing the standard input byte count (`-c`) on Windows using **PowerShell**, you may notice a 3-byte discrepancy compared to reading the file directly (e.g., `342187` instead of `342190`). + +**Why?** PowerShell's `cat` alias (`Get-Content`) decodes files into .NET strings and silently strips the 3-byte UTF-8 Byte Order Mark (BOM) before piping the data to external executables like `java.exe`. + +Your Java code is correct. To verify raw byte piping on Windows, use **Command Prompt (`cmd.exe`)** with the `type` command, or use **Git Bash**: +```cmd +:: In cmd.exe +type test.txt | java -cp out ccwc.Main -c +``` + +

Back to top ⬆️

+ +--- + +## Contact + +Mohammad Rokib + +- **[LinkedIn](https://www.linkedin.com/in/m0hammadrokib/)** +- **[Email](mailto:mohammadrokibkhan@gmail.com)** +- **[GitHub](https://github.com/MohammadRokib)** +- **[Project Link: wc-tool-java](https://github.com/MohammadRokib/wc-tool-java)** + +

Back to top ⬆️

diff --git a/src/ccwc/Counter.java b/src/ccwc/Counter.java new file mode 100644 index 0000000..dff2350 --- /dev/null +++ b/src/ccwc/Counter.java @@ -0,0 +1,123 @@ +package ccwc; + +import java.io.BufferedReader; +import java.io.IOException; +import java.io.InputStream; +import java.io.InputStreamReader; +import java.nio.charset.StandardCharsets; +import java.nio.file.Files; +import java.nio.file.Path; + +/** + * Accumulates byte, line, word, and character counts for a file. + * A single pass is made over the file and all requested counts are + * collected concurrently. + */ +public class Counter { + /** Number of lines counted in the file. */ + public long lines = 0; + /** Number of words counted in the file. */ + public long words = 0; + /** Number of bytes counted in the file. */ + public long bytes = 0; + /** Number of characters counted in the file. */ + public long chars = 0; + + /** + * Counts the specified metrics from the given file. Which metrics + * are collected is controlled by the flags set on {@code opts}. + * The byte count is obtained directly from the file system; the + * remaining metrics are gathered by streaming UTF-8 decoded + * characters through a {@code BufferedReader}. + * + * @param path the path to the file to count + * @param opts specifies which counters to enable + * @throws IOException if an I/O error occurs + */ + public void count(Path path, Options opts) throws IOException { + lines = words = bytes = chars = 0; + + if (opts.countBytes) { + bytes = Files.size(path); + } + + boolean needCharStream = opts.countLines || opts.countWords || opts.countChars; + if (needCharStream) { + try (InputStream in = Files.newInputStream(path)) { + countFromStream(in, opts); + } + } + } + + /** + * Counts the specified metrics from the given input stream. Which + * metrics are collected is controlled by the flags set on {@code opts}. + *

+ * When only byte counting is requested the stream is read directly + * in a raw byte loop. Otherwise, the stream is wrapped in a + * {@link CountingInputStream} so that line, word, and character + * counting can share the single read pass while bytes are still + * accumulated. + * + * @param in the input stream to read from + * @param opts specifies which counters to enable + * @throws IOException if an I/O error occurs + */ + public void count(InputStream in, Options opts) throws IOException { + lines = words = chars = bytes = 0; + + if (opts.countBytes && !opts.countLines && !opts.countWords && !opts.countChars) { + byte[] buffer = new byte[8192]; + int bytesRead; + while((bytesRead = in.read(buffer)) != -1) { + bytes += bytesRead; + } + return; + } + + CountingInputStream countingIn = new CountingInputStream(in); + countFromStream(countingIn, opts); + + if (opts.countBytes) { + bytes = countingIn.getBytesRead(); + } + } + + /** + * Reads UTF-8 decoded characters from the given input stream and + * increments line, word, and character counters as requested by + * {@code opts}. If a {@link CountingInputStream} is passed, its + * byte counter is also accumulated into {@link #bytes}. + * + * @param in the input stream to read from (may be a + * {@code CountingInputStream} for byte tracking) + * @param opts specifies which counters to enable + * @throws IOException if an I/O error occurs + */ + private void countFromStream(InputStream in, Options opts) throws IOException { + try (BufferedReader reader = new BufferedReader( + new InputStreamReader(in, StandardCharsets.UTF_8))) { + + boolean inWord = false; + int ch; + while((ch = reader.read())!= -1) { + if (opts.countChars) { + chars++; + } + + if (opts.countLines && ch == '\n') { + lines++; + } + + if (opts.countWords) { + if (Character.isWhitespace(ch)) { + inWord = false; + } else if (!inWord) { + words++; + inWord = true; + } + } + } + } + } +} diff --git a/src/ccwc/CountingInputStream.java b/src/ccwc/CountingInputStream.java new file mode 100644 index 0000000..6a210cf --- /dev/null +++ b/src/ccwc/CountingInputStream.java @@ -0,0 +1,71 @@ +package ccwc; + +import java.io.FilterInputStream; +import java.io.IOException; +import java.io.InputStream; + +/** + * An {@link java.io.InputStream} wrapper that tracks the total number of bytes + * read from the underlying stream. Useful for counting bytes without an + * additional pass over the data. + */ +public class CountingInputStream extends FilterInputStream { + private long bytesRead = 0; + + /** + * Constructs a {@code CountingInputStream} wrapping the given input stream. + * + * @param in the input stream to wrap + */ + CountingInputStream(InputStream in) { + super(in); + } + + /** + * Reads a single byte from the underlying stream and increments the + * byte counter if a byte was successfully read. + * + * @return the next byte of data, or {@code -1} if the end of the stream + * has been reached + * @throws IOException if an I/O error occurs + */ + @Override + public int read() throws IOException { + int b = super.read(); + if (b != -1) { + bytesRead++; + } + return b; + } + + /** + * Reads up to {@code len} bytes from the underlying stream into the + * given buffer and increments the byte counter by the number of bytes + * actually read. + * + * @param b the buffer into which the data is read + * @param off the start offset in the buffer at which the data is written + * @param len the maximum number of bytes to read + * @return the total number of bytes read into the buffer, or {@code -1} + * if there is no more data + * @throws IOException if an I/O error occurs + */ + @Override + public int read(byte[] b, int off, int len) throws IOException { + int n = super.read(b, off, len); + if (n != -1) { + bytesRead += n; + } + return n; + } + + /** + * Returns the total number of bytes that have been read from the + * underlying stream since this object was created. + * + * @return the total bytes read + */ + long getBytesRead() { + return bytesRead; + } +} diff --git a/src/ccwc/Main.java b/src/ccwc/Main.java index 8ffb773..ee0c693 100644 --- a/src/ccwc/Main.java +++ b/src/ccwc/Main.java @@ -1,118 +1,62 @@ package ccwc; -import java.io.BufferedReader; import java.io.IOException; -import java.io.InputStream; -import java.io.InputStreamReader; -import java.nio.charset.StandardCharsets; -import java.nio.file.Files; import java.nio.file.Path; import java.nio.file.Paths; public class Main { - public static void main(String[] args) throws IOException { - if (args.length < 2) { - System.err.println("Usage: ccwc (-c | -l) "); - System.exit(1); - } - - String flag = args[0]; - String fileName = args[1]; - Path path = Paths.get(fileName); - - switch (flag) { - case "-c": - System.out.println(countBytes(path) + " " + fileName); - break; - case "-l": - System.out.println(countLines(path) + " " + fileName); - break; - case "-w": - System.out.println(countWords(path) + " " + fileName); - break; - case "-m": - System.out.println(countChars(path) + " " + fileName); - break; - default: - System.err.println("Unknown flag: " + flag); - System.exit(1); - } - } /** - * Counts the number of bytes in a file by streaming it through a buffer, - * without ever loading the whole file into memory - * */ - private static long countBytes(Path path) throws IOException { - long count = 0; - - try (InputStream in = Files.newInputStream(path)) { - byte[] buffer = new byte[8192]; - int bytesRead; - - while((bytesRead = in.read(buffer)) != -1) { - count += bytesRead; - } + * Entry point for the ccwc word count utility. Parses command-line arguments + * and delegates to the {@link Counter} class to perform the requested counts. + * + * @param args command-line arguments, expected to contain a flag and a filename + * @throws IOException if an I/O error occurs reading the file + */ + public static void main(String[] args) throws IOException { + Options opts = Options.parse(args); + Counter counter = new Counter(); + + if (opts.fileName == null) { + counter.count(System.in, opts); + printResults(counter, opts, null); + } else { + Path path = Paths.get(opts.fileName); + counter.count(path, opts); + printResults(counter, opts, opts.fileName); } - return count; } /** - * Counts newline characters by streaming decoded characters through a BufferedReader - * */ - private static long countLines(Path path) throws IOException { - long count = 0; - - try (BufferedReader reader = new BufferedReader( - new InputStreamReader(Files.newInputStream(path), StandardCharsets.UTF_8))) { - int ch; - while ((ch = reader.read()) != -1) { - if (ch == '\n') { - count++; - } + * Prints the results from the counter to standard output. When all of + * {@code -c}, {@code -l}, {@code -w} are active and {@code -m} is not, + * the output is formatted as {@code "%8d %8d %8d [filename]"} matching + * the classic {@code wc} default format. Otherwise, each enabled metric + * is printed on its own line. + * + * @param counter the counter whose results to print + * @param opts the options indicating which metrics are enabled + * @param fileName the file name to include in the output, or {@code null} + * when reading from standard input + */ + private static void printResults(Counter counter, Options opts, String fileName) { + String suffix = fileName == null ? "" : fileName; + + if (opts.countBytes && opts.countLines && opts.countWords && !opts.countChars) { + System.out.printf("%8d %8d %8d %s\n", counter.lines, counter.words, counter.bytes, suffix); + } else { + if (opts.countBytes) { + System.out.println(counter.bytes + " " + suffix); } - } - return count; - } - - /** - * Counts words by streaming decoded characters and detecting transitions - * from whitespace to non-whitespace. - * */ - private static long countWords(Path path) throws IOException { - long count = 0; - boolean inWord = false; - - try (BufferedReader reader = new BufferedReader( - new InputStreamReader(Files.newInputStream(path), StandardCharsets.UTF_8))) { - int ch; - while ((ch = reader.read()) != -1) { - if (Character.isWhitespace(ch)) { - inWord = false; - } else if (!inWord) { - count++; - inWord = true; - } + if (opts.countLines) { + System.out.println(counter.lines + " " + suffix); } - } - return count; - } - - /** - * Counts characters by streaming decoded UTF-8 characters. The InputStreamReader's - * internal decoder handles multibyte sequences, so each read() returns one logical - * character regardless of how many bytes it occupies on disk. - * */ - private static long countChars(Path path) throws IOException { - long count = 0; - - try (BufferedReader reader = new BufferedReader( - new InputStreamReader(Files.newInputStream(path), StandardCharsets.UTF_8))) { - int ch; - while((ch = reader.read()) != -1) { - count++; + if (opts.countWords) { + System.out.println(counter.words + " " + suffix); + } + if (opts.countChars) { + System.out.println(counter.chars + " " + suffix); } } - return count; } } diff --git a/src/ccwc/Options.java b/src/ccwc/Options.java new file mode 100644 index 0000000..4449ddf --- /dev/null +++ b/src/ccwc/Options.java @@ -0,0 +1,55 @@ +package ccwc; + +/** + * Parses and stores command-line flag options for the ccwc utility. + * If no flag is specified, all metrics (bytes, lines, words, chars) + * are enabled by default. + */ +public class Options { + /** Whether to count bytes ({@code -c}). */ + public boolean countBytes = false; + /** Whether to count lines ({@code -l}). */ + public boolean countLines = false; + /** Whether to count words ({@code -w}). */ + public boolean countWords = false; + /** Whether to count characters ({@code -m}). */ + public boolean countChars = false; + /** The filename argument, or {@code null} if not provided. */ + public String fileName = null; + + /** + * Parses the command-line arguments and returns an {@code Options} + * instance with the appropriate flags set. + *

+ * Recognized flags are {@code -c}, {@code -l}, {@code -w}, and + * {@code -m}. The first non-flag argument is treated as the filename. + * If no flags are specified, all four metrics are enabled. + * + * @param args the command-line arguments to parse + * @return an {@code Options} instance reflecting the parsed flags + */ + public static Options parse(String[] args) { + Options opts = new Options(); + for (String arg : args) { + switch (arg) { + case "-c": opts.countBytes = true; break; + case "-l": opts.countLines = true; break; + case "-w": opts.countWords = true; break; + case "-m": opts.countChars = true; break; + default: + if (opts.fileName == null) { + opts.fileName = arg; + } + break; + } + } + + boolean anyFlagSet = opts.countBytes || opts.countLines || opts.countWords || opts.countChars; + if (!anyFlagSet) { + opts.countBytes = true; + opts.countLines = true; + opts.countWords = true; + } + return opts; + } +}