diff --git a/avro/src/test/scala/dev/constructive/eo/avro/AvroBytesSpec.scala b/avro/src/test/scala/dev/constructive/eo/avro/AvroBytesSpec.scala index 0dc8cc8a..6c8a913f 100644 --- a/avro/src/test/scala/dev/constructive/eo/avro/AvroBytesSpec.scala +++ b/avro/src/test/scala/dev/constructive/eo/avro/AvroBytesSpec.scala @@ -382,4 +382,133 @@ class AvroBytesSpec extends Specification with ScalaCheck: .and(AvroBinaryCursor.zigZagInt(Int.MinValue).length === 5) } + // ---- byte-cursor refusal identities ------------------------------------- + // + // Every row below asserts WHICH `AvroFailure` comes back, never `isLeft`. That is the whole + // point: each of these guards, when removed, lets the cursor run on into the Avro runtime, which + // throws, which `locateFrom` catches and reports as `BinaryParseFailed`. A `Left` either way — + // so an `isLeft` assertion pins nothing, and the structured diagnostic the `.record` face + // publishes silently degrades to "parse failed". + // + // Called at the `AvroBinaryCursor` seam (`private[avro]`, same package) rather than through a + // prism: several rows need a path the public drilling macros refuse to build, and the strictness + // flag is not reachable from the surface at all. + // + // covers: AvroBinaryCursor.scala:558 Field step against a non-record schema, + // AvroBinaryCursor.scala:568 UnionBranch step against a non-union schema, + // AvroBinaryCursor.scala:684 declaredBranches' non-union guard, + // AvroBinaryCursor.scala:675 branchOrdinalOf's end-of-alternatives guard, + // AvroBinaryCursor.scala:572 the `requested < 0` refusal under the LENIENT policy, + // AvroBinaryCursor.scala:126 locateElements' non-array terminal, + // AvroBinaryCursor.scala:125 locateElements' strict-union prefix policy + "AvroBinaryCursor.locate: each refusal reports its OWN AvroFailure, not BinaryParseFailed" >> { + val txBytes = toBinary(transactionRecord(Transaction("t-1", Some(42L))), transactionSchema) + val txNullBytes = toBinary(transactionRecord(Transaction("t-2", None)), transactionSchema) + val personBytes = toBinary(personRecord(Person("Alice", 30)), personSchema) + val basketBytes = + toBinary(basketRecord(Basket("ann", List(Order("tea", 2.5, 1)))), basketSchema) + + def field(n: String) = PathStep.Field(n) + def branch(n: String) = PathStep.UnionBranch(n) + def at(bytes: Array[Byte], schema: Schema, strict: Boolean, steps: PathStep*) = + AvroBinaryCursor.locate(bytes, schema, steps.toArray, strictTerminalUnion = strict) + + // A Field step whose parent schema is a union, not a record. + val notARecord = at(txBytes, transactionSchema, true, field("amount"), field("x")) === + Left(AvroFailure.NotARecord(field("x"))) + + // A UnionBranch step whose parent schema is a plain string — and the diagnostic's branch list + // must degrade to Nil rather than interrogating the non-union for alternatives. + val notAUnion = at(personBytes, personSchema, true, field("name"), branch("string")) === + Left(AvroFailure.UnionResolutionFailed(Nil, branch("string"))) + + // A branch name that is not declared: the ordinal scan must run off the end and report -1 + // rather than indexing past the alternative list. Both policies refuse, and they must refuse + // for THIS reason — the lenient row is the one that separates the `requested < 0` guard from + // the runtime-vs-requested comparison downstream of it. + val unknownBranch = branch("eo.avro.test.NotDeclared") + val declared = Left(AvroFailure.UnionResolutionFailed(List("null", "long"), unknownBranch)) + val unknownStrict = + at(txBytes, transactionSchema, true, field("amount"), unknownBranch) === declared + val unknownLenient = + at(txBytes, transactionSchema, false, field("amount"), unknownBranch) === declared + + // locateElements: a prefix that resolves to a string, not an array. + val notAnArray = AvroBinaryCursor.locateElements( + basketBytes, + basketSchema, + Array(PathStep.Field("owner")), + Array.empty[PathStep], + ) === Left(AvroFailure.NotAnArray(PathStep.Field("owner"))) + + // locateElements resolves its PREFIX strictly: a union prefix whose runtime branch is `null` + // must fail as a branch mismatch, not be tolerated into a "terminal is not an array". + val elementsPrefixStrict = AvroBinaryCursor.locateElements( + txNullBytes, + transactionSchema, + Array(PathStep.Field("amount"), PathStep.UnionBranch("long")), + Array.empty[PathStep], + ) === Left( + AvroFailure.UnionResolutionFailed(List("null", "long"), PathStep.UnionBranch("long")) + ) + + notARecord + .and(notAUnion) + .and(unknownStrict) + .and(unknownLenient) + .and(notAnArray) + .and(elementsPrefixStrict) + } + + // The union-branch POLICY, at both levels of the byte face. Both halves are invisible to every + // other assertion in this module: the payloads still decode, and the decoded values are right. + // + // covers: AvroBinaryCursor.scala:589 the terminal-vs-interior union split, + // AvroPrism.scala:157 the byte-face read's strict terminal-union resolution + "byte-face union policy: an interior step does not anchor the span, a mismatch does not decode" >> { + // An interior union step must NOT anchor the returned span — the span belongs to the step the + // path ends on. Every observable downstream of a mis-anchored span (`getOption`, `.modify`) + // still round-trips, because the span still opens on a decodable value; only `valueSchema` + // gives it away. + val envelope = WireEnvelope("e-1", 7L, Cash(100L), "note") + val envBytes = toBinary(envelopeRecord(envelope), envelopeSchema) + val cashName = summon[AvroCodec[Cash]].schema.getFullName + + val interiorOk = AvroBinaryCursor + .locate( + envBytes, + envelopeSchema, + Array(PathStep.Field("payment"), PathStep.UnionBranch(cashName), PathStep.Field("amount")), + strictTerminalUnion = true, + ) + .map(_.valueSchema.getType) === Right(Schema.Type.LONG) + + // ... and the READ resolves its terminal union strictly. `TwinA` and `TwinB` encode + // identically, so a lenient read does not fail: it DECODES the other branch and hands back a + // well-formed value of the wrong type. The lenient policy is correct for graft/write only. + val twinBytes = toBinaryValue(summon[AvroCodec[TwinA]].encode(TwinA(100L)), twinSchema) + val sameBranch = + AvroPrism.codecPrism[Twin].union[TwinA].getOption(twinBytes) === Some(TwinA(100L)) + val crossBranch = AvroPrism.codecPrism[Twin].union[TwinB].getOption(twinBytes) === None + + interiorOk.and(sameBranch).and(crossBranch) + } + + // covers: AvroTraversal.scala:193 — `.each`'s per-element drilling resolves its prefix to an + // ARRAY or refuses loudly. The public `.each` macro rejects a non-array focus at COMPILE time, + // so this guard is only reachable by constructing the traversal at the internal seam — which + // is exactly what a future drilling entry point would do. + "AvroTraversal: drilling through a non-array prefix throws IllegalArgumentException" >> { + val leaf = new AvroFocus.Leaf[String](Array.empty[PathStep], summon[AvroCodec[String]]) + val bogus = new AvroTraversal[String](Array(PathStep.Field("name")), leaf, personSchema) + + val thrown = + try + bogus.widenSuffixNamed[String]("whatever") + "no throw" + catch case e: IllegalArgumentException => e.getMessage + + thrown must contain("prefix does not point at an array") + } + end AvroBytesSpec diff --git a/avro/src/test/scala/dev/constructive/eo/avro/AvroCodecDecoderReuseSpec.scala b/avro/src/test/scala/dev/constructive/eo/avro/AvroCodecDecoderReuseSpec.scala index 152bad64..ba46623c 100644 --- a/avro/src/test/scala/dev/constructive/eo/avro/AvroCodecDecoderReuseSpec.scala +++ b/avro/src/test/scala/dev/constructive/eo/avro/AvroCodecDecoderReuseSpec.scala @@ -1,12 +1,14 @@ package dev.constructive.eo.avro +import scala.language.implicitConversions + import java.io.ByteArrayInputStream import java.util.concurrent.ConcurrentLinkedQueue import org.apache.avro.Schema import org.apache.avro.generic.{GenericData, GenericDatumReader, GenericRecord, IndexedRecord} import org.apache.avro.io.DecoderFactory import org.scalacheck.Gen -import org.scalacheck.Prop.forAll +import org.scalacheck.Prop.forAllNoShrink import org.specs2.ScalaCheck import org.specs2.mutable.Specification @@ -18,7 +20,9 @@ import org.specs2.mutable.Specification * false` opt-out — must produce a record equal to the OLD fresh-allocation reference decode * (reproduced verbatim in [[freshDecode]]), across writer→reader evolution (field dropped, * added-with-default, reordered, promoted) and union-typed payloads. This is the load-bearing - * correctness guard: a wrong decode in a serde library corrupts every consumer. + * correctness guard: a wrong decode in a serde library corrupts every consumer. The evolution + * shapes are DATA ([[reuseScenarios]]) rather than one example each, so every shape is put + * through both entry points. * 2. '''Thread-safe + non-aliased.''' Concurrent decodes on many threads stay correct, and a * record decoded earlier on a thread is never mutated by a later decode on that thread (fresh * datum, no `Utf8`/bytes aliasing). @@ -93,81 +97,96 @@ class AvroCodecDecoderReuseSpec extends Specification with ScalaCheck: yield AvroSpecFixtures.Transaction(id, amount) // ---- 1. Byte-identical decode vs the fresh-allocation reference ---- + // + // Seven near-identical properties collapsed into one: they differed ONLY in the (generator, + // writer schema, reader schema) triple and in which entry point they called, so the triple is + // now data and both entry points — the thread-local cache and the `threadLocalStorage = false` + // opt-out — are checked on every scenario instead of one each. + + /** One writer→reader shape. `extra` is a scenario-specific sanity check on the reference decode, + * so a scenario can assert that the evolution it names really happened. + */ + final private case class Reuse( + name: String, + bytes: Gen[Array[Byte]], + writer: Schema, + reader: Schema, + extra: IndexedRecord => Boolean, + ) - "non-resolved decodeRecord matches the fresh reference decode (WriterEvent)" >> forAll( - genWriterEvent - ) { e => - val bytes = binaryOf(e) - AvroCodec - .decodeRecord(bytes, writerSchema) - .exists(_ == freshDecode(bytes, writerSchema, writerSchema)) - } - - "non-resolved decodeRecord matches the fresh reference decode (union payload)" >> forAll( - genTransaction - ) { t => - val bytes = binaryOf(t) - AvroCodec - .decodeRecord(bytes, transactionSchema) - .exists(_ == freshDecode(bytes, transactionSchema, transactionSchema)) - } - - "resolved decode matches the fresh reference (WriterEvent→ReaderEvent, field dropped)" >> forAll( - genWriterEvent - ) { e => - val bytes = binaryOf(e) - AvroCodec - .decodeResolvedRecord(bytes, writerSchema, readerEventSchema) - .exists(_ == freshDecode(bytes, writerSchema, readerEventSchema)) - } - - "resolved decode matches the fresh reference (fields reordered, resolved by name)" >> forAll( - genReorderWriter - ) { w => - val bytes = binaryOf(w) - AvroCodec - .decodeResolvedRecord(bytes, reorderWriterSchema, reorderReaderSchema) - .exists(_ == freshDecode(bytes, reorderWriterSchema, reorderReaderSchema)) - } - - "resolved decode matches the fresh reference (int→long promotion)" >> forAll(genPromoteWriter) { - w => - val bytes = binaryOf(w) - AvroCodec - .decodeResolvedRecord(bytes, promoteWriterSchema, promoteReaderSchema) - .exists(_ == freshDecode(bytes, promoteWriterSchema, promoteReaderSchema)) - } + private def always: IndexedRecord => Boolean = _ => true - "resolved decode matches the fresh reference (field added with default)" >> forAll(genId) { id => + private val addWriterBytes: Gen[Array[Byte]] = genId.map { id => val rec = new GenericData.Record(addWriterSchema) rec.put("id", id) - val bytes = AvroSpecFixtures.toBinary(rec, addWriterSchema) - val cached = AvroCodec.decodeResolvedRecord(bytes, addWriterSchema, addReaderSchema) - val reference = freshDecode(bytes, addWriterSchema, addReaderSchema) - // Sanity: the reader default really did materialise, so this is a live added-field case. - val defaultApplied = reference.get(addReaderSchema.getField("note").pos).toString == "n/a" - cached.exists(_ == reference) && defaultApplied + AvroSpecFixtures.toBinary(rec, addWriterSchema) } - // ---- 2. threadLocalStorage = false opt-out (fresh allocation per call) ---- - - "resolved decode with threadLocalStorage = false matches the fresh reference decode" >> forAll( - genWriterEvent - ) { e => - val bytes = binaryOf(e) - AvroBinaryCursor - .records - .read( - bytes, - 0, - bytes.length, - writerSchema, - readerEventSchema, - threadLocalStorage = false, - ) == freshDecode(bytes, writerSchema, readerEventSchema) - } + private val reuseScenarios: List[Reuse] = List( + Reuse("non-resolved", genWriterEvent.map(binaryOf(_)), writerSchema, writerSchema, always), + Reuse( + "non-resolved, union payload", + genTransaction.map(binaryOf(_)), + transactionSchema, + transactionSchema, + always, + ), + Reuse( + "field dropped", + genWriterEvent.map(binaryOf(_)), + writerSchema, + readerEventSchema, + always, + ), + Reuse( + "fields reordered, resolved by name", + genReorderWriter.map(binaryOf(_)), + reorderWriterSchema, + reorderReaderSchema, + always, + ), + Reuse( + "int→long promotion", + genPromoteWriter.map(binaryOf(_)), + promoteWriterSchema, + promoteReaderSchema, + always, + ), + // The reader default must really materialise, else this is not a live added-field case. + Reuse( + "field added with default", + addWriterBytes, + addWriterSchema, + addReaderSchema, + r => r.get(addReaderSchema.getField("note").pos).toString == "n/a", + ), + ) + + private val genScenario: Gen[(Reuse, Array[Byte])] = + Gen.oneOf(reuseScenarios).flatMap(s => s.bytes.map(b => (s, b))) + + "every decode entry point reproduces the fresh-allocation reference, on every evolution shape" >> + forAllNoShrink(genScenario) { (scenario, bytes) => + val reference = freshDecode(bytes, scenario.writer, scenario.reader) + val cached = + if scenario.writer == scenario.reader then AvroCodec.decodeRecord(bytes, scenario.writer) + else AvroCodec.decodeResolvedRecord(bytes, scenario.writer, scenario.reader) + val uncached = AvroBinaryCursor + .records + .read( + bytes, + 0, + bytes.length, + scenario.writer, + scenario.reader, + threadLocalStorage = false, + ) + (cached.exists(_ == reference) && (uncached == reference) && scenario.extra( + reference + )) :| scenario.name + } - // ---- 3. Concurrent + non-aliased ---- + // ---- 2. Concurrent + non-aliased ---- "concurrent decodes across threads are correct and never alias an earlier record" >> { val events = (0 until 8).map(i => WriterEvent(s"id-$i", i)).toVector diff --git a/avro/src/test/scala/dev/constructive/eo/avro/AvroFieldNamingSpec.scala b/avro/src/test/scala/dev/constructive/eo/avro/AvroFieldNamingSpec.scala index 1e3d0c7e..a328e0a1 100644 --- a/avro/src/test/scala/dev/constructive/eo/avro/AvroFieldNamingSpec.scala +++ b/avro/src/test/scala/dev/constructive/eo/avro/AvroFieldNamingSpec.scala @@ -75,65 +75,47 @@ class AvroFieldNamingSpec extends Specification: private val clickBytes = toBinary(clickCodec.encode(click).asInstanceOf[GenericRecord], clickCodec.schema) - "the fixture really uses a divergent (snake_case) schema name" >> { - // Guards the whole spec: if the codec stopped transforming, these fixtures would prove nothing. + // The four byte-face examples (fixture guard, `.field` read, `.field` modify, `selectDynamic`) + // were one code path with one fixture, so they are one example. The leading clause is the + // tripwire: if the codec stopped transforming, nothing below would prove anything. + "byte face: the derived schema really diverges, and .field / selectDynamic both resolve it" >> { + val modified = codecPrism[Click].field(_.clickId).modify(_.toUpperCase)(clickBytes) (clickCodec.schema.getField("clickId") must beNull) .and(clickCodec.schema.getField("click_id") must not(beNull)) + .and(codecPrism[Click].field(_.clickId).getOption(clickBytes) must beSome("abc")) + .and(codecPrism[Click].clickId.getOption(clickBytes) must beSome("abc")) + .and(codecPrism[Click].getOption(modified) must beSome(Click("ABC", 7))) } - "byte face: .field(_.clickId).getOption resolves click_id (issue #35 repro)" >> { - codecPrism[Click].field(_.clickId).getOption(clickBytes) must beSome("abc") - } - - "byte face: .modify round-trips through the snake_case field" >> { - val out = codecPrism[Click].field(_.clickId).modify(_.toUpperCase)(clickBytes) - codecPrism[Click].getOption(out) must beSome(Click("ABC", 7)) - } - - "byte face: selectDynamic sugar resolves the schema name too" >> { - codecPrism[Click].clickId.getOption(clickBytes) must beSome("abc") - } - - "record face: read + modify resolve the schema name" >> { + "record face and a nested descent resolve the same snake_case names" >> { val rec = clickCodec.encode(click).asInstanceOf[GenericRecord] - val readOk = codecPrism[Click].field(_.landingPageId).record.getOption(rec) must beSome(7) - val modified = codecPrism[Click].field(_.landingPageId).record.modifyUnsafe(_ + 1)(rec) - val modOk = clickCodec.decodeEither(modified) must beRight(Click("abc", 8)) - readOk.and(modOk) - } - - "nested record: .field(_.meta).field(_.performanceSourceId) descends under snake_case" >> { - val ev = Event("e1", Meta(42, "hot")) + val bumped = codecPrism[Click].field(_.landingPageId).record.modifyUnsafe(_ + 1)(rec) val evCodec = summon[AvroCodec[Event]] + val ev = Event("e1", Meta(42, "hot")) val bytes = toBinary(evCodec.encode(ev).asInstanceOf[GenericRecord], evCodec.schema) - val optic = codecPrism[Event].field(_.meta).field(_.performanceSourceId) - val readOk = optic.getOption(bytes) must beSome(42) - val writeOk = codecPrism[Event].getOption(optic.modify(_ + 1)(bytes)) must beSome( - Event("e1", Meta(43, "hot")) - ) - readOk.and(writeOk) + val nested = codecPrism[Event].field(_.meta).field(_.performanceSourceId) + (codecPrism[Click].field(_.landingPageId).record.getOption(rec) must beSome(7)) + .and(clickCodec.decodeEither(bumped) must beRight(Click("abc", 8))) + .and(nested.getOption(bytes) must beSome(42)) + .and( + codecPrism[Event].getOption(nested.modify(_ + 1)(bytes)) must beSome( + Event("e1", Meta(43, "hot")) + ) + ) } + // Not folded into the block above: this is the ONE shape in the module where the focus codec's + // own field names diverge from the parent's, so the grouped read is only correct if the bytes + // face projects by RESOLVED PARENT name rather than handing the parent datum to the NT codec. "byte face: .fields(...) grouped read projects by schema name, not NT-codec name" >> { codecPrism[Click].fields(_.landingPageId, _.clickId).getOption(clickBytes) must beSome( (landingPageId = 7, clickId = "abc"): LpAndClick ) } - ".fieldNamed escape hatch navigates by explicit schema name" >> { - codecPrism[Click].fieldNamed[String]("click_id").getOption(clickBytes) must beSome("abc") - } - - // This example used to assert `None` — "a bad explicit .fieldNamed misses, it does not corrupt". - // That PINNED the defect (issue #95): a silent miss on a name the reader schema never carried, - // decided at run time although the schema was available at construction. It is now a refusal. - "a bad explicit .fieldNamed is refused at construction, not missed at runtime" >> { - codecPrism[Click].fieldNamed[String]("no_such_field") must - throwAn[IllegalArgumentException].like { - case e => - (e.getMessage must contain("click_id, landing_page_id")) - .and(e.getMessage must contain("no field of that name")) - } - } + // `.fieldNamed`'s two verdicts — a name the schema HAS builds and reads, a name it LACKS is + // refused at construction (issue #95, which used to be a silent runtime miss) — are pinned on a + // divergent-name fixture by `AvroNominalResolutionSpec`, with strictly more of the refusal + // message asserted than the duplicate that used to live here. end AvroFieldNamingSpec diff --git a/avro/src/test/scala/dev/constructive/eo/avro/AvroNominalDoctrineSpec.scala b/avro/src/test/scala/dev/constructive/eo/avro/AvroNominalDoctrineSpec.scala new file mode 100644 index 00000000..982c996b --- /dev/null +++ b/avro/src/test/scala/dev/constructive/eo/avro/AvroNominalDoctrineSpec.scala @@ -0,0 +1,170 @@ +package dev.constructive.eo.avro + +import scala.util.control.NonFatal + +import java.util.ArrayList +import org.apache.avro.Schema +import org.specs2.mutable.Specification + +/** The ALL-OR-NOTHING nominal rung, stated as a declarative oracle and checked EXHAUSTIVELY. + * + * `AvroWalk.fieldNameAt` resolves `.field(_.x)` by NAME when the whole case-field list maps to + * DISTINCT schema fields (exact, else uniquely up to `_` / `-` / `.` and case), and abstains to + * position otherwise. Four clauses — totality, injectivity, exact-beats-normalised, ambiguity ⇒ + * abstain — and before this spec only totality was pinned. Issue #104 recorded the cost: an + * experimental alias rung silently re-aimed a case the rung gets right while every avro test + * stayed green. A MIS-TARGETED optic satisfies get-put, put-get, put-put and modify fusion + * perfectly, so no optic law family can see this; the contract is a naming contract on a + * resolution function, which is why this is a property and not a `checkAll` registration. + * + * The oracle below is written from the DOCTRINE, deliberately not as a frozen copy of the + * implementation: parity against a frozen copy of the code cannot detect a wrong doctrine, which + * is exactly the #104 failure mode. + * + * That is the division of labour with [[NominalResolutionOracle]] and + * `vulcan.NominalResolutionParitySpec`, which #107 added alongside its index rewrite. That oracle + * is the OLD ALGORITHM frozen verbatim, and its spec asks "does the cached-index rung still agree + * with the scan it replaced" — a refactor gate, which cannot fire if the algorithm was already + * resolving to the wrong field. This one asks "does the rung agree with what the rung is FOR", and + * is written without reading the implementation. Both are wanted: #107's rewrite deleted + * `sameFieldName` and the fuzzy per-field scan outright, and the fact that this doctrine oracle + * still agrees cell for cell is the evidence that the rewrite preserved the CONTRACT and not just + * the code path. + * + * The corpus is an exhaustive enumeration rather than a `Gen`: the discriminating cells are + * threshold cells (a duplicate resolution landing on slot 0, a fuzzy ambiguity whose FIRST hit is + * at index 0 versus at index >= 1, `declIdx` exactly equal to the schema's field count) that a + * default generator would essentially never produce. + * + * Subsumes the deleted `"AvroWalk.fieldNameAt: non-record parent and out-of-range declIdx both + * throw loudly"` example, which tested `declIdx = 99` against a 2-field schema — a cell where the + * `>=` guard and a `>` mutant agree. + * + * covers (against post-#107 `AvroWalk`, which replaced the fuzzy per-field scan with + * `normalisedName` + a cached `normalisedNameIndex`): `if exact != null` in `totalNominalIndex` + * (exact-beats-normalised fast path), `j < 0` and the `&&` in its `seen` injectivity scan, the + * `idx < 0` verdict on a `null` or `Ambiguous` index hit (ambiguity => abstain), the char + * classification and `Character.toLowerCase` in `normalisedName`, the four disjuncts of the + * all-or-nothing precondition, and `declIdx >= fields.size` in `fieldNameAt` at the exact + * boundary. Line numbers are deliberately not quoted: the pre-#107 ones this spec was written + * against are already gone. + */ +class AvroNominalDoctrineSpec extends Specification: + + // ---- Corpus ------------------------------------------------------------- + // + // Every pool entry is BOTH a legal Avro field name (`[A-Za-z_][A-Za-z0-9_]*`) and a legal Scala + // identifier, and ASCII-only so `toLowerCase` agrees with the per-char `Character.toLowerCase` + // the implementation uses. `-` and `.` are therefore absent by construction: those two `skip` + // arms are unreachable from any legal schema paired with any legal Scala identifier. + private val namePool: List[String] = List("a", "A", "_a", "b", "a_b", "aB") + + /** The sub-pool used for arity-3 case lists — keeps the corpus at ~10^2 case shapes. */ + private val smallPool: List[String] = List("a", "A", "_a", "b") + + private def seqsWithRep(pool: List[String], len: Int): List[List[String]] = + if len <= 0 then List(Nil) + else + for + h <- pool + t <- seqsWithRep(pool, len - 1) + yield h :: t + + private def injectiveSeqs(pool: List[String], len: Int): List[List[String]] = + if len <= 0 then List(Nil) + else + for + h <- pool + t <- injectiveSeqs(pool.filterNot(_ == h), len - 1) + yield h :: t + + private def recordOf(names: List[String], tag: Int): Schema = + val fields = new ArrayList[Schema.Field]() + names.foreach(n => + fields.add(new Schema.Field(n, Schema.create(Schema.Type.STRING), null, null)) + ) + Schema.createRecord(s"R$tag", null, "eo.avro.test", false, fields) + + /** All injective ordered field lists of length 1..3 — Avro forbids duplicate field names. */ + private val schemas: List[(List[String], Schema)] = + val lists = (1 to 3).toList.flatMap(injectiveSeqs(namePool, _)) + lists.zipWithIndex.map((ns, i) => (ns, recordOf(ns, i))) + + /** Case-field lists WITH repetition — two identical case names is the degenerate injectivity + * failure — plus the empty list (a NamedTuple parent, which has no case fields). + */ + private val caseCorpus: List[List[String]] = + Nil :: (seqsWithRep(namePool, 1) ++ seqsWithRep(namePool, 2) ++ seqsWithRep(smallPool, 3)) + + // ---- The declarative oracle -------------------------------------------- + + private def norm(s: String): String = + s.filter(c => c != '_' && c != '-' && c != '.').toLowerCase + + /** Exact name wins; otherwise the UNIQUE normalised match; otherwise no signal. */ + private def oracleIdx(fields: List[String], name: String): Int = + val exact = fields.indexOf(name) + if exact >= 0 then exact + else + val hits = fields.indices.filter(i => norm(fields(i)) == norm(name)) + if hits.sizeIs == 1 then hits.head else -1 + + /** What the DOCTRINE says `fieldNameAt` returns, as either a name or a `throw:` tag. */ + private def oracle(fields: List[String], caseNames: List[String], declIdx: Int): String = + def positional: String = + if declIdx >= fields.size then "throw:IllegalArgumentException" + else if declIdx < 0 then "throw:IndexOutOfBoundsException" + else fields(declIdx) + val arity = caseNames.size + if arity == 0 || declIdx < 0 || declIdx >= arity || arity > fields.size then positional + else + val ix = caseNames.map(oracleIdx(fields, _)) + val total = ix.forall(_ >= 0) + val injective = ix.distinct.sizeIs == ix.size + if total && injective then fields(ix(declIdx)) else positional + + private def observed(schema: Schema, caseNames: List[String], declIdx: Int): String = + try + AvroWalk.fieldNameAt( + schema, + caseNames.lift(declIdx).getOrElse("?"), + declIdx, + caseNames, + "test", + ) + catch case NonFatal(e) => "throw:" + e.getClass.getSimpleName + + private val disagreements: List[String] = + for + (fieldNames, schema) <- schemas + caseNames <- caseCorpus + declIdx <- -1 to (caseNames.size + 1) + want = oracle(fieldNames, caseNames, declIdx) + got = observed(schema, caseNames, declIdx) + if want != got + yield s"schema=$fieldNames case=$caseNames declIdx=$declIdx want=$want got=$got" + + /** The corpus says nothing about non-record parents — they are the one input shape the oracle + * does not model, because resolution never begins. Checked here rather than in a block of its + * own: it is the same function's same contract, and the guard is on the parent's schema TYPE, so + * it must hold at every declIdx. + */ + private val nonRecordOutcomes: List[String] = + val stringSchema = Schema.create(Schema.Type.STRING) + val parents = + List(stringSchema, Schema.createArray(stringSchema), Schema.createMap(stringSchema)) + for + parent <- parents + declIdx <- List(0, 1) + yield observed(parent, List("a", "b"), declIdx) + + "AvroWalk.fieldNameAt agrees with the nominal-resolution doctrine over the whole corpus" >> { + val shown: List[String] = disagreements.take(10) + val cells: Int = schemas.size * caseCorpus.size + (shown === List.empty[String]) + .and(disagreements.size === 0) + .and(cells must be_>(10000)) + .and(nonRecordOutcomes === List.fill(6)("throw:IllegalArgumentException")) + } + +end AvroNominalDoctrineSpec diff --git a/avro/src/test/scala/dev/constructive/eo/avro/AvroSpecFixtures.scala b/avro/src/test/scala/dev/constructive/eo/avro/AvroSpecFixtures.scala index 16f5d337..1cf4973a 100644 --- a/avro/src/test/scala/dev/constructive/eo/avro/AvroSpecFixtures.scala +++ b/avro/src/test/scala/dev/constructive/eo/avro/AvroSpecFixtures.scala @@ -103,6 +103,33 @@ object AvroSpecFixtures: */ lazy val transactionSchema: Schema = summon[AvroCodec[Transaction]].schema + /** Two union branches with IDENTICAL field shapes and different names. `TwinA(100)` and + * `TwinB(100)` encode to the same bytes, so a union walk that TOLERATES a branch mismatch + * decodes the wrong branch SUCCESSFULLY instead of refusing — silent wrong data rather than a + * failure. Every other union fixture here mismatches into a decode error, which any `isLeft` / + * `=== None` assertion cannot tell apart from a correct refusal. + */ + sealed trait Twin + + object Twin: + + given AvroEncoder[Twin] = AvroEncoder.derived + given AvroDecoder[Twin] = AvroDecoder.derived + given AvroSchemaFor[Twin] = AvroSchemaFor.derived + + given AvroEncoder[TwinA] = AvroEncoder.derived + given AvroDecoder[TwinA] = AvroDecoder.derived + given AvroSchemaFor[TwinA] = AvroSchemaFor.derived + + given AvroEncoder[TwinB] = AvroEncoder.derived + given AvroDecoder[TwinB] = AvroDecoder.derived + given AvroSchemaFor[TwinB] = AvroSchemaFor.derived + + case class TwinA(v: Long) extends Twin + case class TwinB(v: Long) extends Twin + + lazy val twinSchema: Schema = summon[AvroCodec[Twin]].schema + /** Sealed-trait sum used by the `.union[Branch]` happy-path tests. Mirrors the probe ADT — two * record-shaped subclasses, deliberately top-level so the kindlings macros aren't tripped by * outer accessors. diff --git a/avro/src/test/scala/dev/constructive/eo/avro/AvroWalkSpec.scala b/avro/src/test/scala/dev/constructive/eo/avro/AvroWalkSpec.scala index 6f99e80a..829eeef9 100644 --- a/avro/src/test/scala/dev/constructive/eo/avro/AvroWalkSpec.scala +++ b/avro/src/test/scala/dev/constructive/eo/avro/AvroWalkSpec.scala @@ -42,6 +42,14 @@ class AvroWalkSpec extends Specification: fields.add(new Schema.Field("amount", unionSchema, null, null)) Schema.createRecord("MaybeLong", null, "eo.avro.test", false, fields) + /** Schema for `record Outer { MaybeLong inner; }` — a union two records deep, which is what + * separates the branch-list recovery's `pIdx >= 0` cursor guard from a `pIdx == 0` one. + */ + private val outerSchema: Schema = + val fields = new ArrayList[Schema.Field]() + fields.add(new Schema.Field("inner", maybeLongSchema, null, null)) + Schema.createRecord("Outer", null, "eo.avro.test", false, fields) + private val colorSchema: Schema = Schema.createEnum("Color", null, "eo.avro.test", Arrays.asList("RED", "GREEN", "BLUE")) @@ -129,23 +137,57 @@ class AvroWalkSpec extends Specification: negOne.and(atSize) } - // covers: walk into a map entry by key returns the entry value - "Map walk: by string key" >> { - val tags = new LinkedHashMap[String, String]() - tags.put("env", "prod") - tags.put("region", "us") - val record = buildRecord(taggedMapSchema)("tags" -> tags) - - AvroWalk.walkPath(record, Array(PathStep.Field("tags"), PathStep.Field("env"))) match - case Right((cur, _)) => cur.toString === "prod" - case other => - org.specs2.execute.Failure(s"expected Right, got $other"): org.specs2.execute.Result + /** A `map` record put through the binary codec — which is the ONLY way to get the key + * shape production code actually sees. `GenericDatumReader` decodes map keys as + * [[org.apache.avro.util.Utf8]], so `asMap.get(name: String)` misses every entry and only the + * `direct == null` Utf8 retry finds them. A hand-built `LinkedHashMap[String, _]` cannot + * exercise that retry at all. + */ + private def tagsRecord(keys: List[String]): IndexedRecord = + val m = new LinkedHashMap[String, String]() + keys.foreach(k => m.put(k, s"v-$k")) + buildRecord(taggedMapSchema)("tags" -> m) + + private def decodedTags(keys: List[String]): IndexedRecord = + fromBinaryValue(toBinaryValue(tagsRecord(keys), taggedMapSchema), taggedMapSchema) + .asInstanceOf[IndexedRecord] + + private def tagAt(rec: IndexedRecord, key: String): Either[AvroFailure, String] = + AvroWalk + .walkPath(rec, Array(PathStep.Field("tags"), PathStep.Field(key))) + .map((c, _) => String.valueOf(c)) + + // covers: walk a map entry by key on a hand-built String-keyed map, + // walk a map entry by key on the Utf8-keyed map a DECODE produces (the Utf8 retry), + // an absent key surfaces PathMissing at map sizes 0 / 1 / 2 / 5 — both present and absent + // probes occur at every size, so neither arm of the retry can be dropped silently + "Map walk: by string key, over hand-built and decoded (Utf8-keyed) maps at sizes 0/1/2/5" >> { + val handBuilt = tagAt(tagsRecord(List("env", "region")), "env") === Right("v-env") + + val keyPool = List("env", "region", "zone", "tier", "shard", "cell") + val disagreements = + for + size <- List(0, 1, 2, 5) + present = keyPool.take(size) + decoded = decodedTags(present) + probe <- keyPool + want = + if present.contains(probe) then Right(s"v-$probe") + else Left(AvroFailure.PathMissing(PathStep.Field(probe))) + got = tagAt(decoded, probe) + if want != got + yield s"size=$size probe=$probe want=$want got=$got" + + handBuilt.and(disagreements === List.empty[String]) } // covers: union branch matching resolves "long" alt to its long value, // mismatched union branch ("string" on long) surfaces UnionResolutionFailed, + // a null payload against the "null" branch resolves to a null focus, + // the diagnostic's declared-branch list at depth 2 (parents cursor > 0), + // the diagnostic degrades to Nil when the recovered Field step names a non-union, // terminalOf returns Field("") for empty, last step otherwise - "Union walk + terminalOf: long-alt resolution, branch mismatch, terminalOf endpoints" >> { + "Union walk + terminalOf: long-alt resolution, branch mismatch, null branch, diagnostics, terminalOf endpoints" >> { val record = buildRecord(maybeLongSchema)("amount" -> java.lang.Long.valueOf(42L)) val longOk = AvroWalk.walkPath( @@ -168,11 +210,46 @@ class AvroWalkSpec extends Specification: .execute .Failure(s"expected UnionResolutionFailed, got $other"): org.specs2.execute.Result + // A null payload matches the "null" branch and nothing else — the one branch name for which a + // null focus is the CORRECT answer rather than a resolution failure. + val nullBranchOk = AvroWalk + .walkPath( + buildRecord(maybeLongSchema)("amount" -> null.asInstanceOf[Object]), + Array(PathStep.Field("amount"), PathStep.UnionBranch("null")), + ) + .map((cur, _) => cur == null) === Right(true) + + // The diagnostic's PAYLOAD, not just its constructor: the declared alternative list is + // recovered by walking back to the nearest Field step. A union two records deep leaves the + // parents cursor at 1, so a cursor guard that only accepts 0 silently reports "no branches". + val deepBranchesOk = AvroWalk.walkPath( + buildRecord(outerSchema)("inner" -> record), + Array(PathStep.Field("inner"), PathStep.Field("amount"), PathStep.UnionBranch("string")), + ) === Left( + AvroFailure.UnionResolutionFailed(List("null", "long"), PathStep.UnionBranch("string")) + ) + + // ... and when the recovered Field step names something that is NOT a union (an array reached + // through an Index step), the diagnostic degrades to an EMPTY list. Asking a non-union schema + // for its alternatives throws an AvroRuntimeException straight out of the walk. + val people2: GenericData.Array[GenericRecord] = + buildArray(personSchema, Vector(personRecord(Person("Alice", 30)))) + val nonUnionParentOk = AvroWalk.walkPath( + buildRecord(wrapperSchema)("people" -> people2), + Array(PathStep.Field("people"), PathStep.Index(0), PathStep.UnionBranch("string")), + ) === Left(AvroFailure.UnionResolutionFailed(Nil, PathStep.UnionBranch("string"))) + val terminalEmpty = AvroWalk.terminalOf(Array.empty[PathStep]) === PathStep.Field("") val terminalLast = AvroWalk.terminalOf(Array(PathStep.Field("a"), PathStep.Index(2))) === PathStep.Index(2) - longOk.and(mismatchOk).and(terminalEmpty).and(terminalLast) + longOk + .and(mismatchOk) + .and(nullBranchOk) + .and(deepBranchesOk) + .and(nonUnionParentOk) + .and(terminalEmpty) + .and(terminalLast) } // covers: enum value resolves its full-name union branch, @@ -341,23 +418,8 @@ class AvroWalkSpec extends Specification: "literalName" } - // covers: fieldNameAt on a non-record schema throws IllegalArgumentException, - // fieldNameAt with declIdx past the field count throws IllegalArgumentException - "AvroWalk.fieldNameAt: non-record parent and out-of-range declIdx both throw loudly" >> { - val stringSchema = Schema.create(Schema.Type.STRING) - val nonRecordThrows = - try - AvroWalk.fieldNameAt(stringSchema, "x", 0, Nil, "test") - false - catch case _: IllegalArgumentException => true - - val outOfRangeThrows = - try - AvroWalk.fieldNameAt(personSchema, "x", 99, Nil, "test") - false - catch case _: IllegalArgumentException => true - - (nonRecordThrows === true).and(outOfRangeThrows === true) - } + // `fieldNameAt`'s own arms — non-record parent, every `declIdx` boundary including the + // `declIdx == fields.size` cell this file's old example missed — now live in + // `AvroNominalDoctrineSpec`, which checks them exhaustively against a declarative oracle. end AvroWalkSpec diff --git a/avro/src/test/scala/dev/constructive/eo/avro/AvroWriteCorrectnessSpec.scala b/avro/src/test/scala/dev/constructive/eo/avro/AvroWriteCorrectnessSpec.scala index 1dada321..e3b3b4a9 100644 --- a/avro/src/test/scala/dev/constructive/eo/avro/AvroWriteCorrectnessSpec.scala +++ b/avro/src/test/scala/dev/constructive/eo/avro/AvroWriteCorrectnessSpec.scala @@ -93,14 +93,23 @@ class AvroWriteCorrectnessSpec extends Specification with ScalaCheck: codecPrism[FullName].getOption(out) === Some(FullName("Jane", "Smith")) } - // covers: per-element .each.fields byte write (was: every element silently kept) - "byte-face .each.fields: per-element multi-field write applies to every element" >> { + // covers: per-element .each.fields byte write (was: every element silently kept), on BOTH array + // framings — canonical goes through the splice path, blocked additionally through the reframe + // path, and the by-name field overlay has to survive either. Two examples that differed only + // in how the same basket was framed. + "byte-face .each.fields: per-element multi-field write, canonical and blocked framing" >> { val basket = Basket("ann", List(Order("tea", 2.5, 1), Order("mate", 4.0, 2))) val bytes = toBinary(basketRecord(basket), basketSchema) + val blocked = toBlockedBinary(basketRecord(basket), basketSchema) val T = codecPrism[Basket].items.each.fields(_.name, _.price) - val out = T.modify(nt => (name = nt.name.toUpperCase, price = nt.price * 2))(bytes) - codecPrism[Basket].getOption(out) === Some( + val doubled = T.modify(nt => (name = nt.name.toUpperCase, price = nt.price * 2))(bytes) + val bumped = T.modify(nt => (name = nt.name.toUpperCase, price = nt.price + 1.0))(blocked) + (codecPrism[Basket].getOption(doubled) === Some( Basket("ann", List(Order("TEA", 5.0, 1), Order("MATE", 8.0, 2))) + )).and( + codecPrism[Basket].getOption(bumped) === Some( + Basket("ann", List(Order("TEA", 3.5, 1), Order("MATE", 5.0, 2))) + ) ) } @@ -201,20 +210,64 @@ class AvroWriteCorrectnessSpec extends Specification with ScalaCheck: totalOk.and(namesOk) } - // ---- Post-re-review combination axes (reframe × overlay × union) ----- + /** A CANONICAL (positive-count) array framing that is nonetheless MULTI-BLOCK — spec-legal, and + * the only shape that separates an in-place splice from a whole-region re-frame. Every other + * fixture here is single-block, where a re-frame reproduces the input byte-for-byte and the two + * write paths are indistinguishable. Collapsing two blocks into one costs a count byte, so the + * re-frame is visible in the payload LENGTH while the decoded value is unchanged. + */ + private def multiBlockBasket(owner: String, items: List[Order]): Array[Byte] = + val out = new ByteArrayOutputStream() + def put(bs: Array[Byte]): Unit = out.write(bs, 0, bs.length) + def block(bs: List[Array[Byte]]): Unit = + put(AvroBinaryCursor.zigZagLong(bs.length.toLong)) + bs.foreach(put) + put(AvroBinaryCursor.writeDatum(owner, basketSchema.getField("owner").schema)) + val (head, last) = items + .map(o => AvroBinaryCursor.writeDatum(summon[AvroCodec[Order]].encode(o), orderSchema)) + .splitAt(items.length - 1) + block(head) + block(last) + put(AvroBinaryCursor.zigZagLong(0L)) + out.toByteArray - // covers: .each.fields on a NON-canonical array — the reframe path AND the by-name field - // overlay together (each individually pinned above; this exercises the combination) - "byte-face .each.fields on blocked framing: reframe + overlay both apply" >> { - val basket = Basket("ann", List(Order("tea", 2.5, 1), Order("mate", 4.0, 2))) - val blocked = toBlockedBinary(basketRecord(basket), basketSchema) - val T = codecPrism[Basket].items.each.fields(_.name, _.price) - val out = T.modify(nt => (name = nt.name.toUpperCase, price = nt.price + 1.0))(blocked) - codecPrism[Basket].getOption(out) === Some( - Basket("ann", List(Order("TEA", 3.5, 1), Order("MATE", 5.0, 2))) - ) + // covers: AvroTraversal.scala:119 the no-op guard (`active.isEmpty || arity mismatch`), + // AvroTraversal.scala:125 the canonical-splice vs re-frame write split, + // AvroBinaryCursor.scala:143 the initial canonical flag threaded through the block walk + // + // These are BYTE-level assertions on purpose. Every write below decodes to the right value + // under every mutation of the three guards above — the Optional-law property block at the foot + // of this file quantifies over decoded values and cannot see any of it. What changes is the + // FRAMING: a payload that arrives blocked and leaves canonical, or arrives multi-block and + // leaves single-block, is a rewrite of bytes the caller never asked us to touch. + "array FRAMING survives a write: a no-op touches no byte, a splice touches only the focus" >> { + // (a) NOTHING is focused (every element is on the other union branch) under NON-canonical + // framing. The write must be the identity on the bytes, not a canonical re-frame. + val cardsOnly = Ledger("ann", List(Card("4111"), Card("4222"))) + val blocked = toBlockedBinary(ledgerRecord(cardsOnly), ledgerSchema) + val cashT = codecPrism[Ledger].field(_.entries).each.union[Cash] + val noFociOk = + (Arrays.equals(blocked, toBinary(ledgerRecord(cardsOnly), ledgerSchema)) === false) + .and(cashT.foldMap(List(_))(blocked) === Nil) + .and(Arrays.equals(cashT.modify(c => Cash(c.amount + 1L))(blocked), blocked) === true) + + // (b) CANONICAL multi-block framing plus a length-preserving edit: the splice happens in + // place and the block structure is left exactly as it arrived. + val items = List(Order("tea", 2.5, 1), Order("mate", 4.0, 2), Order("cola", 1.0, 3)) + val multi = multiBlockBasket("ann", items) + val framingOk = + (Arrays.equals(multi, toBinary(basketRecord(Basket("ann", items)), basketSchema)) === false) + .and(codecPrism[Basket].getOption(multi) === Some(Basket("ann", items))) + + val spliced = codecPrism[Basket].items.each.name.modify(_.toUpperCase)(multi) + val upper = multiBlockBasket("ann", items.map(o => o.copy(name = o.name.toUpperCase))) + val spliceOk = (spliced.length === multi.length).and(Arrays.equals(spliced, upper) === true) + + noFociOk.and(framingOk).and(spliceOk) } + // ---- Post-re-review combination axes (reframe × overlay × union) ----- + // covers: Fields focus UNDER a .union step — the union index re-synthesis (from branchOrdinal) // and the by-name parent overlay interacting on both faces "byte-face .union[Branch].fields: index re-synthesis + overlay, siblings survive" >> { @@ -236,51 +289,33 @@ class AvroWriteCorrectnessSpec extends Specification with ScalaCheck: } // covers: per-element .each.union[Branch] — narrows every element to one union alternative, - // folding the elements that ARE that branch and leaving the others untouched. Exercises the - // element-level branch-index re-synthesis in reframeArray under blocked framing (canonical - // framing goes through spliceAll's index re-synth, already covered by graft tests). - "byte-face .each.union[Branch] on blocked framing: per-element branch focus reframes" >> { + // folding the elements that ARE that branch and leaving the others untouched. ONE fixture and + // ONE optic across all three carriers, because the three used to be three examples that + // differed only in which carrier the same Ledger was handed to: blocked framing exercises the + // element-level branch-index re-synthesis in reframeArray, canonical framing exercises + // spliceAll's, and the record face the parsed Ior walk. + "the .each.union[Branch] per-element focus: blocked framing, canonical framing, record face" >> { val ledger = Ledger("ann", List(Cash(100L), Card("4111"), Cash(250L))) - val blocked = toBlockedBinary(ledgerRecord(ledger), ledgerSchema) + val rec = ledgerRecord(ledger) + val blocked = toBlockedBinary(rec, ledgerSchema) + val bytes = toBinary(rec, ledgerSchema) val cashT = codecPrism[Ledger].field(_.entries).each.union[Cash] + val recordOut = cashT.record.modifyUnsafe(c => Cash(c.amount + 5L))(rec) match + case out: IndexedRecord => + codecPrism[Ledger].getOption(toBinary(out.asInstanceOf[GenericRecord], ledgerSchema)) // Read: only the Cash-branch elements fold in, in order. - val readOk = cashT.foldMap(List(_))(blocked) === List(Cash(100L), Cash(250L)) - // Write: Cash elements bumped, the Card element rides through untouched. - val out = cashT.modify(c => Cash(c.amount + 1L))(blocked) - val writeOk = codecPrism[Ledger].getOption(out) === Some( - Ledger("ann", List(Cash(101L), Card("4111"), Cash(251L))) - ) - readOk.and(writeOk) - } - - // covers: .each.union[Branch] on canonical framing too — the spliceAll index re-synth path for - // per-element union foci - "byte-face .each.union[Branch] on canonical framing: per-element branch focus splices" >> { - val ledger = Ledger("ann", List(Cash(100L), Card("4111"), Cash(250L))) - val bytes = toBinary(ledgerRecord(ledger), ledgerSchema) - val out = codecPrism[Ledger] - .field(_.entries) - .each - .union[Cash] - .modify(c => Cash(c.amount * 2))( - bytes + (cashT.foldMap(List(_))(blocked) === List(Cash(100L), Cash(250L))) + // Write, blocked: Cash elements bumped, the Card element rides through untouched. + .and( + codecPrism[Ledger].getOption(cashT.modify(c => Cash(c.amount + 1L))(blocked)) === + Some(Ledger("ann", List(Cash(101L), Card("4111"), Cash(251L)))) ) - codecPrism[Ledger].getOption(out) === Some( - Ledger("ann", List(Cash(200L), Card("4111"), Cash(500L))) - ) - } - - // covers: .each.union[Branch] on the RECORD face — the same per-element branch narrowing - // through the parsed walk (Ior surface), Cash elements modified, Card untouched - ".each.union[Branch] record face: per-element branch modify, non-branch elements ride through" >> { - val ledger = Ledger("ann", List(Cash(100L), Card("4111"), Cash(250L))) - val rec = ledgerRecord(ledger) - val T = codecPrism[Ledger].field(_.entries).each.union[Cash].record - T.modifyUnsafe(c => Cash(c.amount + 5L))(rec) match - case out: IndexedRecord => - codecPrism[Ledger].getOption(toBinary(out.asInstanceOf[GenericRecord], ledgerSchema)) === - Some(Ledger("ann", List(Cash(105L), Card("4111"), Cash(255L)))) + .and( + codecPrism[Ledger].getOption(cashT.modify(c => Cash(c.amount * 2))(bytes)) === + Some(Ledger("ann", List(Cash(200L), Card("4111"), Cash(500L)))) + ) + .and(recordOut === Some(Ledger("ann", List(Cash(105L), Card("4111"), Cash(255L))))) } // covers: Long.MinValue block count (negates to itself) is a structured parse failure, not a diff --git a/avro/src/test/scala/dev/constructive/eo/avro/ConfluentReaderSpec.scala b/avro/src/test/scala/dev/constructive/eo/avro/ConfluentReaderSpec.scala index 1fea168f..22d27c63 100644 --- a/avro/src/test/scala/dev/constructive/eo/avro/ConfluentReaderSpec.scala +++ b/avro/src/test/scala/dev/constructive/eo/avro/ConfluentReaderSpec.scala @@ -110,44 +110,26 @@ class ConfluentReaderSpec extends Specification: // covers: strict frame contract — an unframed payload raises NotConfluentFramed rather than // silently direct-decoding (which could accidentally succeed on corrupt bytes and yield // garbage); the hook is never consulted. Mixed-topic callers opt into their own fallback by - // catching this and decoding directly (AvroCodec.decodeValue). - "reader: unframed bytes raise NotConfluentFramed in F, hook never consulted" >> { + // catching this and decoding directly (AvroCodec.decodeValue). `recordReader` shares the + // contract verbatim, so the two entry points are one example over one predicate — and the + // two payload shapes (valid-but-unframed Avro, and short garbage) are swapped between them. + private def framedRefusal(r: Res[Any]): Boolean = r match + case Left(e: AvroFailureException) => + e.failure match + case AvroFailure.NotConfluentFramed(_) => true + case _ => false + case _ => false + + "reader / recordReader: unframed bytes raise NotConfluentFramed in F, hook never consulted" >> { val boom: Int => Res[Schema] = _ => Left(new AssertionError("hook must not be called")) val raw = AvroSpecFixtures.toBinaryValue( summon[AvroCodec[ReaderEvent]].encode(ReaderEvent("e-3")), readerSchema, ) - ConfluentWire.reader[Res, ReaderEvent](boom)(raw) match - case Left(e: AvroFailureException) => - (e.failure match - case AvroFailure.NotConfluentFramed(_) => true - case _ => false - ) === true - case other => - org - .specs2 - .execute - .Failure( - s"expected Left(AvroFailureException(NotConfluentFramed)), got $other" - ): org.specs2.execute.Result - } - - // covers: recordReader shares the strict frame contract - "recordReader: unframed bytes raise NotConfluentFramed in F" >> { - val boom: Int => Res[Schema] = _ => Left(new AssertionError("hook must not be called")) - ConfluentWire.recordReader[Res](boom, readerSchema)(Array[Byte](1, 2, 3)) match - case Left(e: AvroFailureException) => - (e.failure match - case AvroFailure.NotConfluentFramed(_) => true - case _ => false - ) === true - case other => - org - .specs2 - .execute - .Failure( - s"expected Left(AvroFailureException(NotConfluentFramed)), got $other" - ): org.specs2.execute.Result + val viaReader = ConfluentWire.reader[Res, ReaderEvent](boom)(raw) + val viaRecord = ConfluentWire.recordReader[Res](boom, readerSchema)(Array[Byte](1, 2, 3)) + (framedRefusal(viaReader).aka(s"reader gave $viaReader") must beTrue) + .and(framedRefusal(viaRecord).aka(s"recordReader gave $viaRecord") must beTrue) } "reader: incompatible reader (field absent from writer, no default) raises ResolveFailed in F" >> { @@ -197,35 +179,25 @@ class ConfluentReaderSpec extends Specification: readOk.and(writeOk) } - "reader: fields MOVED (reordered writer→reader) resolve by name, not position" >> { - val ws = summon[AvroCodec[ReorderWriter]].schema + /** Frame `a` under id 7 and hand `reader` a registry that knows only that id. */ + private def readFramed[A, B](a: A)(using wc: AvroCodec[A], rc: AvroCodec[B]): Res[B] = val reg: Int => Res[Schema] = - case 7 => Right(ws) + case 7 => Right(wc.schema) case id => Left(new NoSuchElementException(s"no schema for id $id")) - val bytes = ConfluentWire.attach( - 7, - AvroSpecFixtures.toBinaryValue( - summon[AvroCodec[ReorderWriter]].encode(ReorderWriter("x", 42, true)), - ws, - ), + ConfluentWire.reader[Res, B](reg)( + ConfluentWire.attach(7, AvroSpecFixtures.toBinaryValue(wc.encode(a), wc.schema)) ) - // gamma/alpha/beta land on the right reader fields despite the flipped declaration order. - ConfluentWire.reader[Res, ReorderReader](reg)(bytes) === Right(ReorderReader(true, "x", 42)) - } - "reader: a field whose TYPE CHANGED (int writer → long reader) is promoted, siblings intact" >> { - val ws = summon[AvroCodec[PromoteWriter]].schema - val reg: Int => Res[Schema] = - case 7 => Right(ws) - case id => Left(new NoSuchElementException(s"no schema for id $id")) - val bytes = ConfluentWire.attach( - 7, - AvroSpecFixtures.toBinaryValue( - summon[AvroCodec[PromoteWriter]].encode(PromoteWriter("n", 42)), - ws, - ), - ) - ConfluentWire.reader[Res, PromoteReader](reg)(bytes) === Right(PromoteReader("n", 42L)) + // Two evolution shapes, one entry point and one registry rig: `reader` resolves by NAME, so a + // flipped declaration order must not move a value, and an int writer field must widen into a + // long reader field with its siblings intact. + "reader: MOVED fields resolve by name and a PROMOTED field widens, siblings intact" >> { + (readFramed[ReorderWriter, ReorderReader](ReorderWriter("x", 42, true)) === + Right(ReorderReader(true, "x", 42))) + .and( + readFramed[PromoteWriter, PromoteReader](PromoteWriter("n", 42)) === + Right(PromoteReader("n", 42L)) + ) } "resolving Prism: read across a MOVED-field schema, modify, re-frame back to the reader shape" >> { diff --git a/avro/src/test/scala/dev/constructive/eo/avro/circe/AvroJsonSpec.scala b/avro/src/test/scala/dev/constructive/eo/avro/circe/AvroJsonSpec.scala index ad6ab3a0..fef8d048 100644 --- a/avro/src/test/scala/dev/constructive/eo/avro/circe/AvroJsonSpec.scala +++ b/avro/src/test/scala/dev/constructive/eo/avro/circe/AvroJsonSpec.scala @@ -207,33 +207,39 @@ class AvroJsonSpec extends Specification with ScalaCheck: .and(envPrism.getOption(Json.fromString("not an object")) === None) } - // covers: enum symbol / fixed length / bytes shape parse both ways - "record prism: enum, bytes and fixed round-trip and reject malformed leaves" >> { + // covers: enum symbol / fixed length / bytes shape, in BOTH directions and in every refusal — + // the render conventions (enum -> fromString, bytes and fixed -> arrays of signed byte ints), + // the prism round-trip back, and the three malformed leaves. One schema and one value: the + // render example and the round-trip example used to build two near-identical three-leaf + // fixtures, and the render half now asserts the WHOLE document rather than three lookups. + "enum, bytes and fixed leaves: rendering, round-trip, and every malformed rejection" >> { val schema = new Schema.Parser().parse( - """{"type":"record","name":"LeavesRT","namespace":"dev.constructive.eo.avro.circe.test","fields":[ - | {"name":"color","type":{"type":"enum","name":"ColorRT","symbols":["RED","GREEN"]}}, + """{"type":"record","name":"Leaves","namespace":"dev.constructive.eo.avro.circe.test","fields":[ + | {"name":"color","type":{"type":"enum","name":"Color","symbols":["RED","GREEN"]}}, | {"name":"blob","type":"bytes"}, - | {"name":"tag","type":{"type":"fixed","name":"TagRT","size":3}} + | {"name":"tag","type":{"type":"fixed","name":"Tag","size":3}} |]}""".stripMargin ) + val record = new GenericData.Record(schema) + record.put("color", new GenericData.EnumSymbol(schema.getField("color").schema, "GREEN")) + record.put("blob", java.nio.ByteBuffer.wrap(Array[Byte](1, -2, 3))) + record.put("tag", new GenericData.Fixed(schema.getField("tag").schema, Array[Byte](-1, 0, 127))) + val prism = AvroJson.record(schema) val json = Json.obj( "color" -> Json.fromString("GREEN"), - "blob" -> Json.arr(Json.fromInt(1), Json.fromInt(-2)), + "blob" -> Json.arr(Json.fromInt(1), Json.fromInt(-2), Json.fromInt(3)), "tag" -> Json.arr(Json.fromInt(-1), Json.fromInt(0), Json.fromInt(127)), ) - (prism.getOption(json).map(prism.reverseGet) === Some(json)) - .and( - prism.getOption(json.mapObject(_.add("color", Json.fromString("BLUE")))) === None - ) + (AvroJson.avroToJson(record) === json) + .and(prism.getOption(json).map(prism.reverseGet) === Some(json)) + .and(prism.getOption(json.mapObject(_.add("color", Json.fromString("BLUE")))) === None) .and( prism.getOption( json.mapObject(_.add("tag", Json.arr(Json.fromInt(1), Json.fromInt(2)))) ) === None ) - .and( - prism.getOption(json.mapObject(_.add("blob", Json.arr(Json.fromInt(200))))) === None - ) + .and(prism.getOption(json.mapObject(_.add("blob", Json.arr(Json.fromInt(200))))) === None) } // ---- valuePrism and its torn/mended diagonal family ---- @@ -311,34 +317,6 @@ class AvroJsonSpec extends Specification with ScalaCheck: AvroJson.bytesPrism[Combo](writer).getOption(binary(wrec, writer)) === Some(combo) } - // ---- Leaf renderings with no source coverage in the property schema - - // covers: enum → fromString; bytes (ByteBuffer) → array of signed byte ints; fixed → same - "enum, bytes and fixed leaf renderings" >> { - val schema = new Schema.Parser().parse( - """{"type":"record","name":"Leaves","namespace":"dev.constructive.eo.avro.circe.test","fields":[ - | {"name":"color","type":{"type":"enum","name":"Color","symbols":["RED","GREEN"]}}, - | {"name":"blob","type":"bytes"}, - | {"name":"tag","type":{"type":"fixed","name":"Tag","size":3}} - |]}""".stripMargin - ) - val record = new GenericData.Record(schema) - record.put("color", new GenericData.EnumSymbol(schema.getField("color").schema, "GREEN")) - record.put("blob", java.nio.ByteBuffer.wrap(Array[Byte](1, -2, 3))) - record.put("tag", new GenericData.Fixed(schema.getField("tag").schema, Array[Byte](-1, 0, 127))) - - val json = AvroJson.avroToJson(record) - (json.asObject.flatMap(_("color")) === Some(Json.fromString("GREEN"))) - .and( - json.asObject.flatMap(_("blob")) === - Some(Json.arr(Json.fromInt(1), Json.fromInt(-2), Json.fromInt(3))) - ) - .and( - json.asObject.flatMap(_("tag")) === - Some(Json.arr(Json.fromInt(-1), Json.fromInt(0), Json.fromInt(127))) - ) - } - end AvroJsonSpec /** Top-level so kindlings derivation needs no outer accessor (same reason the fixture ADTs in diff --git a/avro/src/test/scala/dev/constructive/eo/avro/jsoniter/AvroJsoniterSpec.scala b/avro/src/test/scala/dev/constructive/eo/avro/jsoniter/AvroJsoniterSpec.scala index c10282e1..07fded87 100644 --- a/avro/src/test/scala/dev/constructive/eo/avro/jsoniter/AvroJsoniterSpec.scala +++ b/avro/src/test/scala/dev/constructive/eo/avro/jsoniter/AvroJsoniterSpec.scala @@ -213,6 +213,19 @@ class AvroJsoniterSpec extends Specification: prism.getOption(json).map(r => str(prism.reverseGet(r))) === Some(str(json)) } + // The first six arms are VALUE faults — the key set is right and only a leaf is wrong. + // + // The last five are STRUCTURE faults, and each one is deliberately COUNT-PRESERVING, because + // `readRecord`'s trailing arity check (`count != fields.size`) otherwise refuses the document + // for an unrelated reason and masks the clause actually under test. The earlier duplicate-key + // arm (`"i":42` appended, 14 keys against 13 fields) was exactly that: it passed on the arity + // clause, so the duplicate detector itself was never exercised. Replacing a SIBLING key with a + // repeat of `i` keeps the count at 13 and leaves the `seen` scan as the only refusal. + // + // covers: AvroJsoniter.scala:331 `(field eq null) || seen(field.pos)` and its `||`, + // AvroJsoniter.scala:333 `seen(field.pos) = true`, + // AvroJsoniter.scala:341 the record's objectEndOrCommaError, + // AvroJsoniter.scala:368 / 380 / 395 the array / map / byte-array end-or-comma refusals "miss on anything the schema does not pin" in { val prism = AvroJsoniter.record(schema) (prism.getOption(mutated(_.replace("{\"s\":", "{\"zzz\":1,\"s\":"))) === None) // extra key @@ -221,10 +234,23 @@ class AvroJsoniterSpec extends Specification: .and(prism.getOption(mutated(_.replace("GREEN", "BLUE"))) === None) // not a symbol .and(prism.getOption(mutated(_.replace("[1,-128]", "[1]"))) === None) // fixed length .and(prism.getOption(mutated(_.replace("[0,-1]", "[0,200]"))) === None) // byte range - .and( - prism.getOption(mutated(_.replace("\"i\":42", "\"i\":42,\"i\":42"))) === None - ) // duplicate key .and(prism.getOption(mutated(_.replace("\"s\":\"", "\"s\":9,\"x\":\""))) === None) + .and( + prism.getOption(mutated(_.replace("\"l\":9007199254740993", "\"i\":42"))) === None + ) // duplicate key, count-preserving: 13 keys, `i` twice, `l` unset + .and( + prism.getOption(mutated(_.replace("{\"v\":7}", "{\"v\":7]"))) === None + ) // nested record closed with `]` + .and( + prism.getOption(mutated(_.replace("\"arr\":[1,2]", "\"arr\":[1,2}"))) === None + ) // array closed with `}` + .and( + prism.getOption(mutated(_.replace("\"map\":{\"k\":\"v\"}", "\"map\":{\"k\":\"v\"]"))) === + None + ) // map closed with `]` + .and( + prism.getOption(mutated(_.replace("\"by\":[0,-1]", "\"by\":[0,-1}"))) === None + ) // bytes array closed with `}` } }