diff --git a/.changeset/plutus-data-encoder.md b/.changeset/plutus-data-encoder.md new file mode 100644 index 00000000..a1601046 --- /dev/null +++ b/.changeset/plutus-data-encoder.md @@ -0,0 +1,13 @@ +--- +"@evolution-sdk/evolution": minor +--- + +Plutus data now has its own encoder, so three kinds of value get the bytes the node writes under every option preset. Their bytes and datum hashes change: + +- A constructor index above 127 writes its `[index, fields]` pair as a definite two-item array. `Data.constr(128n, [1n])` was `d8669f18809f01ffff` and is now `d8668218809f01ff`. +- The integer -2^64 is a plain negative integer, `3bffffffffffffffff`, where it was a negative bignum. +- A bignum whose bytes are longer than 64 is written in 64-byte chunks, as a long byte string already was. + +`Data.toCBORBytes`, `Data.toCBORHex`, `Data.toDatumHash` and the encode side of `Data.FromCBORBytes` and `Data.FromCBORHex` use the new encoder, and so do redeemers, witness datums, inline datums, UPLC data constants and `Redeemers.toScriptDataHash`. All other data keeps its bytes under every preset. Decoding is unchanged. `Data.toCBORBytes` and `Data.toCBORHex` no longer validate their input with the schema first; a value that is not Plutus data now throws `DataError` instead of `ParseError`. + +The CBOR encoder also writes -2^64 as `3bffffffffffffffff`. When a decoded transaction or witness set is written back, an integer decoded from a bignum tag keeps that tag, so a decoded -2^64 keeps its bytes in either form. diff --git a/packages/evolution/src/CBOR.ts b/packages/evolution/src/CBOR.ts index 16359ddf..0fcac8ed 100644 --- a/packages/evolution/src/CBOR.ts +++ b/packages/evolution/src/CBOR.ts @@ -228,15 +228,11 @@ export const CML_DEFAULT_OPTIONS: CodecOptions = { * - Maps: definite-length * - Empty list: `80`; empty map: `a0` * - * This matches `encodeData` in the Haskell `PlutusCore.Data` module, - * `cardano-cli hash-script-data`, Aiken `cbor.serialise()`, and - * `aiken blueprint apply`. Two cases still differ from the node: a - * constructor index above 127 (tag 102) writes its `[index, fields]` pair - * indefinite where the node writes `82`, and the integer -2^64 is written as a - * negative bignum where the node writes `3bffffffffffffffff`. The - * `bounded_bytes` constraint (Conway CDDL: byte strings of at most 64 bytes) - * is enforced at the data-type layer via the `BoundedBytes` CBOR node, - * independent of these codec options. + * `Data.toCBORBytes` with these options writes the bytes the node writes. + * The rules that do not depend on options are applied by the Plutus data + * encoder under every preset: byte strings over 64 bytes in 64-byte chunks, + * bignums only outside -2^64 to 2^64 - 1 with their bytes chunked the same + * way, and a definite `[index, fields]` pair under tag 102. * * @since 2.0.0 * @category constants @@ -260,8 +256,8 @@ export const PLUTUS_DATA_OPTIONS: CodecOptions = { * Uses indefinite-length lists, constructor fields, and maps. It differs from * the node layout ({@link PLUTUS_DATA_OPTIONS}) by writing non-empty maps * indefinite. The `bounded_bytes` constraint (Conway CDDL: byte strings of at - * most 64 bytes) is enforced at the data-type layer via the `BoundedBytes` - * CBOR node, independent of these codec options. + * most 64 bytes) is applied by the Plutus data encoder, independent of these + * codec options. * * @since 1.0.0 * @category constants @@ -1091,6 +1087,17 @@ export const internalEncodeSync = (value: CBOR, options: CodecOptions = CML_DEFA throw new CBORError({ message: `Unsupported CBOR value type: ${typeof value}` }) } +// An integer in the 64-bit range that was decoded from a bignum tag keeps the +// tag on replay, when its captured chunk lengths still fit its bytes. +const replaysAsBignum = (fmt: CBORFormat.Tag, bytes: Uint8Array): boolean => { + if (fmt.child._tag !== "bytes") return false + const encoding = fmt.child.encoding + if (encoding?.tag !== "indefinite") return true + let length = 0 + for (const chunk of encoding.chunks) length += chunk.length + return length === bytes.length +} + const encodeUintSync = (value: bigint, options: CodecOptions, fmt?: CBORFormat): Uint8Array => { if (value < 0n) throw new CBORError({ message: `Cannot encode negative value ${value} as unsigned integer` }) const maxUint64 = 18446744073709551615n @@ -1098,6 +1105,10 @@ const encodeUintSync = (value: bigint, options: CodecOptions, fmt?: CBORFormat): const bytes = bigintToBytes(value) return encodeTagSync(2, bytes, options, fmt) } + if (fmt?._tag === "tag") { + const bytes = bigintToBytes(value) + if (replaysAsBignum(fmt, bytes)) return encodeTagSync(2, bytes, options, fmt) + } // Use specific ByteSize from format metadata if (fmt?._tag === "uint" && fmt.byteSize !== undefined) { return encodeIntHeader(0, value, fmt.byteSize) @@ -1136,12 +1147,16 @@ const encodeUintSync = (value: bigint, options: CodecOptions, fmt?: CBORFormat): const encodeNintSync = (value: bigint, options: CodecOptions, fmt?: CBORFormat): Uint8Array => { if (value >= 0n) throw new CBORError({ message: `Cannot encode non-negative value ${value} as negative integer` }) - const minInt64 = -18446744073709551615n + const minInt64 = -18446744073709551616n if (value < minInt64) { const positiveValue = -(value + 1n) const bytes = bigintToBytes(positiveValue) return encodeTagSync(3, bytes, options, fmt) } + if (fmt?._tag === "tag") { + const bytes = bigintToBytes(-(value + 1n)) + if (replaysAsBignum(fmt, bytes)) return encodeTagSync(3, bytes, options, fmt) + } const positiveValue = -value - 1n // Use specific ByteSize from format metadata if (fmt?._tag === "nint" && fmt.byteSize !== undefined) { diff --git a/packages/evolution/src/Data.ts b/packages/evolution/src/Data.ts index b16a55c4..091bbe14 100644 --- a/packages/evolution/src/Data.ts +++ b/packages/evolution/src/Data.ts @@ -1,6 +1,7 @@ import { blake2b } from "@noble/hashes/blake2.js" import { Data as EffectData, Effect, Equal, FastCheck, Hash, ParseResult, Schema } from "effect" +import * as Bytes from "./Bytes.js" import * as CBOR from "./CBOR.js" import * as DatumHash from "./DatumHash.js" import * as Numeric from "./Numeric.js" @@ -513,6 +514,247 @@ const bytesToBigint = (bytes: Uint8Array): bigint => { return result } +// ============================================================================ +// Encoding +// ============================================================================ + +// The layout choices that CodecOptions make for Plutus data, read once per call +type Layout = { + readonly indefiniteArrays: boolean + readonly indefiniteMaps: boolean + readonly sortMapKeys: boolean + readonly minimal: boolean + readonly mapsAsPairs: boolean +} + +const toLayout = (options: CBOR.CodecOptions): Layout => + options.mode === "custom" + ? { + indefiniteArrays: options.useIndefiniteArrays, + indefiniteMaps: options.useIndefiniteMaps, + sortMapKeys: options.sortMapKeys, + minimal: options.useMinimalEncoding, + mapsAsPairs: options.encodeMapAsPairs === true + } + : { + indefiniteArrays: false, + indefiniteMaps: false, + sortMapKeys: true, + minimal: true, + mapsAsPairs: options.encodeMapAsPairs === true + } + +const MAX_UINT64 = 0xffffffffffffffffn +const BYTES_CHUNK_SIZE = 64 + +// A growable output buffer +type Writer = { buf: Uint8Array; pos: number } + +const reserve = (w: Writer, n: number): void => { + if (w.pos + n <= w.buf.length) return + let size = w.buf.length * 2 + while (size < w.pos + n) size *= 2 + const buf = new Uint8Array(size) + buf.set(w.buf.subarray(0, w.pos)) + w.buf = buf +} + +const writeByte = (w: Writer, b: number): void => { + reserve(w, 1) + w.buf[w.pos++] = b +} + +const writeBytes = (w: Writer, bytes: Uint8Array): void => { + reserve(w, bytes.length) + w.buf.set(bytes, w.pos) + w.pos += bytes.length +} + +// A head with the shortest argument, for a length or tag below 2^32 +const writeHead = (w: Writer, major: number, n: number): void => { + reserve(w, 5) + const mt = major << 5 + const buf = w.buf + if (n < 24) { + buf[w.pos++] = mt | n + } else if (n < 0x100) { + buf[w.pos++] = mt | 24 + buf[w.pos++] = n + } else if (n < 0x10000) { + buf[w.pos++] = mt | 25 + buf[w.pos++] = n >>> 8 + buf[w.pos++] = n & 0xff + } else { + buf[w.pos++] = mt | 26 + buf[w.pos++] = n >>> 24 + buf[w.pos++] = (n >>> 16) & 0xff + buf[w.pos++] = (n >>> 8) & 0xff + buf[w.pos++] = n & 0xff + } +} + +// An integer head for 0 <= n <= 2^64 - 1. Without minimal encoding, an +// argument of 24 or more takes the 8-byte form, as the CBOR encoder writes it +const writeIntHead = (w: Writer, major: number, n: bigint, minimal: boolean): void => { + if (n < 0x100000000n && (minimal || n < 24n)) { + writeHead(w, major, Number(n)) + return + } + reserve(w, 9) + const buf = w.buf + const high = Number(n >> 32n) + const low = Number(n & 0xffffffffn) + buf[w.pos++] = (major << 5) | 27 + buf[w.pos++] = high >>> 24 + buf[w.pos++] = (high >>> 16) & 0xff + buf[w.pos++] = (high >>> 8) & 0xff + buf[w.pos++] = high & 0xff + buf[w.pos++] = low >>> 24 + buf[w.pos++] = (low >>> 16) & 0xff + buf[w.pos++] = (low >>> 8) & 0xff + buf[w.pos++] = low & 0xff +} + +// A byte string of at most 64 bytes is written definite. A longer one is +// written indefinite, in 64-byte chunks +const writeBoundedBytes = (w: Writer, bytes: Uint8Array): void => { + const length = bytes.length + if (length <= BYTES_CHUNK_SIZE) { + writeHead(w, 2, length) + writeBytes(w, bytes) + return + } + writeByte(w, 0x5f) + for (let offset = 0; offset < length; offset += BYTES_CHUNK_SIZE) { + const end = Math.min(offset + BYTES_CHUNK_SIZE, length) + writeHead(w, 2, end - offset) + writeBytes(w, bytes.subarray(offset, end)) + } + writeByte(w, 0xff) +} + +// The big-endian bytes of a positive integer, with no leading zero byte +const toBigEndian = (n: bigint): Uint8Array => { + const hex = n.toString(16) + const offset = hex.length & 1 + const bytes = new Uint8Array((hex.length + offset) >> 1) + bytes[0] = parseInt(hex.slice(0, 2 - offset), 16) + for (let i = 1; i < bytes.length; i++) { + bytes[i] = parseInt(hex.slice(2 * i - offset, 2 * i + 2 - offset), 16) + } + return bytes +} + +// An integer from -2^64 to 2^64 - 1 is a CBOR integer. Outside that range it +// is a bignum, tag 2 or 3, whose bytes follow the byte string rule +const writeInt = (w: Writer, layout: Layout, value: bigint): void => { + const major = value < 0n ? 1 : 0 + const n = value < 0n ? -1n - value : value + if (n <= MAX_UINT64) { + writeIntHead(w, major, n, layout.minimal) + return + } + writeByte(w, 0xc2 + major) + writeBoundedBytes(w, toBigEndian(n)) +} + +const writeList = (w: Writer, layout: Layout, items: ReadonlyArray): void => { + const length = items.length + if (length === 0) { + writeByte(w, 0x80) + return + } + if (layout.indefiniteArrays) writeByte(w, 0x9f) + else writeHead(w, 4, length) + for (let i = 0; i < length; i++) writeData(w, layout, items[i]) + if (layout.indefiniteArrays) writeByte(w, 0xff) +} + +// Shorter encoded keys first, then bytewise, as the CBOR encoder sorts map keys +const compareEncodedKeys = (a: readonly [Uint8Array, Uint8Array], b: readonly [Uint8Array, Uint8Array]): number => { + const x = a[0] + const y = b[0] + if (x.length !== y.length) return x.length - y.length + for (let i = 0; i < x.length; i++) { + if (x[i] !== y[i]) return x[i] - y[i] + } + return 0 +} + +const writeMap = (w: Writer, layout: Layout, map: globalThis.Map): void => { + const size = map.size + if (layout.mapsAsPairs) { + if (size === 0) { + writeByte(w, 0x80) + return + } + if (layout.indefiniteArrays) writeByte(w, 0x9f) + else writeHead(w, 4, size) + for (const [key, value] of map) { + writeByte(w, layout.indefiniteArrays ? 0x9f : 0x82) + writeData(w, layout, key) + writeData(w, layout, value) + if (layout.indefiniteArrays) writeByte(w, 0xff) + } + if (layout.indefiniteArrays) writeByte(w, 0xff) + return + } + if (size === 0) { + writeByte(w, 0xa0) + return + } + if (layout.indefiniteMaps) writeByte(w, 0xbf) + else if (size < 24 || layout.minimal) writeHead(w, 5, size) + else throw new CBOR.CBORError({ message: `Map too long: ${size} entries` }) + if (layout.sortMapKeys) { + const entries = Array.from(map, ([key, value]) => [encodeData(key, layout), encodeData(value, layout)] as const) + entries.sort(compareEncodedKeys) + for (const [key, value] of entries) { + writeBytes(w, key) + writeBytes(w, value) + } + } else { + for (const [key, value] of map) { + writeData(w, layout, key) + writeData(w, layout, value) + } + } + if (layout.indefiniteMaps) writeByte(w, 0xff) +} + +// Indices 0 to 6 are tags 121 to 127, indices 7 to 127 are tags 1280 to 1400, +// and any other index is tag 102 over the definite pair [index, fields] +const writeConstr = (w: Writer, layout: Layout, constr: Constr): void => { + const index = constr.index + if (index < 0n || index > MAX_UINT64) { + throw new DataError({ message: `Constructor index out of range: ${index}` }) + } + const tag = index <= 6n ? 121 + Number(index) : index <= 127n ? 1280 + Number(index) - 7 : 102 + // Without minimal encoding the CBOR encoder cannot write a tag of 24 or more + if (!layout.minimal) throw new CBOR.CBORError({ message: `Tag ${tag} too large` }) + writeHead(w, 6, tag) + if (tag === 102) { + writeByte(w, 0x82) + writeIntHead(w, 0, index, true) + } + writeList(w, layout, constr.fields) +} + +const writeData = (w: Writer, layout: Layout, data: Data): void => { + if (typeof data === "bigint") writeInt(w, layout, data) + else if (data instanceof Uint8Array) writeBoundedBytes(w, data) + else if (Array.isArray(data)) writeList(w, layout, data) + else if (data instanceof globalThis.Map) writeMap(w, layout, data) + else if (data instanceof Constr || isConstr(data)) writeConstr(w, layout, data) + else throw new DataError({ message: `Unsupported PlutusData type: ${String(data)}` }) +} + +const encodeData = (data: Data, layout: Layout): Uint8Array => { + const w: Writer = { buf: new Uint8Array(256), pos: 0 } + writeData(w, layout, data) + return w.buf.slice(0, w.pos) +} + // ============================================================================ // Combinators // ============================================================================ @@ -794,20 +1036,33 @@ export const FromCDDL = Schema.transformOrFail(CDDLSchema, Schema.typeSchema(Dat /** * CBOR bytes transformation schema for PlutusData using CDDL. - * Transforms between CBOR bytes and Data using CDDL encoding. + * Decodes through {@link FromCDDL}. Encodes as {@link toCBORBytes} does. * * @since 2.0.0 * @category schemas */ -export const FromCBORBytes = (options: CBOR.CodecOptions = CBOR.CML_DATA_DEFAULT_OPTIONS) => - Schema.compose( - CBOR.FromBytes(options), // Uint8Array → CBOR - FromCDDL // CBOR → Data - ).annotations({ +export const FromCBORBytes = (options: CBOR.CodecOptions = CBOR.CML_DATA_DEFAULT_OPTIONS) => { + const fromBytes = Schema.compose(CBOR.FromBytes(options), FromCDDL) + const layout = toLayout(options) + return Schema.transformOrFail(Schema.Uint8ArrayFromSelf, Schema.typeSchema(DataSchema), { + strict: true, + decode: (bytes, parseOptions) => ParseResult.decode(fromBytes)(bytes, parseOptions), + encode: (data, _, ast) => + ParseResult.try({ + try: () => encodeData(data, layout), + catch: (error) => + new ParseResult.Type( + ast, + data, + `Failed to encode CBOR value: ${error instanceof Error ? error.message : String(error)}` + ) + }) + }).annotations({ identifier: "Data.FromCBORBytes", title: "Data from CBOR Bytes using CDDL", description: "Transforms CBOR bytes to Data using CDDL encoding" }) +} /** * CBOR hex transformation schema for PlutusData using CDDL. @@ -827,22 +1082,31 @@ export const FromCBORHex = (options: CBOR.CodecOptions = CBOR.CML_DATA_DEFAULT_O }) /** - * Encode PlutusData to CBOR bytes + * Encode PlutusData to CBOR bytes. + * + * `options` choose definite or indefinite lists and maps, map key order and + * integer widths. These rules hold under every option: + * + * - A byte string over 64 bytes is written indefinite, in 64-byte chunks. + * - An integer from -2^64 to 2^64 - 1 is a CBOR integer. Outside that range it + * is a bignum (tag 2 or 3) whose bytes follow the byte string rule. + * - A constructor index above 127 is tag 102 over a definite pair + * `[index, fields]`; only the fields follow `options`. * * @since 2.0.0 * @category transformation */ -export const toCBORBytes = (data: Data, options: CBOR.CodecOptions = CBOR.CML_DATA_DEFAULT_OPTIONS) => - Schema.encodeSync(FromCBORBytes(options))(data) +export const toCBORBytes = (data: Data, options: CBOR.CodecOptions = CBOR.CML_DATA_DEFAULT_OPTIONS): Uint8Array => + encodeData(data, toLayout(options)) /** - * Encode PlutusData to CBOR hex string + * Encode PlutusData to CBOR hex string, as {@link toCBORBytes} writes it. * * @since 2.0.0 * @category transformation */ -export const toCBORHex = (data: Data, options: CBOR.CodecOptions = CBOR.CML_DATA_DEFAULT_OPTIONS) => - Schema.encodeSync(FromCBORHex(options))(data) +export const toCBORHex = (data: Data, options: CBOR.CodecOptions = CBOR.CML_DATA_DEFAULT_OPTIONS): string => + Bytes.toHex(toCBORBytes(data, options)) /** * Decode PlutusData from CBOR bytes diff --git a/packages/evolution/src/Redeemer.ts b/packages/evolution/src/Redeemer.ts index e79c47d5..4fa54d06 100644 --- a/packages/evolution/src/Redeemer.ts +++ b/packages/evolution/src/Redeemer.ts @@ -282,14 +282,14 @@ export const FromCBORBytes = (options: CBOR.TxCodecOptions | CBOR.CodecOptions = decode: (bytes, parseOptions) => ParseResult.decode(Schema.compose(CBOR.FromBytes(ledger), FromCDDL))(bytes, parseOptions), encode: (redeemer, parseOptions, ast) => - Effect.flatMap(ParseResult.encode(FromCDDL)(redeemer, parseOptions), ([tag, index, data, exUnits]) => + Effect.flatMap(ParseResult.encode(FromCDDL)(redeemer, parseOptions), ([tag, index, _, exUnits]) => ParseResult.try({ try: () => encodeArray( [ CBOR.toCBORBytes(tag, ledger), CBOR.toCBORBytes(index, ledger), - CBOR.toCBORBytes(data, plutusData), + PlutusData.toCBORBytes(redeemer.data, plutusData), CBOR.toCBORBytes(exUnits, ledger) ], ledger diff --git a/packages/evolution/src/Redeemers.ts b/packages/evolution/src/Redeemers.ts index 345e1ecf..f257436e 100644 --- a/packages/evolution/src/Redeemers.ts +++ b/packages/evolution/src/Redeemers.ts @@ -499,14 +499,17 @@ export const FromCBORBytesMap = (options: CBOR.TxCodecOptions | CBOR.CodecOption encode: (redeemers, parseOptions, ast) => Eff.flatMap(ParseResult.encode(FromMapCDDL)(redeemers, parseOptions), (map) => ParseResult.try({ - try: () => - encodeMap( - Array.from(map, ([key, [data, exUnits]]) => [ + try: () => { + // The map holds the entries in the order of `redeemers.value` + const values = Array.from(redeemers.value.values()) + return encodeMap( + Array.from(map, ([key, [, exUnits]], i) => [ CBOR.toCBORBytes(key, ledger), - encodeArray([CBOR.toCBORBytes(data, plutusData), CBOR.toCBORBytes(exUnits, ledger)], ledger) + encodeArray([Data.toCBORBytes(values[i].data, plutusData), CBOR.toCBORBytes(exUnits, ledger)], ledger) ]), ledger - ), + ) + }, catch: (error) => encodeError(ast, redeemers, error) }) ) diff --git a/packages/evolution/test/CBOR.test.ts b/packages/evolution/test/CBOR.test.ts index 4c3df3fa..dd9f0907 100644 --- a/packages/evolution/test/CBOR.test.ts +++ b/packages/evolution/test/CBOR.test.ts @@ -192,6 +192,19 @@ describe("CBOR Implementation Tests", () => { expect(decoded).toBe(maxUint64) }) + it("writes -2^64 as a negative integer and bignums only outside the 64-bit range", () => { + expect(CBOR.toCBORHex(-(2n ** 64n))).toBe("3bffffffffffffffff") + expect(CBOR.toCBORHex(-(2n ** 64n) - 1n)).toBe("c349010000000000000000") + expect(CBOR.toCBORHex(2n ** 64n)).toBe("c249010000000000000000") + }) + + it("replays a bignum decoded inside the 64-bit range as a bignum", () => { + for (const hex of ["c348ffffffffffffffff", "3bffffffffffffffff", "c24101", "c35f4101ff", "82c24101d87980"]) { + const decoded = CBOR.fromCBORHexWithFormat(hex) + expect(CBOR.toCBORHexWithFormat(decoded.value, decoded.format)).toBe(hex) + } + }) + it("should handle various CBOR integer encoding sizes", () => { const testCases = [ { name: "direct encoding (0-23)", value: 23n }, diff --git a/packages/evolution/test/Data.encoder.test.ts b/packages/evolution/test/Data.encoder.test.ts new file mode 100644 index 00000000..2de92629 --- /dev/null +++ b/packages/evolution/test/Data.encoder.test.ts @@ -0,0 +1,272 @@ +import { FastCheck, Schema } from "effect" +import { describe, expect, it } from "vitest" + +import * as Bytes from "../src/Bytes.js" +import * as CBOR from "../src/CBOR.js" +import * as Data from "../src/Data.js" +import * as DatumOption from "../src/DatumOption.js" +import * as InlineDatum from "../src/InlineDatum.js" +import * as Redeemer from "../src/Redeemer.js" +import * as Redeemers from "../src/Redeemers.js" +import * as TransactionWitnessSet from "../src/TransactionWitnessSet.js" +import * as UPLC from "../src/UPLC.js" + +// Hex of the bytes from..from+n-1 +const seq = (from: number, n: number): string => Bytes.toHex(Uint8Array.from({ length: n }, (_, i) => from + i)) + +const big65 = BigInt("0x" + seq(1, 65)) +const big200 = BigInt("0x" + seq(1, 200)) +const bigPayload65 = `5f5840${seq(1, 64)}41${seq(65, 1)}ff` +const bigPayload200 = `5f5840${seq(1, 64)}5840${seq(65, 64)}5840${seq(129, 64)}48${seq(193, 8)}ff` + +// Bytes and datum hashes from two independent reference encoders of the node +// layout, which agree on every case +const oracle: ReadonlyArray<{ name: string; data: Data.Data; hex: string; hash: string }> = [ + { + name: "Constr 128 [1]", + data: Data.constr(128n, [1n]), + hex: "d8668218809f01ff", + hash: "fd05a09eedb9747bc38ac3dcc2eec2f9790f1ddf9dbe125577dea6eba9ce7405" + }, + { + name: "Constr 1000 [1]", + data: Data.constr(1000n, [1n]), + hex: "d866821903e89f01ff", + hash: "ba07c7523ca414b7b4a379b8bdf2dd41b47bb1d95cb0ef17bc048b5ed7d48ea8" + }, + { + name: "Constr 128 []", + data: Data.constr(128n, []), + hex: "d86682188080", + hash: "e4886280ddea3623dfcce52c34b42d69d7932b6967798a1819442c6d9532c175" + }, + { + name: "nested tag 102", + data: Data.constr(200n, [ + Data.constr(300n, [2n, [3n]]), + Data.constr(0n, []), + Data.constr(7n, [Bytes.fromHex("ab")]) + ]), + hex: "d8668218c89fd8668219012c9f029f03ffffd87980d905009f41abffff", + hash: "467913537d7e5be4dd016e8bbe0c19aae048a99c6d0ce10b32293ad9486f3993" + }, + { + name: "65-byte bignum", + data: big65, + hex: `c2${bigPayload65}`, + hash: "d75c7051c8d273e945ae880b5165f50d0b97f83d492f5c58fbe1c78ab04452bd" + }, + { + name: "65-byte negative bignum", + data: -1n - big65, + hex: `c3${bigPayload65}`, + hash: "6952d7f09ded3b1a2011daeef0985493ab1c2ead3f0eec80271d651e42091d3e" + }, + { + name: "200-byte bignum", + data: big200, + hex: `c2${bigPayload200}`, + hash: "3bd58ca6d456c55524db48c55479a67d633dca1bd846e1e2365833ffd947a862" + }, + { + name: "200-byte negative bignum", + data: -1n - big200, + hex: `c3${bigPayload200}`, + hash: "f83089b61dce7020b3bc2a7d3733ab74f5272e6b674a26b103f195792313d7f6" + }, + { + name: "-2^64", + data: -(2n ** 64n), + hex: "3bffffffffffffffff", + hash: "69393b55a0ae218f47bfa0376159277a159c76662f738592c1553e3d904c5ac7" + }, + { + name: "2^64 - 1", + data: 2n ** 64n - 1n, + hex: "1bffffffffffffffff", + hash: "3fac877e3f1eaefb3cbb71eb8d49248b51571fd18dc02e74f464f038eb734935" + }, + { + name: "2^64", + data: 2n ** 64n, + hex: "c249010000000000000000", + hash: "0b854352f6a4c02db6f13ac878a41f0b11f54950ad9b88170fb509c916ff0a71" + }, + { + name: "-2^64 - 1", + data: -(2n ** 64n) - 1n, + hex: "c349010000000000000000", + hash: "42e2692b0e46ba0dc699ee2f1aa07e1bffb800c35c7b138ddb756d8bf998bc89" + }, + { + name: "65-byte byte string", + data: Bytes.fromHex(seq(0, 65)), + hex: `5f5840${seq(0, 64)}41${seq(64, 1)}ff`, + hash: "1aeac7f533c4d772fea626e7a9233cb1c58ef171f585c05f3070791ff695b898" + } +] + +const presets: ReadonlyArray<[string, CBOR.CodecOptions]> = [ + ["PLUTUS_DATA_OPTIONS", CBOR.PLUTUS_DATA_OPTIONS], + ["CML_DATA_DEFAULT_OPTIONS", CBOR.CML_DATA_DEFAULT_OPTIONS], + ["CML_DATA_DEFINITE_OPTIONS", CBOR.CML_DATA_DEFINITE_OPTIONS], + ["CANONICAL_OPTIONS", CBOR.CANONICAL_OPTIONS], + ["CML_DEFAULT_OPTIONS", CBOR.CML_DEFAULT_OPTIONS], + ["STRUCT_FRIENDLY_OPTIONS", CBOR.STRUCT_FRIENDLY_OPTIONS], + ["canonical, map as pairs", { mode: "canonical", encodeMapAsPairs: true }], + [ + "custom, sorted indefinite maps", + { + mode: "custom", + useIndefiniteArrays: true, + useIndefiniteMaps: true, + useDefiniteForEmpty: false, + sortMapKeys: true, + useMinimalEncoding: true + } + ], + [ + "custom, non-minimal", + { + mode: "custom", + useIndefiniteArrays: false, + useIndefiniteMaps: false, + useDefiniteForEmpty: true, + sortMapKeys: false, + useMinimalEncoding: false + } + ] +] + +const indefiniteLists = (options: CBOR.CodecOptions) => options.mode === "custom" && options.useIndefiniteArrays + +describe("Data encoder", () => { + describe("matches the reference encoders", () => { + it.each(oracle)("$name", ({ data, hash, hex }) => { + expect(Data.toCBORHex(data, CBOR.PLUTUS_DATA_OPTIONS)).toBe(hex) + expect(Data.toDatumHash(data, CBOR.PLUTUS_DATA_OPTIONS).hash).toEqual(Bytes.fromHex(hash)) + expect(Data.fromCBORHex(hex)).toEqual(data) + }) + + it("writes the oracle bytes through FromCBORHex and withSchema", () => { + const codec = Data.withSchema(Schema.typeSchema(Data.DataSchema), CBOR.PLUTUS_DATA_OPTIONS) + for (const { data, hex } of oracle) { + expect(Schema.encodeSync(Data.FromCBORHex(CBOR.PLUTUS_DATA_OPTIONS))(data)).toBe(hex) + expect(codec.toCBORHex(data)).toBe(hex) + } + }) + }) + + describe("rules that hold under every preset", () => { + const leaves = oracle.filter(({ data }) => !Data.isConstr(data)) + + it.each(presets)("integers and byte strings under %s", (_, options) => { + for (const { data, hex } of leaves) { + expect(Data.toCBORHex(data, options)).toBe(hex) + } + }) + + it.each(presets.filter(([, options]) => options.mode === "canonical" || options.useMinimalEncoding))( + "tag 102 writes a definite pair under %s", + (_, options) => { + const fields = indefiniteLists(options) ? "9f01ff" : "8101" + expect(Data.toCBORHex(Data.constr(128n, [1n]), options)).toBe(`d866821880${fields}`) + expect(Data.toCBORHex(Data.constr(1000n, [1n]), options)).toBe(`d866821903e8${fields}`) + expect(Data.toCBORHex(Data.constr(2n ** 64n - 1n, []), options)).toBe("d866821bffffffffffffffff80") + } + ) + + it("bignum bytes inside lists, maps and constructors", () => { + const data = Data.constr(0n, [[big65], new Map([[big65, -1n - big65]])]) + expect(Data.toCBORHex(data, CBOR.PLUTUS_DATA_OPTIONS)).toBe( + `d8799f9fc2${bigPayload65}ffa1c2${bigPayload65}c3${bigPayload65}ff` + ) + }) + }) + + describe("matches the CBOR tree encoder outside the rule cases", () => { + // Integers in (-2^64, 2^64), byte strings up to 64 bytes, indices up to 127 + const leaf = FastCheck.oneof( + FastCheck.bigInt({ min: -(2n ** 64n) + 1n, max: 2n ** 64n - 1n }), + FastCheck.constantFrom(0n, 23n, 24n, 255n, 256n, 65535n, 65536n, 2n ** 32n, -1n, -24n, -25n, -(2n ** 64n) + 1n), + FastCheck.uint8Array({ maxLength: 64 }) + ) + const { tree } = FastCheck.letrec<{ tree: Data.Data; list: Data.List; map: Data.Map; constr: Data.Constr }>( + (tie) => ({ + tree: FastCheck.oneof({ depthSize: "small", maxDepth: 4 }, leaf, tie("list"), tie("map"), tie("constr")), + list: FastCheck.array(tie("tree"), { maxLength: 6 }), + map: FastCheck.array(FastCheck.tuple(tie("tree"), tie("tree")), { maxLength: 6 }).map( + (entries) => new Map(entries) + ), + constr: FastCheck.tuple(FastCheck.bigInt({ min: 0n, max: 127n }), FastCheck.array(tie("tree"), { maxLength: 4 })).map( + ([index, fields]) => Data.constr(index, fields) + ) + }) + ) + const samples = FastCheck.sample(tree, { seed: 614, numRuns: 200 }) + + const viaTree = (data: Data.Data, options: CBOR.CodecOptions): string => { + try { + return CBOR.toCBORHex(Data.plutusDataToCBORValue(data), options) + } catch { + return "throws" + } + } + const direct = (data: Data.Data, options: CBOR.CodecOptions): string => { + try { + return Data.toCBORHex(data, options) + } catch { + return "throws" + } + } + + it.each(presets)("%s", (_, options) => { + for (const data of samples) { + expect(direct(data, options)).toBe(viaTree(data, options)) + } + }) + }) + + describe("encoders that write Plutus data", () => { + // Constr 128 [-2^64, 65-byte bignum] in the node layout + const data = Data.constr(128n, [-(2n ** 64n), big65]) + const dataHex = `d8668218809f3bffffffffffffffffc2${bigPayload65}ff` + const exUnits = new Redeemer.ExUnits({ mem: 1n, steps: 2n }) + const redeemer = new Redeemer.Redeemer({ tag: "spend", index: 0n, data, exUnits }) + const options = { ledger: CBOR.CML_DEFAULT_OPTIONS, plutusData: CBOR.PLUTUS_DATA_OPTIONS } + + it("redeemers and witness datums", () => { + expect(Redeemer.toCBORHex(redeemer, options)).toBe(`840000${dataHex}820102`) + expect(Redeemers.toCBORHex(new Redeemers.RedeemerArray({ value: [redeemer] }), options)).toBe( + `81840000${dataHex}820102` + ) + expect(Redeemers.toCBORHexMap(Redeemers.makeRedeemerMap([redeemer]), options)).toBe( + `a182000082${dataHex}820102` + ) + const witnessSet = new TransactionWitnessSet.TransactionWitnessSet({ plutusData: [data] }) + expect(TransactionWitnessSet.toCBORHex(witnessSet, options)).toBe(`a104d9010281${dataHex}`) + }) + + it("a decoded witness set replays -2^64 in the form it was written", () => { + const fresh = TransactionWitnessSet.toCBORHex( + new TransactionWitnessSet.TransactionWitnessSet({ plutusData: [-(2n ** 64n)] }) + ) + expect(fresh).toBe("a104d90102813bffffffffffffffff") + for (const hex of [ + fresh, + "a104d9010281c348ffffffffffffffff", + "a105a1820000823bffffffffffffffff820000", + "a105a182000082c348ffffffffffffffff820000" + ]) { + const decoded = TransactionWitnessSet.fromCBORHexWithFormat(hex) + expect(TransactionWitnessSet.toCBORHexWithFormat(decoded.value, decoded.format)).toBe(hex) + } + }) + + it("inline datums and UPLC data constants", () => { + const inline = new InlineDatum.InlineDatum({ data }) + expect(DatumOption.toCBORHex(inline)).toContain(Data.toCBORHex(data)) + expect(UPLC.dataConstant(data)).toMatchObject({ value: Bytes.fromHex(dataHex) }) + }) + }) +}) diff --git a/packages/evolution/test/Data.golden.test.ts b/packages/evolution/test/Data.golden.test.ts index b2eba557..34a76a68 100644 --- a/packages/evolution/test/Data.golden.test.ts +++ b/packages/evolution/test/Data.golden.test.ts @@ -201,7 +201,23 @@ const getTestCases = ( ): Array => { const entries = getGoldenEntries(type) const count = TEST_CONFIG[type][category] - return entries.slice(0, count) + const sampled = entries.slice(0, count) + // The golden files hold the bytes of the first release, which wrote the tag + // 102 pair of a constructor index above 127 with the list options. The + // encoder now writes that pair definite (#601), so those entries are checked + // on decode and round trip only + return category === "encoding" ? sampled.filter((entry) => !holdsGeneralConstr(entry.sample)) : sampled +} + +/** + * Whether a sample holds a constructor index above 127 at any depth + * + */ +const holdsGeneralConstr = (sample: unknown): boolean => { + if (isConstrSample(sample) && BigInt(sample.index) > 127n) return true + if (Array.isArray(sample)) return sample.some(holdsGeneralConstr) + if (typeof sample === "object" && sample !== null) return Object.values(sample).some(holdsGeneralConstr) + return false } /** diff --git a/packages/evolution/test/Data.test.ts b/packages/evolution/test/Data.test.ts index 80d3fea4..dbee2324 100644 --- a/packages/evolution/test/Data.test.ts +++ b/packages/evolution/test/Data.test.ts @@ -141,7 +141,7 @@ describe("CBOR Encoding/Decoding", () => { { name: "large constructor index", value: Data.constr(999999n, [42n]), - expectedHex: "d8669f1a000f423f9f182affff" + expectedHex: "d866821a000f423f9f182aff" }, { name: "empty list", diff --git a/packages/evolution/test/TransactionWitnessSet.DataOptions.test.ts b/packages/evolution/test/TransactionWitnessSet.DataOptions.test.ts index 23f47ce7..b753ea1b 100644 --- a/packages/evolution/test/TransactionWitnessSet.DataOptions.test.ts +++ b/packages/evolution/test/TransactionWitnessSet.DataOptions.test.ts @@ -241,9 +241,36 @@ describe("byte identity with the single-options encoder", () => { ["canonical map pairs", { mode: "canonical", encodeMapAsPairs: true }] ] + // The tree writes a constructor index above 127 and an integer outside + // (-2^64, 2^64) as the Plutus data encoder does not, so the sampled data + // stays inside those ranges + const inTreeRange = (data: Data.Data): Data.Data => { + if (typeof data === "bigint") return data % 2n ** 64n + if (data instanceof Uint8Array) return data + if (Array.isArray(data)) return data.map(inTreeRange) + if (data instanceof Map) return new Map(Array.from(data, ([k, v]) => [inTreeRange(k), inTreeRange(v)])) + const constr = data as Data.Constr + return Data.constr(constr.index % 128n, constr.fields.map(inTreeRange)) + } + const withDataInTreeRange = (tx: Transaction.Transaction): Transaction.Transaction => { + const { plutusData, redeemers } = tx.witnessSet + const mapped = redeemers + ?.toArray() + .map((r) => new Redeemer.Redeemer({ tag: r.tag, index: r.index, data: inTreeRange(r.data), exUnits: r.exUnits })) + return withWitnessSet(tx, { + plutusData: plutusData?.map(inTreeRange), + redeemers: + mapped === undefined + ? undefined + : redeemers?._tag === "RedeemerMap" + ? Redeemers.makeRedeemerMap(mapped) + : new Redeemers.RedeemerArray({ value: mapped }) + }) + } + // Each sampled transaction, and the same transaction without datums and redeemers const txs = FastCheck.sample(Transaction.arbitrary, { seed: 604, numRuns: 20 }).flatMap((tx) => [ - tx, + withDataInTreeRange(tx), withWitnessSet(tx, { plutusData: undefined, redeemers: undefined }) ]) const holdsPlutusData = (tx: Transaction.Transaction) =>