diff --git a/CHANGELOG.md b/CHANGELOG.md index fca9b994e..651577513 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -53,6 +53,22 @@ The sections do not signal obligations; the prefixes do. ## Unreleased +### Changed + +- The `offset` of a segment counts bytes from the most significant byte of the + slot: byte `0` is the first byte of the slot's big-endian word. The + description now says so, says how to convert from a layout that counts from + the low-order end (`$wordsize - o - n`), and allows writing that conversion as + an expression. New segment examples pack an `address` and a `uint32` into + one slot, each written with literal offsets (`12` and `8`) and with an + expression that takes the value's size from the region's own `length` + (`{ ".length": "$this" }`) ([#309]). + - Schemas: **ethdebug/format/pointer/scheme/segment** + - Producers: no change needed. The reference implementation and bugc already + resolve offsets this way; the text states existing meaning. + - Consumers: no change needed. A consumer that resolved `offset` against the + big-endian word already follows this. + ## 0.1.0-draft.0 — 2026-09-21 The version scheme changed: prerelease versions of the specification are now @@ -616,3 +632,4 @@ First published version of the specification. [#285]: https://github.com/ethdebug/format/pull/285 [#286]: https://github.com/ethdebug/format/pull/286 [#303]: https://github.com/ethdebug/format/pull/303 +[#309]: https://github.com/ethdebug/format/pull/309 diff --git a/packages/pointers/src/segment.examples.test.ts b/packages/pointers/src/segment.examples.test.ts new file mode 100644 index 000000000..ea2dc7794 --- /dev/null +++ b/packages/pointers/src/segment.examples.test.ts @@ -0,0 +1,87 @@ +import { expect, describe, it } from "vitest"; + +import type { Pointer } from "@ethdebug/format"; +import { schemas } from "@ethdebug/format"; + +import type { Machine } from "#machine"; +import { Data } from "#data"; +import { dereference } from "./dereference/index.js"; + +const { examples } = schemas[ + "schema:ethdebug/format/pointer/scheme/segment" +] as { + examples: Record[]; +}; + +// the packed examples all address slot 2 +const packed = examples.filter((example) => example.slot === 2); + +// slot 2 packs an address (bytes 12 to 31) and a uint32 (bytes 8 to 11); a +// byte's number counts from the most significant end of the word +const address = Uint8Array.from({ length: 20 }, (_, i) => 0xa0 + i); +const count = new Uint8Array([0xde, 0xad, 0xbe, 0xef]); +const word = new Uint8Array(32); +word.set(address, 12); +word.set(count, 8); + +const state = { + stack: { length: Promise.resolve(0n) }, + storage: { + read: async ({ + slot, + slice, + }: { + slot: Data; + slice?: Machine.State.Slice; + }) => { + expect(slot).toEqual(Data.fromNumber(2)); + return Data.fromBytes( + slice + ? word.slice( + Number(slice.offset), + Number(slice.offset + slice.length), + ) + : word, + ); + }, + }, +} as unknown as Machine.State; + +async function resolve(example: Record): Promise { + const pointer = { location: "storage", ...example } as Pointer; + const cursor = await dereference(pointer); + const { regions, read } = await cursor.view(state); + return read(regions[0]); +} + +describe("segment schema examples (packed)", () => { + it("include literal and expression offsets", () => { + expect(packed.map((example) => example.offset)).toEqual([ + 12, + 8, + { + $difference: ["$wordsize", { ".length": "$this" }], + }, + { + $difference: ["$wordsize", { $sum: [20, { ".length": "$this" }] }], + }, + ]); + }); + + it("count offsets from the most significant byte", async () => { + const [literalAddress, literalCount] = packed; + + expect(await resolve(literalAddress)).toEqual(Data.fromBytes(address)); + expect(await resolve(literalCount)).toEqual(Data.fromBytes(count)); + expect(await resolve({ slot: 2, offset: 0, length: 1 })).toEqual( + Data.fromBytes(new Uint8Array([0x00])), + ); + }); + + it("resolve each expression like its literal twin", async () => { + const [literalAddress, literalCount, exprAddress, exprCount] = packed; + + expect(await resolve(exprAddress)).toEqual(await resolve(literalAddress)); + expect(await resolve(exprCount)).toEqual(await resolve(literalCount)); + }); +}); diff --git a/packages/web/spec/pointer/concepts.mdx b/packages/web/spec/pointer/concepts.mdx index da47aa746..4e41e1c70 100644 --- a/packages/web/spec/pointer/concepts.mdx +++ b/packages/web/spec/pointer/concepts.mdx @@ -283,6 +283,9 @@ This format currently defines two such addressing schemes: single continuous byte array and addressing a slot or collection of slots in a word-arranged locaton (respectively). +Within a slot, the segment scheme numbers bytes from the most significant +byte: byte `0` is the first byte of the slot's big-endian word. +
**Example**: Slice-based region diff --git a/schemas/pointer/scheme/segment.schema.yaml b/schemas/pointer/scheme/segment.schema.yaml index 487795bf5..4404a431d 100644 --- a/schemas/pointer/scheme/segment.schema.yaml +++ b/schemas/pointer/scheme/segment.schema.yaml @@ -24,9 +24,32 @@ properties: description: | The starting byte index within the slot. + Bytes within a slot are numbered from the most significant byte. A + slot's value is its `$wordsize`-byte big-endian word, and byte `0` is + the first byte of that word, as if the word were written to memory. An + `offset` of `0` therefore addresses the most significant byte of the + slot, and an `offset` of `$wordsize - 1` addresses the least + significant byte. + This field is **optional**. If unspecified, it has the default value of `0`, indicating that the segment begins at the start of the specified - slot. + slot (its most significant byte). + + A data layout that counts bytes from the low-order end of a slot must + convert: a value of `n` bytes that sits `o` bytes from the low-order end + is at `offset` `$wordsize - o - n`. An emitter may write that number as + a literal, which is the simplest form to read and resolve. It may also + write the conversion as an expression, such as + + ```json + { + "$difference": ["$wordsize", { "$sum": [o, { ".length": "$this" }] }] + } + ``` + + which can take `n` from the region's own `length`, keeps the layout's + own numbers visible, needs no arithmetic in the emitter, and does not + depend on a fixed word size. This field's expression must resolve to a non-negative value. It is **not** bounded by the word size: an offset that meets or exceeds @@ -77,3 +100,27 @@ examples: - slot: 0 offset: $wordsize length: 4 + # packed values: an `address` (20 bytes) at the low-order end of slot 2, + # and a `uint32` (4 bytes) just above it, written with literal offsets + - slot: 2 + offset: 12 + length: 20 + - slot: 2 + offset: 8 + length: 4 + # the same two values with the conversion `$wordsize - (o + n)` written as + # an expression that takes `n` from the region's own length + - slot: 2 + offset: + $difference: + - $wordsize + - .length: $this + length: 20 + - slot: 2 + offset: + $difference: + - $wordsize + - $sum: + - 20 + - .length: $this + length: 4